Merge remote-tracking branch 'upstream/main' into feat/profiles-and-alias

# Conflicts:
#	packages/coding-agent/src/cli/args.ts
This commit is contained in:
Ogrodev
2026-06-14 19:10:31 -03:00
498 changed files with 28242 additions and 9623 deletions
+19 -6
View File
@@ -36,6 +36,12 @@ runs:
toolchain: nightly-2026-04-29
components: ${{ inputs.rust_checks == 'true' && 'clippy, rustfmt' || '' }}
targets: ${{ inputs.target }}
- name: Install Linux build prerequisites
if: runner.os == 'Linux'
shell: bash
run: |
sudo apt-get update
sudo apt-get install -y build-essential
- name: Prepend rustup toolchain bin to PATH
shell: bash
run: |
@@ -97,18 +103,25 @@ runs:
- name: Setup sccache
uses: mozilla-actions/sccache-action@v0.0.10
- name: Enable sccache for cargo
# `CARGO_INCREMENTAL=0` is required: sccache silently skips caching
# when incremental is enabled, which would turn the wrapper into a
# no-op. rust-cache also sets this today, but pin it here so sccache
# stays effective if rust-cache is removed or changes defaults. Also
# overrides `profile.dev.incremental = true` for `bun run test:rs`.
# `CARGO_INCREMENTAL=0` is required: sccache silently skips caching when
# incremental is enabled, turning the wrapper into a no-op. The backend
# is conditional: self-hosted omp-kata runners inject a shared S3 (RustFS)
# cache via pod env (SCCACHE_BUCKET/ENDPOINT/REGION + AWS creds) that
# sccache reads from the inherited environment; GitHub-hosted runners
# (macOS, ubuntu-arm) can't reach the private RustFS and keep the GHA
# cache backend.
shell: bash
run: |
{
echo "SCCACHE_GHA_ENABLED=true"
echo "RUSTC_WRAPPER=sccache"
echo "CARGO_INCREMENTAL=0"
} >> "$GITHUB_ENV"
if [ -n "${SCCACHE_BUCKET:-}" ]; then
echo "sccache backend: shared S3 ($SCCACHE_BUCKET @ $SCCACHE_ENDPOINT)"
else
echo "SCCACHE_GHA_ENABLED=true" >> "$GITHUB_ENV"
echo "sccache backend: GitHub Actions cache"
fi
- uses: taiki-e/install-action@v2
if: inputs.target == ''
with:
@@ -0,0 +1,26 @@
name: Setup system deps
description: >-
Install the canvas/native runtime deps CI needs (cairo/pango stack, fd,
ripgrep, imagemagick). No-op on the preloaded omp-kata runner image, which
already ships them; self-heals on a stock runner by installing via apt.
runs:
using: composite
steps:
- name: Install system deps (skip when preloaded)
shell: bash
run: |
# The preloaded omp-kata runner image bakes these in. Detect that
# and skip the apt round-trip; otherwise install the exact same set so
# stock runners (and any future host) still work.
if command -v fd >/dev/null 2>&1 \
&& command -v rg >/dev/null 2>&1 \
&& command -v magick >/dev/null 2>&1 \
&& pkg-config --exists cairo pango 2>/dev/null; then
echo "System deps already present (preloaded runner image); skipping apt."
exit 0
fi
sudo apt-get update
sudo apt-get install -y libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev fd-find ripgrep imagemagick
sudo ln -sf "$(command -v fdfind)" /usr/local/bin/fd
sudo ln -sf /usr/bin/convert /usr/local/bin/magick
+293 -54
View File
@@ -12,13 +12,25 @@ on:
type: boolean
default: false
# Release runs publish a `v*` tag pushed atomically with main HEAD; sharing
# the cheap branch-wide `CI-refs/heads/main` group meant a later main push
# silently cancelled the in-flight release and left the tag without a GitHub
# Release or npm publish (#2564). Detect release runs at workflow-scheduling
# time via the release-script commit subject (`chore: bump version to vX.Y.Z`),
# via `v*` tag-ref dispatches, and via manual dispatches whose tag-on-HEAD
# status is only known after checkout; scope them to a per-sha group with no
# cancellation. Every other event keeps branch-wide cancellation for PR/main churn.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
group: "${{ github.workflow }}-${{ (startsWith(github.event.head_commit.message, 'chore: bump version to ') || startsWith(github.ref, 'refs/tags/v') || github.event_name == 'workflow_dispatch') && format('release-{0}', github.sha) || github.ref }}"
cancel-in-progress: "${{ !(startsWith(github.event.head_commit.message, 'chore: bump version to ') || startsWith(github.ref, 'refs/tags/v') || github.event_name == 'workflow_dispatch') }}"
env:
FORCE_JAVASCRIPT_ACTIONS_TO_NODE24: true
permissions:
contents: read
actions: read
jobs:
# scripts/release.ts pushes the version-bump commit and its `v*` tag
# atomically (`git push --atomic origin refs/heads/main:refs/heads/main
@@ -32,7 +44,7 @@ jobs:
# ref (or from a tagged main HEAD) is also treated as a release.
release_metadata:
name: Resolve release metadata
runs-on: ubuntu-22.04
runs-on: omp-kata
outputs:
is-release: ${{ steps.detect.outputs.is-release }}
release-tag: ${{ steps.detect.outputs.release-tag }}
@@ -75,7 +87,7 @@ jobs:
# native artifacts for this hash. Two independent outputs:
# * `linux-x64-run-id` — set when the linux x64 canary
# (`pi-natives-linux-x64-modern-h<hash>`) is present on a prior main run,
# so `test`/`native_linux_x64` can reuse it.
# so native-dependent TS test jobs and `native_linux_x64` can reuse it.
# * `cross-platform-run-id` — set when ALL cross-platform native artifacts
# also have non-expired artifacts on that same prior run, so
# `native_cross_platform` can skip the cold rebuild on main pushes after
@@ -84,7 +96,7 @@ jobs:
# retention window (see build-native action) is the effective TTL.
native_artifact_lookup:
name: Look up cached native artifacts
runs-on: ubuntu-22.04
runs-on: omp-kata
outputs:
source-hash: ${{ steps.compute.outputs.source-hash }}
linux-x64-run-id: ${{ steps.find.outputs.linux-x64-run-id }}
@@ -175,7 +187,7 @@ jobs:
# Fast lint, type check, and browser bundle build (no Rust, no native build needed)
check:
name: Lint, type check & web build
runs-on: ubuntu-22.04
runs-on: omp-kata
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
@@ -199,7 +211,7 @@ jobs:
name: "Native: Linux x64 (${{ matrix.variant }})"
needs: [release_metadata, native_artifact_lookup]
if: ${{ needs.release_metadata.outputs.is-release == 'true' || needs.native_artifact_lookup.outputs.linux-x64-run-id == '' }}
runs-on: ubuntu-22.04
runs-on: omp-kata
strategy:
fail-fast: false
matrix:
@@ -228,10 +240,10 @@ jobs:
fail-fast: false
matrix:
include:
- { os: ubuntu-22.04, platform: linux, arch: arm64, target: aarch64-unknown-linux-gnu }
- { os: omp-kata, platform: linux, arch: arm64, target: aarch64-unknown-linux-gnu }
- { os: macos-15-intel, platform: darwin, arch: x64, variant: baseline }
- { os: macos-14, platform: darwin, arch: arm64 }
- { os: ubuntu-22.04, platform: win32, arch: x64, target: x86_64-pc-windows-msvc, variant: baseline }
- { os: omp-kata, platform: win32, arch: x64, target: x86_64-pc-windows-msvc, variant: baseline }
runs-on: ${{ matrix.os }}
steps:
- uses: actions/checkout@v4
@@ -244,12 +256,12 @@ jobs:
target: ${{ matrix.target }}
save_cache: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }}
test:
name: Test & smoke (TS)
runs-on: ubuntu-22.04
test_workspace:
name: Test TS workspace fast
runs-on: omp-kata
needs: [native_linux_x64, native_artifact_lookup]
if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }}
timeout-minutes: 30
timeout-minutes: 20
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
@@ -260,12 +272,6 @@ jobs:
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- name: Install system deps
run: |
sudo apt-get update
sudo apt-get install -y libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev fd-find ripgrep imagemagick
sudo ln -sf "$(command -v fdfind)" /usr/local/bin/fd
sudo ln -sf /usr/bin/convert /usr/local/bin/magick
- run: bun install --frozen-lockfile
- name: Resolve Linux x64 native artifact run
id: source
@@ -284,18 +290,242 @@ jobs:
merge-multiple: true
run-id: ${{ steps.source.outputs.artifact-run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: Test workspace (TS)
# `test:ts` sets GITHUB_ACTIONS=0 inline so `bun test` skips its
# per-file `::group::`/`::endgroup::` annotations. Under `--workspaces`
# every line is prefixed with `<pkg> test: `, which breaks GHA's
# column-0 parsing and would leak those markers as literal log spam.
run: bun run test:ts
- name: Test workspace packages and repo scripts (TS)
run: bun run ci:test:ts:workspace
test_coding_agent_singleton:
name: Test coding-agent singleton/global-state (TS)
runs-on: omp-kata
needs: [native_linux_x64, native_artifact_lookup]
if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }}
timeout-minutes: 20
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3"
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- run: bun install --frozen-lockfile
- name: Resolve Linux x64 native artifact run
id: source
shell: bash
run: |
if [ "${{ needs.native_linux_x64.result }}" = "success" ]; then
echo "artifact-run-id=${{ github.run_id }}" >> "$GITHUB_OUTPUT"
else
echo "artifact-run-id=${{ needs.native_artifact_lookup.outputs.linux-x64-run-id }}" >> "$GITHUB_OUTPUT"
fi
- name: Download native addons
uses: actions/download-artifact@v4
with:
pattern: pi-natives-linux-x64-*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
merge-multiple: true
run-id: ${{ steps.source.outputs.artifact-run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: Test coding-agent singleton/global-state bucket
# Keep global Settings/env/fake-timer tests serial; native addon
# artifacts are still available like every other coding-agent bucket.
run: bun run ci:test:coding-agent:singleton
test_ts_native:
name: Test TS native/integration packages
runs-on: omp-kata
needs: [native_linux_x64, native_artifact_lookup]
if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }}
timeout-minutes: 25
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3"
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- uses: ./.github/actions/setup-system-deps
- run: bun install --frozen-lockfile
- name: Resolve Linux x64 native artifact run
id: source
shell: bash
run: |
if [ "${{ needs.native_linux_x64.result }}" = "success" ]; then
echo "artifact-run-id=${{ github.run_id }}" >> "$GITHUB_OUTPUT"
else
echo "artifact-run-id=${{ needs.native_artifact_lookup.outputs.linux-x64-run-id }}" >> "$GITHUB_OUTPUT"
fi
- name: Download native addons
uses: actions/download-artifact@v4
with:
pattern: pi-natives-linux-x64-*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
merge-multiple: true
run-id: ${{ steps.source.outputs.artifact-run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: Test native/TUI/browser-ish packages (TS)
run: bun run ci:test:ts:native
test_coding_agent_ui:
name: Test coding-agent UI/TUI (TS)
runs-on: omp-kata
needs: [native_linux_x64, native_artifact_lookup]
if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }}
timeout-minutes: 25
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3"
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- uses: ./.github/actions/setup-system-deps
- run: bun install --frozen-lockfile
- name: Resolve Linux x64 native artifact run
id: source
shell: bash
run: |
if [ "${{ needs.native_linux_x64.result }}" = "success" ]; then
echo "artifact-run-id=${{ github.run_id }}" >> "$GITHUB_OUTPUT"
else
echo "artifact-run-id=${{ needs.native_artifact_lookup.outputs.linux-x64-run-id }}" >> "$GITHUB_OUTPUT"
fi
- name: Download native addons
uses: actions/download-artifact@v4
with:
pattern: pi-natives-linux-x64-*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
merge-multiple: true
run-id: ${{ steps.source.outputs.artifact-run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: Test coding-agent UI/TUI bucket
run: bun run ci:test:coding-agent:ui
test_coding_agent_runtime:
name: Test coding-agent runtime/session (TS)
runs-on: omp-kata
needs: [native_linux_x64, native_artifact_lookup]
if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }}
timeout-minutes: 25
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3"
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- run: bun install --frozen-lockfile
- name: Resolve Linux x64 native artifact run
id: source
shell: bash
run: |
if [ "${{ needs.native_linux_x64.result }}" = "success" ]; then
echo "artifact-run-id=${{ github.run_id }}" >> "$GITHUB_OUTPUT"
else
echo "artifact-run-id=${{ needs.native_artifact_lookup.outputs.linux-x64-run-id }}" >> "$GITHUB_OUTPUT"
fi
- name: Download native addons
uses: actions/download-artifact@v4
with:
pattern: pi-natives-linux-x64-*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
merge-multiple: true
run-id: ${{ steps.source.outputs.artifact-run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: Test coding-agent runtime bucket
# Runtime/session tests import native-backed barrels too; keep this
# separate for concurrency, not as a native-free guardrail.
run: bun run ci:test:coding-agent:runtime
test_coding_agent_native:
name: Test coding-agent native/unit (TS)
runs-on: omp-kata
needs: [native_linux_x64, native_artifact_lookup]
if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }}
timeout-minutes: 25
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3"
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- uses: ./.github/actions/setup-system-deps
- run: bun install --frozen-lockfile
- name: Resolve Linux x64 native artifact run
id: source
shell: bash
run: |
if [ "${{ needs.native_linux_x64.result }}" = "success" ]; then
echo "artifact-run-id=${{ github.run_id }}" >> "$GITHUB_OUTPUT"
else
echo "artifact-run-id=${{ needs.native_artifact_lookup.outputs.linux-x64-run-id }}" >> "$GITHUB_OUTPUT"
fi
- name: Download native addons
uses: actions/download-artifact@v4
with:
pattern: pi-natives-linux-x64-*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
merge-multiple: true
run-id: ${{ steps.source.outputs.artifact-run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: Test coding-agent native/unit bucket
run: bun run ci:test:coding-agent:native
test_smoke:
name: Test CLI smoke (TS)
runs-on: omp-kata
needs: [native_linux_x64, native_artifact_lookup]
if: ${{ !cancelled() && needs.native_linux_x64.result != 'failure' }}
timeout-minutes: 15
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3"
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- uses: ./.github/actions/setup-system-deps
- run: bun install --frozen-lockfile
- name: Resolve Linux x64 native artifact run
id: source
shell: bash
run: |
if [ "${{ needs.native_linux_x64.result }}" = "success" ]; then
echo "artifact-run-id=${{ github.run_id }}" >> "$GITHUB_OUTPUT"
else
echo "artifact-run-id=${{ needs.native_artifact_lookup.outputs.linux-x64-run-id }}" >> "$GITHUB_OUTPUT"
fi
- name: Download native addons
uses: actions/download-artifact@v4
with:
pattern: pi-natives-linux-x64-*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
merge-multiple: true
run-id: ${{ steps.source.outputs.artifact-run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: CLI smoke test
run: bun run ci:test:smoke
install_methods:
name: Install method smoke tests
runs-on: ubuntu-22.04
runs-on: omp-kata
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
@@ -316,24 +546,27 @@ jobs:
- name: Setup sccache
uses: mozilla-actions/sccache-action@v0.0.10
- name: Enable sccache for cargo
# Conditional backend: self-hosted omp-kata injects a shared S3
# (RustFS) sccache via pod env; GitHub-hosted runners keep the GHA
# cache. CARGO_INCREMENTAL=0 keeps sccache from silently no-oping.
shell: bash
run: |
{
echo "SCCACHE_GHA_ENABLED=true"
echo "RUSTC_WRAPPER=sccache"
echo "CARGO_INCREMENTAL=0"
} >> "$GITHUB_ENV"
if [ -n "${SCCACHE_BUCKET:-}" ]; then
echo "sccache backend: shared S3 ($SCCACHE_BUCKET @ $SCCACHE_ENDPOINT)"
else
echo "SCCACHE_GHA_ENABLED=true" >> "$GITHUB_ENV"
echo "sccache backend: GitHub Actions cache"
fi
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- name: Install system deps
run: |
sudo apt-get update
sudo apt-get install -y libcairo2-dev libpango1.0-dev libjpeg-dev libgif-dev librsvg2-dev fd-find ripgrep imagemagick
sudo ln -sf "$(command -v fdfind)" /usr/local/bin/fd
sudo ln -sf /usr/bin/convert /usr/local/bin/magick
- uses: ./.github/actions/setup-system-deps
- run: bun install --frozen-lockfile
- name: Install method smoke tests
run: bun run ci:test:install-methods
@@ -342,9 +575,15 @@ jobs:
name: "Release binary: ${{ matrix.target_id }}"
if: ${{ needs.release_metadata.outputs.is-release == 'true' && !cancelled() &&
needs.native_linux_x64.result == 'success' && needs.native_cross_platform.result ==
'success' && needs.test.result == 'success' && needs.check.result ==
'success' && needs.install_methods.result == 'success' }}
needs: [release_metadata, check, native_linux_x64, native_cross_platform, test, install_methods, native_artifact_lookup]
'success' && needs.test_workspace.result == 'success' &&
needs.test_coding_agent_singleton.result == 'success' &&
needs.test_ts_native.result == 'success' &&
needs.test_coding_agent_ui.result == 'success' &&
needs.test_coding_agent_runtime.result == 'success' &&
needs.test_coding_agent_native.result == 'success' &&
needs.test_smoke.result == 'success' && needs.check.result == 'success' &&
needs.install_methods.result == 'success' }}
needs: [release_metadata, check, native_linux_x64, native_cross_platform, test_workspace, test_coding_agent_singleton, test_ts_native, test_coding_agent_ui, test_coding_agent_runtime, test_coding_agent_native, test_smoke, install_methods, native_artifact_lookup]
strategy:
fail-fast: false
matrix:
@@ -391,11 +630,11 @@ jobs:
env:
MACOS_SIGNING: ${{ secrets.APPLE_CERTIFICATE_P12 != '' && secrets.APPLE_CERTIFICATE_PASSWORD != '' && secrets.APPLE_API_KEY_ID != '' && secrets.APPLE_API_ISSUER_ID != '' && secrets.APPLE_API_KEY != '' }}
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
bun-version: "1.3"
- uses: actions/setup-node@v4
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: "24"
registry-url: "https://registry.npmjs.org"
@@ -404,13 +643,13 @@ jobs:
if: ${{ !inputs.skip_npm }}
run: npm install -g npm@latest
- name: Cache bun dependencies
uses: actions/cache@v4
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- run: bun install --frozen-lockfile
- name: Download native addon(s)
uses: actions/download-artifact@v4
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
pattern: pi-natives-${{ matrix.platform }}-${{ matrix.arch }}*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
@@ -451,7 +690,7 @@ jobs:
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
run: bun run ci:release:publish-native-leaf ${{ matrix.target_id }}
- name: Upload release binary artifact
uses: actions/upload-artifact@v4
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
with:
name: omp-binary-${{ matrix.target_id }}
path: ${{ matrix.binary_path }}
@@ -465,20 +704,20 @@ jobs:
permissions:
contents: write
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
bun-version: "1.3"
- name: Generate release notes from CHANGELOGs
run: bun scripts/ci-release-notes.ts ${{ needs.release_metadata.outputs.release-tag }}
- name: Download release binaries
uses: actions/download-artifact@v4
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
pattern: omp-binary-*
path: packages/coding-agent/binaries
merge-multiple: true
- name: Create GitHub Release
uses: softprops/action-gh-release@v2
uses: softprops/action-gh-release@3bb12739c298aeb8a4eeaf626c5b8d85266b0e65 # v2.6.2
with:
tag_name: ${{ needs.release_metadata.outputs.release-tag }}
files: |
@@ -537,11 +776,11 @@ jobs:
id-token: write
contents: read
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
with:
bun-version: "1.3"
- uses: actions/setup-node@v4
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: "24"
registry-url: "https://registry.npmjs.org"
@@ -549,7 +788,7 @@ jobs:
- name: Ensure npm supports trusted publishing
run: npm install -g npm@latest
- name: Cache bun dependencies
uses: actions/cache@v4
uses: actions/cache@0057852bfaa89a56745cba8c7296529d2fc39830 # v4.3.0
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
@@ -560,7 +799,7 @@ jobs:
# Release runs always rebuild natives in this same run, so the
# default run-id resolves the artifacts.
- name: Download native addons
uses: actions/download-artifact@v4
uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
with:
pattern: pi-natives-linux-x64-*-h${{ needs.native_artifact_lookup.outputs.source-hash }}
path: packages/natives/native
@@ -587,15 +826,15 @@ jobs:
env:
HAS_TAP_KEY: ${{ secrets.HOMEBREW_TAP_DEPLOY_KEY != '' }}
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
if: env.HAS_TAP_KEY == 'true'
- uses: oven-sh/setup-bun@v2
- uses: oven-sh/setup-bun@0c5077e51419868618aeaa5fe8019c62421857d6 # v2.2.0
if: env.HAS_TAP_KEY == 'true'
with:
bun-version: "1.3"
- name: Check out the Homebrew tap
if: env.HAS_TAP_KEY == 'true'
uses: actions/checkout@v4
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
with:
repository: can1357/homebrew-tap
ssh-key: ${{ secrets.HOMEBREW_TAP_DEPLOY_KEY }}
+1
View File
@@ -237,6 +237,7 @@ Location: `packages/*/CHANGELOG.md` (per package).
**Rules:**
- New entries always go under `## [Unreleased]`.
- Never modify already-released sections (e.g., `## [0.12.2]`) — they are immutable.
- Don't flag changelog section order or formatting in reviews or PRs — `bun run release` runs `fix-changelogs` which normalizes everything automatically.
**Attribution:**
- Internal (from issues): `Fixed foo bar ([#123](https://github.com/can1357/oh-my-pi/issues/123))`.
Generated
+6 -6
View File
@@ -1825,9 +1825,9 @@ dependencies = [
[[package]]
name = "napi"
version = "3.9.1"
version = "3.9.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ad513ff22558f1830b595ea6eb4091da48145d09a222ce157e781896f78be0b9"
checksum = "26d3c7dd60231116a47854321c9ac8df6f13435d11aa3a59d8533a76e07a3730"
dependencies = [
"bitflags 2.13.0",
"ctor",
@@ -2330,7 +2330,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "15.12.5"
version = "15.13.0"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -2398,7 +2398,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "15.12.5"
version = "15.13.0"
dependencies = [
"async-trait",
"libc",
@@ -2410,7 +2410,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "15.12.5"
version = "15.13.0"
dependencies = [
"anyhow",
"arboard",
@@ -2458,7 +2458,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "15.12.5"
version = "15.13.0"
dependencies = [
"anyhow",
"brush-builtins",
+1 -1
View File
@@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"]
resolver = "3"
[workspace.package]
version = "15.12.5"
version = "15.13.0"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
+92 -38
View File
@@ -4,6 +4,11 @@
"workspaces": {
"": {
"name": "omp-monorepo",
"dependencies": {
"sherpa-onnx": "1.12.37",
"sherpa-onnx-darwin-arm64": "1.12.37",
"sherpa-onnx-node": "1.12.37",
},
"devDependencies": {
"@biomejs/biome": "catalog:",
"@types/bun": "catalog:",
@@ -15,7 +20,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "15.12.5",
"version": "15.13.0",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -32,7 +37,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "15.12.5",
"version": "15.13.0",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -46,7 +51,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "15.12.5",
"version": "15.13.0",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -59,7 +64,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "15.12.5",
"version": "15.13.0",
"bin": {
"omp": "src/cli.ts",
},
@@ -104,6 +109,7 @@
},
"optionalDependencies": {
"@huggingface/transformers": "catalog:",
"sherpa-onnx-node": "1.13.2",
},
},
"packages/collab-web": {
@@ -124,7 +130,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "15.12.5",
"version": "15.13.0",
"dependencies": {
"diff": "catalog:",
"lru-cache": "catalog:",
@@ -135,7 +141,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "15.12.5",
"version": "15.13.0",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -161,7 +167,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "15.12.5",
"version": "15.13.0",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -169,7 +175,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "15.12.5",
"version": "15.13.0",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
@@ -181,7 +187,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "15.12.5",
"version": "15.13.0",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -207,7 +213,7 @@
},
"packages/swarm-extension": {
"name": "@oh-my-pi/swarm-extension",
"version": "15.12.5",
"version": "15.13.0",
"bin": {
"omp-swarm": "src/cli.ts",
},
@@ -223,7 +229,7 @@
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "15.12.5",
"version": "15.13.0",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -264,7 +270,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "15.12.5",
"version": "15.13.0",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"beautiful-mermaid": "catalog:",
@@ -278,7 +284,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "15.12.5",
"version": "15.13.0",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -314,18 +320,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "15.12.5",
"@oh-my-pi/omp-stats": "15.12.5",
"@oh-my-pi/pi-agent-core": "15.12.5",
"@oh-my-pi/pi-ai": "15.12.5",
"@oh-my-pi/pi-catalog": "15.12.5",
"@oh-my-pi/pi-coding-agent": "15.12.5",
"@oh-my-pi/pi-mnemopi": "15.12.5",
"@oh-my-pi/pi-natives": "15.12.5",
"@oh-my-pi/pi-tui": "15.12.5",
"@oh-my-pi/pi-utils": "15.12.5",
"@oh-my-pi/pi-wire": "15.12.5",
"@oh-my-pi/snapcompact": "15.12.5",
"@oh-my-pi/hashline": "15.13.0",
"@oh-my-pi/omp-stats": "15.13.0",
"@oh-my-pi/pi-agent-core": "15.13.0",
"@oh-my-pi/pi-ai": "15.13.0",
"@oh-my-pi/pi-catalog": "15.13.0",
"@oh-my-pi/pi-coding-agent": "15.13.0",
"@oh-my-pi/pi-mnemopi": "15.13.0",
"@oh-my-pi/pi-natives": "15.13.0",
"@oh-my-pi/pi-tui": "15.13.0",
"@oh-my-pi/pi-utils": "15.13.0",
"@oh-my-pi/pi-wire": "15.13.0",
"@oh-my-pi/snapcompact": "15.13.0",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -453,7 +459,7 @@
"@emnapi/core": ["@emnapi/core@1.10.0", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.1", "tslib": "^2.4.0" } }, "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw=="],
"@emnapi/runtime": ["@emnapi/runtime@1.10.0", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA=="],
"@emnapi/runtime": ["@emnapi/runtime@1.11.0", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-55coeOFKHv1ywEcUXJtWU5f+Jr/W5tZDvZig8DLKSwUN1JpROQ4rk/SNOQiFWmaR/VKF4zuFyW1B8JduOSv6Pg=="],
"@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.1", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w=="],
@@ -735,9 +741,9 @@
"@opentelemetry/api-logs": ["@opentelemetry/api-logs@0.218.0", "", { "dependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-fmEWp5kXlGEc3i/lR698Hz41DfGyN4Tbe4g7L1AxSc7fF8Xeh/FQ9Quqpa9dVA413Q1Ad43QOLzU4JoXgbFPWw=="],
"@opentelemetry/context-async-hooks": ["@opentelemetry/context-async-hooks@2.7.1", "", { "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-OPFBYuXEn1E4ja3Y6eeA7O+ZnLBNcXTV5Cgsn1VaqBZ6hC5FnpZPLBNme1LJY8ZtF4aOujPKFoeWN4ik487KuQ=="],
"@opentelemetry/context-async-hooks": ["@opentelemetry/context-async-hooks@2.8.0", "", { "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-/3FIraneMcng67SUJCxvyInk/oxzwsxyadufk0wwfOBLf5wqtAGX4MoQASwSbndBPeARzBryUM9Azr5kHIdWLw=="],
"@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="],
"@opentelemetry/core": ["@opentelemetry/core@2.8.0", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-hd1Lfh8p545nNz+jq1Ejfz+Mn1hyLuxYn1YzTfFNrxr8urEWMNQLPf1Th8kjOH+HxwawCrtgBp8JpBUR4ZSgww=="],
"@opentelemetry/exporter-trace-otlp-proto": ["@opentelemetry/exporter-trace-otlp-proto@0.218.0", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/otlp-exporter-base": "0.218.0", "@opentelemetry/otlp-transformer": "0.218.0", "@opentelemetry/resources": "2.7.1", "@opentelemetry/sdk-trace-base": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-r1Msf8SNLRmwh9J6XQ5uh82D7CdDWMNHnPB7LAVHjzut0TkSeKc5KcIvr4SvHvfk/xwN5gxC+VLKQ1k0o8PSPw=="],
@@ -745,15 +751,15 @@
"@opentelemetry/otlp-transformer": ["@opentelemetry/otlp-transformer@0.218.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.218.0", "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/sdk-logs": "0.218.0", "@opentelemetry/sdk-metrics": "2.7.1", "@opentelemetry/sdk-trace-base": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": "^1.3.0" } }, "sha512-CFaKH87WAzjuJ4awowTTLzUvMfaRfiOFG5+qm5S5ncyalRtN4ecQ+YmuANJSCrVPuvZFEkUgKhBPBndxi3rHsQ=="],
"@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="],
"@opentelemetry/resources": ["@opentelemetry/resources@2.8.0", "", { "dependencies": { "@opentelemetry/core": "2.8.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-qmXQ27ilDbUK/vGMqwL8D4/rhn76C+sherM4wTbjlfknR8Nvfc/hCxjRJPhkzZzUsPiNg16SA31NxMabwttRjg=="],
"@opentelemetry/sdk-logs": ["@opentelemetry/sdk-logs@0.218.0", "", { "dependencies": { "@opentelemetry/api-logs": "0.218.0", "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.4.0 <1.10.0" } }, "sha512-QvnNdugatFTVCJXH0Mcu7GOOJSylA9j127kIezOE4YwTI4YbowRons2K4WZTv5FMS8T4q9P0NdaRHdkSmeAIag=="],
"@opentelemetry/sdk-metrics": ["@opentelemetry/sdk-metrics@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": ">=1.9.0 <1.10.0" } }, "sha512-MpDJdkiFDs3Pm1RHO3KByuZbuBdJEXEAkiC0+yJdsZGVCdf1RpHR6n+LHDcS7ffmfrt5kVCzJSCfm4z2C7v0uQ=="],
"@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-NAYIlsF8MPUsKqJMiDQJTMPOmlbawC1Iz/omMLygZ1C9am8fTKYjTaI+OZM+WTY3t3Glo0wnOg/6/pac6RGPPw=="],
"@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.8.0", "", { "dependencies": { "@opentelemetry/core": "2.8.0", "@opentelemetry/resources": "2.8.0", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-mhU4jp+vW0mGbFRd+GeXHvmfA4aDqWjBjLC3pE5XMpLs0IE2ryYb019Ts2AQrOq67gaTF25D91+fgvEHDZEnuQ=="],
"@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.7.1", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.7.1", "@opentelemetry/core": "2.7.1", "@opentelemetry/sdk-trace-base": "2.7.1" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-pCpQxU68lV+I9s9svqMyVu5iHdDDUnqUpSxqwyCU8A9ejEsSnMPCbearwsUO4yk08ZJzAIUCFuReMdVQvHrdvg=="],
"@opentelemetry/sdk-trace-node": ["@opentelemetry/sdk-trace-node@2.8.0", "", { "dependencies": { "@opentelemetry/context-async-hooks": "2.8.0", "@opentelemetry/core": "2.8.0", "@opentelemetry/sdk-trace-base": "2.8.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-nZt9OGufioAc3AfoLTqA9bsAeaMJAictYDdI2VcNQ+PmT+3rfKjAZDZvgPfd8VPX0O5Bw1hdQF6kDK8VSpZiWg=="],
"@opentelemetry/semantic-conventions": ["@opentelemetry/semantic-conventions@1.41.1", "", {}, "sha512-/UhIkaZgPutTFmQ7RnIJGgDXZmtEJ7Dvi86xNTFWcnRxVRNk/aotsqDJYeEvDP+FSMB2SdW+pQzNMcWP0rwuNA=="],
@@ -861,7 +867,7 @@
"@types/bun": ["@types/bun@1.3.14", "", { "dependencies": { "bun-types": "1.3.14" } }, "sha512-h1hFqFVcvAvD9j9K7ZW7vd82aSA+rTdznZa+5bwvCwqSB1jmmfLcbIWhOLx1/+boy/xmjgCs/OMUL8hRJSmnPw=="],
"@types/node": ["@types/node@25.9.2", "", { "dependencies": { "undici-types": ">=7.24.0 <7.24.7" } }, "sha512-G05zqtJhcDLb8uslf5EjCxXg9G1KQxiV8OS0R26IC//Eoyitzqe8z37I7cqvnZlrlSfgocQRfSn/AHBZJJFyGw=="],
"@types/node": ["@types/node@25.9.3", "", { "dependencies": { "undici-types": ">=7.24.0 <7.24.7" } }, "sha512-603BddQMv3pUcr4U2dhujk83N2tTDVr/34wII2B6bJy6g+8WD6yUb11jszNs0gdi4PesVWl7ABt8nYMVpnLUcg=="],
"@types/react": ["@types/react@19.2.17", "", { "dependencies": { "csstype": "^3.2.2" } }, "sha512-MXfmqaVPEVgkBT/aY0aGCkRWWtByiYQXo3xdQ8r5RzuFrPiRn8Gar2tQdXSUQ2GKV3bkXckek89V8wQBY2Q/Aw=="],
@@ -927,7 +933,7 @@
"bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="],
"caniuse-lite": ["caniuse-lite@1.0.30001797", "", {}, "sha512-l8xKG+gwAIExZGl9FrF7KUwuOmk6wbEPC9Xoy/RtnWv1XG0Q4LFlagaLpUv3Kiza3W/wm27zy0yWJEieYKAP6w=="],
"caniuse-lite": ["caniuse-lite@1.0.30001799", "", {}, "sha512-hG1bReV+OUU+MOqK4t/ZWI0tZOyz3rqS9XuhOUz1cIcbwBKjOyJEJuw9ER5JuNyqxNk8u/JUVbGibBOL1yrjFw=="],
"chalk": ["chalk@5.6.2", "", {}, "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA=="],
@@ -1015,7 +1021,7 @@
"enabled": ["enabled@2.0.0", "", {}, "sha512-AKrN98kuwOzMIdAizXGI86UFBoo26CL21UM763y1h/GMSJ4/OHU9k2YlsmBpyScFo/wbLzWQJBMCW4+IO3/+OQ=="],
"enhanced-resolve": ["enhanced-resolve@5.23.0", "", { "dependencies": { "graceful-fs": "^4.2.4", "tapable": "^2.3.3" } }, "sha512-yJN/BOOLxcOW2aQgeif9mSnaUB8KtvmMMp56oA1kx1CRfBKbhZm2pJ+NBY+3eOboHxix8lfjWpHE0Ei5U8RbSA=="],
"enhanced-resolve": ["enhanced-resolve@5.24.0", "", { "dependencies": { "graceful-fs": "^4.2.4", "tapable": "^2.3.3" } }, "sha512-SkE2t82KlkkxQRVMVLAGKxLfORGQfrkx5dkj+vlgXRVNEdPc4eZcR+J/Fvj8C+yKSFH5L0q3NFlyufOVQnCcYQ=="],
"entities": ["entities@7.0.1", "", {}, "sha512-TWrgLOFUQTH994YUyl1yT4uyavY5nNB5muff+RtWaqNVCAK408b5ZnnbNAUEWLTCpum9w6arT70i1XdQ4UeOPA=="],
@@ -1315,6 +1321,22 @@
"sharp": ["sharp@0.34.5", "", { "dependencies": { "@img/colour": "^1.0.0", "detect-libc": "^2.1.2", "semver": "^7.7.3" }, "optionalDependencies": { "@img/sharp-darwin-arm64": "0.34.5", "@img/sharp-darwin-x64": "0.34.5", "@img/sharp-libvips-darwin-arm64": "1.2.4", "@img/sharp-libvips-darwin-x64": "1.2.4", "@img/sharp-libvips-linux-arm": "1.2.4", "@img/sharp-libvips-linux-arm64": "1.2.4", "@img/sharp-libvips-linux-ppc64": "1.2.4", "@img/sharp-libvips-linux-riscv64": "1.2.4", "@img/sharp-libvips-linux-s390x": "1.2.4", "@img/sharp-libvips-linux-x64": "1.2.4", "@img/sharp-libvips-linuxmusl-arm64": "1.2.4", "@img/sharp-libvips-linuxmusl-x64": "1.2.4", "@img/sharp-linux-arm": "0.34.5", "@img/sharp-linux-arm64": "0.34.5", "@img/sharp-linux-ppc64": "0.34.5", "@img/sharp-linux-riscv64": "0.34.5", "@img/sharp-linux-s390x": "0.34.5", "@img/sharp-linux-x64": "0.34.5", "@img/sharp-linuxmusl-arm64": "0.34.5", "@img/sharp-linuxmusl-x64": "0.34.5", "@img/sharp-wasm32": "0.34.5", "@img/sharp-win32-arm64": "0.34.5", "@img/sharp-win32-ia32": "0.34.5", "@img/sharp-win32-x64": "0.34.5" } }, "sha512-Ou9I5Ft9WNcCbXrU9cMgPBcCK8LiwLqcbywW3t4oDV37n1pzpuNLsYiAV8eODnjbtQlSDwZ2cUEeQz4E54Hltg=="],
"sherpa-onnx": ["sherpa-onnx@1.12.37", "", {}, "sha512-3luwSdHwR8BtJiiFwqHfb15FE2FX0KsN4aOBbfq9Ma23r3w9C3bprFc/WBusXk56nUbzcEN5YczN7t9w1JwdtQ=="],
"sherpa-onnx-darwin-arm64": ["sherpa-onnx-darwin-arm64@1.12.37", "", { "os": "darwin", "cpu": "arm64" }, "sha512-zpqbH+2TI6dvg7mxGm30Mnv17aJL3ZfRGshMiBK85dHBfhzqMbKUHNXCzsHhiKBQTzti6JqG1YoRbYrJwvdUjA=="],
"sherpa-onnx-darwin-x64": ["sherpa-onnx-darwin-x64@1.13.2", "", { "os": "darwin", "cpu": "x64" }, "sha512-VyiTiaU/QmBKh5ymEVefc86kMONyJxHaIKpAq0Cf60v/PGEfyfiFaWGGEJyysf8R8PXECB2YLv2wtKAomfcfcw=="],
"sherpa-onnx-linux-arm64": ["sherpa-onnx-linux-arm64@1.13.2", "", { "os": "linux", "cpu": "arm64" }, "sha512-IJNyd6ORcMpy1oR2ZXGNidRiPuIh5KWoOYSIOhW9UQ/BUIY43P6fq8Y3XcFj1xNjBtwke7keMW9DA7ZPYxlYoQ=="],
"sherpa-onnx-linux-x64": ["sherpa-onnx-linux-x64@1.13.2", "", { "os": "linux", "cpu": "x64" }, "sha512-CI2pTKgbOTOpAbm6cSwsFzJZ9qD+xcpEKPfmMCMI7KjaEePnTfky+FYxVBvllbYNk9DQznuTT0Ob9XsBhwtE/Q=="],
"sherpa-onnx-node": ["sherpa-onnx-node@1.12.37", "", { "optionalDependencies": { "sherpa-onnx-darwin-arm64": "^1.12.37", "sherpa-onnx-darwin-x64": "^1.12.37", "sherpa-onnx-linux-arm64": "^1.12.37", "sherpa-onnx-linux-x64": "^1.12.37", "sherpa-onnx-win-ia32": "^1.12.37", "sherpa-onnx-win-x64": "^1.12.37" } }, "sha512-SpblPUl/ODliBk4WzKLRa0VPyc30I3HU9U/qRUmhya7eTNLkJE8sQOmj2kvRL8k2cWrHES2pyvRwwq1jT/MPpw=="],
"sherpa-onnx-win-ia32": ["sherpa-onnx-win-ia32@1.13.2", "", { "os": "win32", "cpu": "ia32" }, "sha512-PJxFuZB6VcwxscP9whLdxBMhHWZ88Ax35LeKpAZWXIvFjIviX1yPJB1+ndhXvF9Nh1L8gPgO9MUBfC1BMKqzvw=="],
"sherpa-onnx-win-x64": ["sherpa-onnx-win-x64@1.13.2", "", { "os": "win32", "cpu": "x64" }, "sha512-D11eEIW4LZLK6Q+yPGmpJ+45gV8EqiwcPNZiE/RoveBsasZCBtl3YXXqvH2APyHafhDk+L8pD/dDR7HdH3GZJw=="],
"signal-exit": ["signal-exit@4.1.0", "", {}, "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw=="],
"slice-ansi": ["slice-ansi@8.0.0", "", { "dependencies": { "ansi-styles": "^6.2.3", "is-fullwidth-code-point": "^5.1.0" } }, "sha512-stxByr12oeeOyY2BlviTNQlYV5xOj47GirPr4yA1hE9JCtxfQN0+tVbkxwCtYDQWhEKWFHsEK48ORg5jrouCAg=="],
@@ -1335,7 +1357,7 @@
"string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="],
"string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="],
"string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="],
"strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="],
@@ -1439,11 +1461,37 @@
"@isaacs/fs-minipass/minipass": ["minipass@7.1.3", "", {}, "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A=="],
"@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.10.0", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.1", "tslib": "^2.4.0" }, "bundled": true }, "sha512-yq6OkJ4p82CAfPl0u9mQebQHKPJkY7WrIuk205cTYnYe+k2Z8YBh11FrbRG/H6ihirqcacOgl2BIO8oyMQLeXw=="],
"@oh-my-pi/pi-coding-agent/sherpa-onnx-node": ["sherpa-onnx-node@1.13.2", "", { "optionalDependencies": { "sherpa-onnx-darwin-arm64": "^1.13.2", "sherpa-onnx-darwin-x64": "^1.13.2", "sherpa-onnx-linux-arm64": "^1.13.2", "sherpa-onnx-linux-x64": "^1.13.2", "sherpa-onnx-win-ia32": "^1.13.2", "sherpa-onnx-win-x64": "^1.13.2" } }, "sha512-uIH6SA5Or4pb8HlCYWB3K54XkMtzdef4/tkw1amtIf8GB1tt6hQLpur9p2jSFNfTYRyzZ8XrXofxefXQ0A7EUA=="],
"@tailwindcss/oxide-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.10.0", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA=="],
"@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="],
"@tailwindcss/oxide-wasm32-wasi/@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.1", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-uTII7OYF+/Mes/MrcIOYp5yOtSMLBWSIoLPpcgwipoiKbli6k322tcoFsxoIIxPDqW01SQGAgko4EzZi2BNv2w=="],
"@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="],
"@opentelemetry/exporter-trace-otlp-proto/@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-NAYIlsF8MPUsKqJMiDQJTMPOmlbawC1Iz/omMLygZ1C9am8fTKYjTaI+OZM+WTY3t3Glo0wnOg/6/pac6RGPPw=="],
"@opentelemetry/otlp-exporter-base/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="],
"@opentelemetry/otlp-transformer/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="],
"@opentelemetry/otlp-transformer/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="],
"@opentelemetry/otlp-transformer/@opentelemetry/sdk-trace-base": ["@opentelemetry/sdk-trace-base@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/resources": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-NAYIlsF8MPUsKqJMiDQJTMPOmlbawC1Iz/omMLygZ1C9am8fTKYjTaI+OZM+WTY3t3Glo0wnOg/6/pac6RGPPw=="],
"@opentelemetry/sdk-logs/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="],
"@opentelemetry/sdk-logs/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="],
"@opentelemetry/sdk-metrics/@opentelemetry/core": ["@opentelemetry/core@2.7.1", "", { "dependencies": { "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.0.0 <1.10.0" } }, "sha512-QAqIj32AtK6+pEVNG7EOVxHdE06RP+FM5qpiEJ4RtDcFIqKUZHYhl7/7UY5efhwmwNAg7j8QbJVBLxMerc0+gw=="],
"@opentelemetry/sdk-metrics/@opentelemetry/resources": ["@opentelemetry/resources@2.7.1", "", { "dependencies": { "@opentelemetry/core": "2.7.1", "@opentelemetry/semantic-conventions": "^1.29.0" }, "peerDependencies": { "@opentelemetry/api": ">=1.3.0 <1.10.0" } }, "sha512-DeT6KKolmC4e/dRQvMQ/RwlnzhaqeiFOXY5ngoOPJ07GgVVKxZOg9EcrNZb5aTzUn+iCrJldAgOfQm1O/QfPAQ=="],
"@rolldown/binding-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.10.0", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-ewvYlk86xUoGI0zQRNq/mC+16R1QeDlKQy21Ki3oSYXNgLb45GV1P6A0M+/s6nyCuNDqe5VpaY84BzXGwVbwFA=="],
"@tailwindcss/oxide-wasm32-wasi/@emnapi/core": ["@emnapi/core@1.11.0", "", { "dependencies": { "@emnapi/wasi-threads": "1.2.2", "tslib": "^2.4.0" }, "bundled": true }, "sha512-l9Oo58x0HOP5znGzVhYW9U3e5wVuA4LAZU2AGezTmkhO1CgQRFDhDg4nneHsu/t3WniXg9QrG2nIXL/ZS8ln8Q=="],
"@tailwindcss/oxide-wasm32-wasi/@emnapi/runtime": ["@emnapi/runtime@1.11.0", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-55coeOFKHv1ywEcUXJtWU5f+Jr/W5tZDvZig8DLKSwUN1JpROQ4rk/SNOQiFWmaR/VKF4zuFyW1B8JduOSv6Pg=="],
"@tailwindcss/oxide-wasm32-wasi/@emnapi/wasi-threads": ["@emnapi/wasi-threads@1.2.2", "", { "dependencies": { "tslib": "^2.4.0" }, "bundled": true }, "sha512-c95qOXkHdydNKhscBTebqEC1CVAZpyqOfVfBzQ1qgzyl3gfeldUjIggDbIZgDKsHLgnsM+igH7TJ/eAasaVuMA=="],
"@tailwindcss/oxide-wasm32-wasi/@napi-rs/wasm-runtime": ["@napi-rs/wasm-runtime@1.1.5", "", { "dependencies": { "@tybys/wasm-util": "^0.10.2" }, "peerDependencies": { "@emnapi/core": "^1.7.1", "@emnapi/runtime": "^1.7.1" }, "bundled": true }, "sha512-AWPoBRJ9tsnVhor4sjO7rkni+7p+2IAEFj6cx06UgP10jkQHqay/36uRV/bFkgrh18D9vb4cr8Q0Pthskgzy+Q=="],
@@ -1489,6 +1537,8 @@
"string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
"string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="],
"wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="],
"xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="],
@@ -1499,6 +1549,8 @@
"@huggingface/transformers/onnxruntime-node/onnxruntime-common": ["onnxruntime-common@1.24.3", "", {}, "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA=="],
"@oh-my-pi/pi-coding-agent/sherpa-onnx-node/sherpa-onnx-darwin-arm64": ["sherpa-onnx-darwin-arm64@1.13.2", "", { "os": "darwin", "cpu": "arm64" }, "sha512-sakY1+WH/Va/vhzwlhIKXaMr0ioKsJ55w785QH+VDFyUqq1+2InbmB3BgX5gyYKeuW40R0U0IW5c7wnMArhpdw=="],
"cliui/strip-ansi/ansi-regex": ["ansi-regex@5.0.1", "", {}, "sha512-quJQXlTSUGL2LH9SUXo8VwsY4soanhgo6LNSm84E1LBcE8s3O0wpdiRzyR9z/ZZJMlMWv37qOOb9pdJlMUEKFQ=="],
"cliui/wrap-ansi/ansi-styles": ["ansi-styles@4.3.0", "", { "dependencies": { "color-convert": "^2.0.1" } }, "sha512-zbB9rCJAT1rbjiVDb2hqKFHNYLxgtk8NURxZ3IZwD3F6NtxbXZQCnnSi1Lkx+IDohdPlFp222wVALIheZJQSEg=="],
@@ -1509,6 +1561,8 @@
"fastembed/onnxruntime-node/tar": ["tar@7.5.16", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w=="],
"jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="],
"log-update/slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="],
"log-update/wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="],
+6
View File
@@ -17,6 +17,12 @@ saveTextLockfile = true
# scratch dirs so a root-level `bun test` doesn't walk into them.
pathIgnorePatterns = [
"**/node_modules/**",
"**/.git/**",
"**/target/**",
"**/dist/**",
"python/**",
"docs/**",
"runs/**",
"python/robomp/data/**",
".wt/**",
".worktrees/**",
+41 -1
View File
@@ -77,8 +77,20 @@ pub fn block_range_at(options: BlockRangeOptions) -> Result<Option<BlockRange>>
};
let root = tree.root_node();
// Query a one-column-wide range over the first content character rather
// than a zero-width point. Some grammars (e.g. tree-sitter-swift) insert a
// zero-width separator node at the start of a statement that follows a
// blank line. An empty point range at that node's start gets absorbed into
// the invisible node, which has no children and is not "relevant", so
// `named_descendant_for_point_range` bubbles back up to the last visible
// ancestor (the enclosing body, or the file root). That made `replace
// block` on a line like `var body: some View {` preceded by a blank line
// resolve to the whole enclosing type body and then fail. Spanning the
// first character skips the zero-width node (its end is < the range end)
// and forces the descent into the node that begins on `row`.
let point = Point::new(row, col);
let Some(leaf) = root.named_descendant_for_point_range(point, point) else {
let point_end = Point::new(row, col + 1);
let Some(leaf) = root.named_descendant_for_point_range(point, point_end) else {
return Ok(None);
};
// A leaf whose own start row is earlier than `row` means `point` landed on
@@ -372,6 +384,34 @@ mod tests {
assert_eq!(resolve(code, "r.rs", 2), Some(BlockRange { start_line: 2, end_line: 4 }));
}
#[test]
fn resolves_swift_computed_property_after_blank_line() {
// Regression: a block whose opening line is preceded by a blank line
// (here the SwiftUI `var body: some View {` computed property) used to
// resolve to nothing. tree-sitter-swift inserts a zero-width separator
// node at the start of a statement that follows a blank line; a
// zero-width point query at the first content column gets absorbed into
// that invisible node and bubbles back up to the enclosing type body. A
// one-column-wide query skips the zero-width node and descends into the
// property that actually begins on the line.
let code = "struct MenuBarUsage: View {\n let metric: AccountMetric\n\n var body: \
some View {\n VStack {\n Text(\"Usage\")\n }\n \
}\n}\n";
assert_eq!(
resolve(code, "MenuBarUsage.swift", 4),
Some(BlockRange { start_line: 4, end_line: 8 })
);
}
#[test]
fn resolves_swift_top_level_decl_after_blank_line() {
// Same zero-width-separator regression one level up: a top-level
// declaration following a blank line. Without the fix the query
// resolved to the whole `source_file` root and was rejected.
let code = "import Foundation\n\nfunc greet() {\n print(\"hi\")\n}\n";
assert_eq!(resolve(code, "g.swift", 3), Some(BlockRange { start_line: 3, end_line: 5 }));
}
fn boundaries(code: &str, path: &str, ranges: &[(u32, u32)]) -> Option<Vec<u32>> {
enclosing_block_boundaries(EnclosingBoundaryOptions {
code: code.to_string(),
+17 -12
View File
@@ -553,7 +553,16 @@ mod imp {
};
if matched {
let extended_info = symlink_extended_info(entry.symlink_target.as_deref());
let extended_info = entry
.symlink_target
.as_deref()
.map(|target| PRJ_EXTENDED_INFO {
InfoType: PRJ_EXT_INFO_TYPE_SYMLINK,
NextInfoOffset: 0,
Anonymous: PRJ_EXTENDED_INFO_0 {
Symlink: PRJ_EXTENDED_INFO_0_0 { TargetName: target.as_ptr() },
},
});
let extended_info_ptr = extended_info
.as_ref()
.map_or(std::ptr::null(), |info| info as *const _);
@@ -599,7 +608,13 @@ mod imp {
Ok(target) => target,
Err(err) => return io_error_to_hresult(&err),
};
let extended_info = symlink_extended_info(symlink_target.as_deref());
let extended_info = symlink_target.as_deref().map(|target| PRJ_EXTENDED_INFO {
InfoType: PRJ_EXT_INFO_TYPE_SYMLINK,
NextInfoOffset: 0,
Anonymous: PRJ_EXTENDED_INFO_0 {
Symlink: PRJ_EXTENDED_INFO_0_0 { TargetName: target.as_ptr() },
},
});
let extended_info_ptr = extended_info
.as_ref()
.map_or(std::ptr::null(), |info| info as *const _);
@@ -782,16 +797,6 @@ mod imp {
Ok(Some(target))
}
fn symlink_extended_info(target: Option<&[u16]>) -> Option<PRJ_EXTENDED_INFO> {
target.map(|target| PRJ_EXTENDED_INFO {
InfoType: PRJ_EXT_INFO_TYPE_SYMLINK,
NextInfoOffset: 0,
Anonymous: PRJ_EXTENDED_INFO_0 {
Symlink: PRJ_EXTENDED_INFO_0_0 { TargetName: target.as_ptr() },
},
})
}
fn callback_relative_path(callback_data: &PRJ_CALLBACK_DATA) -> PathBuf {
if callback_data.FilePathName.is_null() {
return PathBuf::new();
+8 -3
View File
@@ -111,9 +111,7 @@ mod imp {
let resolved = if path.is_absolute() {
path.to_path_buf()
} else {
std::env::current_dir()
.map(|cwd| cwd.join(path))
.unwrap_or_else(|_| path.to_path_buf())
std::env::current_dir().map_or_else(|_| path.to_path_buf(), |cwd| cwd.join(path))
};
let meta = fs::metadata(&resolved).map_err(|err| {
IsoError::other(format!("invalid block-clone source {}: {err}", resolved.display()))
@@ -164,6 +162,13 @@ mod imp {
}
let mut permissions = meta.permissions();
if permissions.readonly() {
// This backend only removes a temporary Windows block-clone tree; clearing
// the readonly file attribute is required so removal can proceed.
#[allow(
clippy::permissions_set_readonly_false,
reason = "Windows block-clone cleanup must clear the readonly file attribute before \
deletion"
)]
permissions.set_readonly(false);
let _ = fs::set_permissions(path, permissions);
}
+146 -2
View File
@@ -53,8 +53,109 @@ pub mod tokens;
pub(crate) mod utils;
pub mod workspace;
#[cfg(target_os = "windows")]
use std::sync::{
Arc,
atomic::{AtomicBool, Ordering},
};
#[cfg(target_os = "windows")]
use napi::bindgen_prelude::create_custom_tokio_runtime;
use napi_derive::{module_init, napi};
/// Upper bound on Windows Tokio *scheduler* workers. These only drive async I/O
/// futures (shell/process/PTY/ISO) and light glue tasks; all CPU-heavy and
/// blocking native work runs elsewhere — libuv tasks (`task::blocking`), Rayon,
/// or Tokio's separate blocking pool via `spawn_blocking` — so a handful of
/// async workers is plenty regardless of core count.
#[cfg(target_os = "windows")]
const NAPI_TOKIO_MAX_WORKER_THREADS: usize = 4;
/// Cap on Tokio's lazily-grown blocking pool (used by `spawn_blocking` offloads
/// such as `iso_start`/`iso_stop`/`pty.start`/`walk_diff`). Threads here are
/// created on demand, not at load, so this only bounds peak fan-out.
#[cfg(target_os = "windows")]
const NAPI_TOKIO_MAX_BLOCKING_THREADS: usize = 8;
/// Windows worker count we'd *like*, before checking what the OS will actually
/// grant: the Tokio default (one per core) clamped to
/// [`NAPI_TOKIO_MAX_WORKER_THREADS`].
#[cfg(target_os = "windows")]
fn desired_worker_threads() -> usize {
std::thread::available_parallelism()
.map_or(1, |threads| threads.get())
.clamp(1, NAPI_TOKIO_MAX_WORKER_THREADS)
}
/// Probe how many worker threads Windows will let us hold alive
/// *simultaneously*, up to `target`. Returns the count actually spawned (0 when
/// not even one extra thread is possible).
///
/// `Builder::build()` for a multi-thread runtime spawns every worker eagerly
/// and **panics** (not `Err`) when Windows refuses one — on a
/// memory-constrained host (tiny pagefile / commit limit, `os error 1455`) that
/// aborts the whole process at addon load before any JS error can surface. The
/// release profile is `panic = "abort"`, so the panic can't even be caught. We
/// instead pre-flight with `std::thread::Builder::spawn`, which returns an
/// `io::Result`, holding each probe thread alive (so their stacks are committed
/// concurrently, matching how real workers coexist) until we know the safe
/// count. Probe threads use the std default stack, exactly like Tokio's workers
/// (it leaves `thread_stack_size` unset), so the probe is representative.
///
/// Keep this Windows-only. On Linux, spawning probe threads from `module_init`
/// can deadlock while Bun is loading the `.node`; napi-rs's default runtime
/// loads cleanly there and avoids any custom loader-time thread probe.
#[cfg(target_os = "windows")]
fn probe_spawnable_workers(target: usize) -> usize {
let keep_running = Arc::new(AtomicBool::new(true));
let mut handles = Vec::with_capacity(target);
for _ in 0..target {
let keep = Arc::clone(&keep_running);
match std::thread::Builder::new().spawn(move || {
while keep.load(Ordering::Relaxed) {
std::thread::park_timeout(std::time::Duration::from_millis(1));
}
}) {
Ok(handle) => handles.push(handle),
Err(_) => break,
}
}
let spawned = handles.len();
keep_running.store(false, Ordering::Relaxed);
for handle in handles {
handle.thread().unpark();
let _ = handle.join();
}
spawned
}
/// Build the custom Tokio runtime napi-rs uses on Windows, sized to what the
/// host can actually spawn. Never panics: backs off from
/// [`desired_worker_threads`] to whatever the probe allows, and falls back to a
/// current-thread runtime (which spawns no workers at build time, so it can't
/// abort under commit-limit pressure) when not even one worker is available.
/// Returns `None` only if even that fails, in which case we leave napi-rs to
/// construct its own default.
#[cfg(target_os = "windows")]
fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
let workers = probe_spawnable_workers(desired_worker_threads());
let multi_thread = (workers > 0)
.then(|| {
tokio::runtime::Builder::new_multi_thread()
.worker_threads(workers)
.max_blocking_threads(NAPI_TOKIO_MAX_BLOCKING_THREADS)
.enable_all()
.build()
.ok()
})
.flatten();
multi_thread.or_else(|| {
tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.ok()
})
}
/// Version sentinel — exists solely so the JS loader can prove at load time
/// that the `.node` file on disk is from the same package release as the
/// `index.js` ESM wrapper invoking it.
@@ -71,12 +172,55 @@ use napi_derive::{module_init, napi};
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV15_12_5")]
#[napi(js_name = "__piNativesV15_13_0")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
/// invoke a panicking or allocating native call. Runs once at `.node` load.
/// invoke a panicking or allocating native call. This runs during `.node`
/// load, while the dynamic-loader lock is held, so it MUST NOT spawn threads —
/// the Tokio runtime is installed afterwards on Windows by
/// [`omp_install_tokio_runtime`], which the JS loader calls once `dlopen` has
/// returned.
///
/// On Windows, the custom Tokio runtime is host-sized to prevent aborts under
/// memory limits (see [`create_windows_napi_tokio_runtime`]). Non-Windows
/// builds intentionally use napi-rs's default path. Linux source builds can
/// deadlock if this module initializer or post-load setup performs its own
/// thread probe, so we keep the probe and custom runtime Windows-only.
#[module_init]
fn install_native_crash_handler() {
crash_handler::install();
}
/// Guards [`omp_install_tokio_runtime`] so the runtime is built at most once
/// per process even if the loader invokes it more than once.
#[cfg(target_os = "windows")]
static TOKIO_RUNTIME_INSTALLED: AtomicBool = AtomicBool::new(false);
/// Install the bounded Tokio runtime napi-rs adopts for async exports.
///
/// The JS loader calls this exactly once, synchronously, right *after* `dlopen`
/// returns and *before* any async native runs — never from `#[module_init]`.
/// Building a multi-thread runtime eagerly spawns worker threads, and doing
/// that during module init (while the dynamic-loader lock is held) deadlocks on
/// some hosts: a fresh worker blocks acquiring the loader lock that the init
/// thread still owns. napi-rs only materializes its runtime on the first async
/// call (`RT` is a `LazyLock`) and `create_custom_tokio_runtime` merely records
/// the runtime in a `OnceLock`, so installing it post-load is still honored.
/// Without it napi builds its own default (one worker per CPU, spawned eagerly)
/// which aborts the process (`os error 1455`) on a memory-constrained Windows
/// host before any JS error can surface; [`create_windows_napi_tokio_runtime`]
/// pre-flights the spawn instead. If no runtime can be built we leave napi-rs
/// to its default. Idempotent.
#[napi(js_name = "__ompInstallTokioRuntime")]
#[allow(clippy::missing_const_for_fn, reason = "napi macro is incompatible with const fn")]
pub fn omp_install_tokio_runtime() {
#[cfg(target_os = "windows")]
if TOKIO_RUNTIME_INSTALLED.swap(true, Ordering::SeqCst) {
return;
}
#[cfg(target_os = "windows")]
if let Some(runtime) = create_windows_napi_tokio_runtime() {
create_custom_tokio_runtime(runtime);
}
}
+6 -2
View File
@@ -424,9 +424,13 @@ mod tests {
.parse::<i32>()
.expect("child pid parses");
// SAFETY: `getsid(0)` only queries the current process session; the
// return value is checked below.
// return value is checked below. Inside a PID namespace (e.g. the
// containerized CI runner) the host's session leader can live outside
// the namespace, so `getsid(0)` legitimately reports 0 — only -1 is a
// real failure. The meaningful invariant is that the child detached
// into its own session (`child_sid == child_pid`, distinct from host).
let host_sid = unsafe { libc::getsid(0) };
assert!(host_sid > 0, "getsid(0) failed: {}", std::io::Error::last_os_error());
assert!(host_sid >= 0, "getsid(0) failed: {}", std::io::Error::last_os_error());
// SAFETY: `child_pid` is a live positive PID reported by the child; the
// return value is checked below.
let child_sid = unsafe { libc::getsid(child_pid) };
+22 -4
View File
@@ -203,10 +203,14 @@ fn io_redirect_is_safe(io: &IoRedirect) -> bool {
IoFileRedirectTarget::Fd(_) => true,
IoFileRedirectTarget::ProcessSubstitution(..) => false,
},
IoRedirect::HereDocument(_, here_doc) => {
!word_has_command_substitution(&here_doc.here_end)
&& !word_has_command_substitution(&here_doc.doc)
},
// Here-docs are never safe to segment. The segmented runner rebuilds
// each chain segment from the brush AST via `pipeline.to_string()`, and
// that Display impl re-emits a quoted/escaped here-doc's *closing*
// delimiter with its quotes intact (`<<'EOF'` … `'EOF'` rather than the
// required bare `EOF`). The reconstructed close tag never matches, so
// the re-run segment fails with "unterminated here document". Leave any
// here-doc-bearing command to the unsegmented single path.
IoRedirect::HereDocument(..) => false,
IoRedirect::HereString(_, word) => !word_has_command_substitution(word),
IoRedirect::OutputAndError(word, _) => !word_has_command_substitution(word),
}
@@ -418,6 +422,20 @@ mod tests {
}
}
#[test]
fn heredoc_chains_are_not_segmented() {
// Regression: a chain whose segment carries a here-doc must not be
// split. The segmented runner rebuilds each segment with the brush AST
// Display impl, which re-emits a quoted/escaped here-doc's closing
// delimiter with quotes (`'EOF'` rather than `EOF`); re-running that
// reconstructed segment fails with "unterminated here document". Both
// quoted and unquoted delimiters bail so the whole command runs whole.
assert_not_chain("cat <<'EOF'\nbody\nEOF\necho done");
assert_not_chain("cat <<\"EOF\"\nbody\nEOF\necho done");
assert_not_chain("cat <<EOF\nbody\nEOF\necho done");
assert_not_chain("cat <<'EOF' && echo done\nbody\nEOF");
}
#[test]
fn rejects_legacy_opaque_shapes() {
assert_eq!(analyze("foo || bar"), CommandPlan::Compound);
+52 -14
View File
@@ -65,6 +65,7 @@ mod platform {
let mut seen: HashSet<i32> = HashSet::new();
let mut out = Vec::new();
let mut children_file_available = false;
for entry in entries.flatten() {
let name = entry.file_name();
let Some(tid_str) = name.to_str() else {
@@ -77,26 +78,56 @@ mod platform {
let Ok(content) = fs::read_to_string(&children_path) else {
continue;
};
// The file is readable -> this kernel has CONFIG_PROC_CHILDREN.
children_file_available = true;
for part in content.split_whitespace() {
let Ok(child_pid) = part.parse::<i32>() else {
continue;
};
if !seen.insert(child_pid) {
continue;
}
let Some(child) = Self::from_pid(child_pid) else {
self.push_validated_child(child_pid, &mut seen, &mut out);
}
}
// Some Kata / microVM guest kernels are built without CONFIG_PROC_CHILDREN,
// so no `.../children` file exists and the walk above finds nothing — which
// would silently turn descendant signaling (cancellation cleanup) into a
// no-op inside such containers. Fall back to scanning `/proc` and grouping
// by parent pid, the same primitive the macOS path uses. Only taken when no
// `children` file was readable, so kernels that support it keep the cheap
// per-task fast path.
if !children_file_available && let Ok(proc_entries) = fs::read_dir("/proc") {
for entry in proc_entries.flatten() {
let name = entry.file_name();
let Some(pid_str) = name.to_str() else {
continue;
};
if child.status() == ProcessStatus::Running
&& current_parent_pid(child.pid) == Some(self.pid)
{
out.push(child);
}
let Ok(child_pid) = pid_str.parse::<i32>() else {
continue;
};
self.push_validated_child(child_pid, &mut seen, &mut out);
}
}
out
}
/// Validate a candidate child pid — dedup, still running, and currently
/// parented to `self` — then push it onto `out`. Shared by the
/// `/proc/<pid>/task/<tid>/children` fast path and the `/proc`-scan
/// fallback for kernels without `CONFIG_PROC_CHILDREN`.
fn push_validated_child(&self, child_pid: i32, seen: &mut HashSet<i32>, out: &mut Vec<Self>) {
if child_pid == self.pid || !seen.insert(child_pid) {
return;
}
let Some(child) = Self::from_pid(child_pid) else {
return;
};
if child.status() == ProcessStatus::Running
&& current_parent_pid(child.pid) == Some(self.pid)
{
out.push(child);
}
}
pub fn parent_pid(&self) -> Option<i32> {
if self.status() == ProcessStatus::Running {
current_parent_pid(self.pid)
@@ -805,7 +836,7 @@ mod platform {
}
}
fn as_raw(&self) -> Handle {
const fn as_raw(&self) -> Handle {
self.raw as Handle
}
}
@@ -932,7 +963,7 @@ mod platform {
unsafe { TerminateProcess(self.handle.as_raw(), 1) != 0 }
}
pub const fn group_id(&self) -> Option<i32> {
pub const fn group_id() -> Option<i32> {
None
}
@@ -1047,7 +1078,7 @@ mod platform {
}
fn read_remote_unicode_string(handle: Handle, value: UnicodeString) -> Option<String> {
if value.length == 0 || value.buffer == 0 || value.length % 2 != 0 {
if value.length == 0 || value.buffer == 0 || !value.length.is_multiple_of(2) {
return None;
}
let code_units = usize::from(value.length) / size_of::<u16>();
@@ -1135,7 +1166,7 @@ mod platform {
OwnedHandle::from_raw(snapshot)
}
fn process_entry() -> PROCESSENTRY32W {
const fn process_entry() -> PROCESSENTRY32W {
PROCESSENTRY32W {
dwSize: mem::size_of::<PROCESSENTRY32W>() as u32,
cntUsage: 0,
@@ -1288,6 +1319,13 @@ impl Process {
}
/// Process group id for this process, when supported by the platform.
#[cfg(target_os = "windows")]
#[must_use]
pub const fn group_id(&self) -> Option<i32> {
platform::Process::group_id()
}
#[cfg(not(target_os = "windows"))]
#[must_use]
pub fn group_id(&self) -> Option<i32> {
self.inner.group_id()
@@ -1357,7 +1395,7 @@ impl Process {
// If self leads its own process group, also signal the group — this catches
// grandchildren reparented to init when their immediate parent died inside
// the descendant walk.
if let Some(pgid) = self.inner.group_id()
if let Some(pgid) = self.group_id()
&& pgid == self.inner.pid()
{
let _ = kill_process_group(pgid, signal);
+44 -67
View File
@@ -2211,6 +2211,32 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
assert_eq!(minimized.output_bytes, 9);
}
/// Regression: a quoted here-doc followed by another command must execute
/// instead of failing with "unterminated here document". The minimizer's
/// segmented runner used to rebuild each segment via the brush AST Display
/// impl, which re-emitted the `<<'PY'` close tag as the quoted `'PY'` — an
/// invalid delimiter that left the body unterminated. Here-doc-bearing
/// commands now bail out of segmentation and run whole via the single path.
#[cfg(unix)]
#[tokio::test(flavor = "multi_thread")]
async fn quoted_heredoc_in_chain_runs_via_single_path() {
let root = unique_temp_dir("heredoc-chain");
let minimizer = printf_minimizer(&root.join("minimizer.toml"), None);
let (result, output) = run_command_capture(
"/bin/cat <<'PY'\nhello $USER\nPY\nprintf 'after\\n'",
None,
Some(minimizer),
CancelToken::default(),
)
.await;
let _ = std::fs::remove_dir_all(&root);
assert_eq!(result.exit_code, Some(0));
// Quoted delimiter keeps the body literal ($USER unexpanded) and the
// trailing command still runs in order.
assert_eq!(output, "hello $USER\nafter\n");
assert!(!output.contains("unterminated"));
}
#[cfg(unix)]
#[tokio::test(flavor = "multi_thread")]
async fn segmented_chain_exceeding_aggregate_capture_cap_stays_raw() {
@@ -2291,9 +2317,12 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
use std::io::Read as _;
// SAFETY: `getsid(0)` only queries the current process session; the return
// value is checked.
// value is checked. Inside a PID namespace (the containerized CI runner)
// the host's session leader can live outside the namespace, so `getsid(0)`
// legitimately reports 0 — only -1 is a real failure. The child-session
// invariants below (own session, distinct from host) stay meaningful.
let host_sid = unsafe { libc::getsid(0) };
assert!(host_sid > 0, "getsid(0) failed: {}", std::io::Error::last_os_error());
assert!(host_sid >= 0, "getsid(0) failed: {}", std::io::Error::last_os_error());
// Build the same kind of session pi-natives uses in production.
let config = ShellConfig { session_env: None, snapshot_path: None, minimizer: None };
@@ -2412,9 +2441,12 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
async fn embedded_pipeline_stage_runs_in_its_own_session() {
use std::io::Read as _;
// SAFETY: `getsid(0)` only queries the current process session; checked below.
// SAFETY: `getsid(0)` only queries the current process session; checked
// below. In a PID namespace (containerized CI) the host's session leader
// can live outside the namespace, so `getsid(0)` reports 0, not an error;
// only -1 is a real failure.
let host_sid = unsafe { libc::getsid(0) };
assert!(host_sid > 0, "getsid(0) failed: {}", std::io::Error::last_os_error());
assert!(host_sid >= 0, "getsid(0) failed: {}", std::io::Error::last_os_error());
let config = ShellConfig { session_env: None, snapshot_path: None, minimizer: None };
let mut session = create_session(&config).await.expect("create_session");
@@ -2511,6 +2543,7 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
);
}
#[cfg(unix)]
#[tokio::test(flavor = "multi_thread")]
async fn wait_accepts_last_background_process_id() {
let options = ShellExecuteOptions {
@@ -2527,6 +2560,7 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
assert!(!result.timed_out);
}
#[cfg(unix)]
#[tokio::test(flavor = "multi_thread")]
async fn wait_n_p_records_completed_process_id() {
let options = ShellExecuteOptions {
@@ -2546,6 +2580,7 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
assert!(!result.timed_out);
}
#[cfg(unix)]
#[tokio::test(flavor = "multi_thread")]
async fn wait_f_accepts_process_id() {
let options = ShellExecuteOptions {
@@ -2576,66 +2611,6 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
assert!(matches!(reason, AbortReason::Signal));
}
#[tokio::test(flavor = "multi_thread")]
async fn cancellation_aborts_internal_background_jobs() {
let unique = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.expect("system clock before epoch")
.as_nanos();
let dir =
std::env::temp_dir().join(format!("pi-shell-bg-cancel-{}-{unique}", std::process::id()));
std::fs::create_dir(&dir).expect("create temp dir");
let started = dir.join("started");
let release = dir.join("release");
let marker = dir.join("marker");
let config = ShellConfig { session_env: None, snapshot_path: None, minimizer: None };
let mut session = create_session(&config).await.expect("create session");
session
.shell
.set_working_dir(dir.to_string_lossy().as_ref())
.expect("set cwd");
let mut params = session.shell.default_exec_params();
params.set_fd(OpenFiles::STDIN_FD, null_file().expect("null stdin"));
params.set_fd(OpenFiles::STDOUT_FD, null_file().expect("null stdout"));
params.set_fd(OpenFiles::STDERR_FD, null_file().expect("null stderr"));
let source_info = SourceInfo::from("pi-shell:test");
let result = session
.shell
.run_string(
"{ echo started > started; while [ ! -f release ]; do sleep 0.05; done; echo done > \
marker; } &",
&source_info,
&params,
)
.await
.expect("spawn background job");
assert_eq!(exit_code(&result), 0);
let mut background_started = false;
for _ in 0..200 {
if started.exists() {
background_started = true;
break;
}
time::sleep(Duration::from_millis(10)).await;
}
assert!(background_started, "background job did not reach its wait loop");
terminate_background_jobs(&mut session.shell);
std::fs::write(&release, b"").expect("release marker");
time::sleep(Duration::from_millis(250)).await;
let marker_exists = marker.exists();
std::fs::remove_dir_all(&dir).expect("cleanup temp dir");
assert!(
!marker_exists,
"internal background job survived cancellation and wrote marker after release",
);
}
#[cfg(unix)]
#[tokio::test]
async fn read_output_stops_when_cancelled_before_pipe_eof() {
@@ -2796,10 +2771,12 @@ replace = [{ pattern = "^.+$", replacement = "PWD" }]
/// own exit status — not nohup's (`125`/`126`/`127`) error codes.
#[tokio::test(flavor = "multi_thread")]
async fn nohup_builtin_propagates_command_exit_code() {
let options = ShellExecuteOptions {
command: "nohup /bin/sh -c 'exit 7'".to_string(),
..Default::default()
let command = if cfg!(windows) {
"nohup cmd /C exit 7"
} else {
"nohup /bin/sh -c 'exit 7'"
};
let options = ShellExecuteOptions { command: command.to_string(), ..Default::default() };
let result = execute_shell(options, None, CancelToken::default())
.await
.expect("execute should succeed");
+1 -1
View File
@@ -264,7 +264,7 @@ Generate a session name using lowercase `<type>:<primary-objective>`.
- Missing `TITLE_SYSTEM.md` keeps the bundled title prompts.
- Discovery uses the same project-then-user config directory pattern as `SYSTEM.md`: project `.omp/TITLE_SYSTEM.md` first, then user `~/.omp/agent/TITLE_SYSTEM.md` and the other supported config bases.
- The override replaces only the automatic session-title generation system prompt; normal `SYSTEM.md` / `APPEND_SYSTEM.md` prompt customization is unaffected.
- The online path still forces the `set_title` tool call. The local tiny-title path keeps the `<title>...</title>` prefill/stop wrapper and uses this file as its system turn.
- The online path forces the `set_title` tool call when the title model honors a forced `tool_choice`. Tool-choice-less providers (chat-completions hosts without `tool_choice` support, Claude Fable/Mythos) instead receive a marker-based prompt and emit the title wrapped in `<title>...</title>`, which is parsed leniently (a plain sentence or a truncated/unclosed tag still works). A `TITLE_SYSTEM.md` override is reused in both modes; in marker mode the wrap-in-`<title>` instruction is appended after it. The local tiny-title path keeps the `<title>...</title>` prefill/stop wrapper and uses this file as its system turn.
## Skills subsystem
+18
View File
@@ -151,12 +151,30 @@ Handlers and tool `execute` receive `ctx` with:
- `cwd`
- `sessionManager` (read-only)
- `modelRegistry`, `model`
- `models` (read-only model query — see below)
- `getContextUsage()`
- `compact(...)`
- `isIdle()`, `hasPendingMessages()`, `abort()`
- `shutdown()`
- `getSystemPrompt()`
### Model selection (`ctx.models`)
`ctx.models` is a read-only facade for picking and comparing models the same way core does:
- `list()` — authenticated models available this session.
- `current()` — the live session model (read lazily, so it reflects `/model` switches).
- `resolve(spec)` — a model string (`provider/id`, bare id) or role alias (`pi/slow`, a configured role) → `Model`, honoring the same settings-backed aliases and match preferences as `--model`. Returns `undefined` when nothing matches.
- `family(model)` — an opaque lineage token for "same family?" checks (Claude point releases share a token; Claude and GPT differ). Compare it; don't persist it (the vocabulary tracks new releases).
```ts
// Pick a model from a different family than the current one (e.g. a cross-family reviewer).
const current = ctx.models.current();
const contrasting = ctx.models
.list()
.find(m => current && ctx.models.family(m) !== ctx.models.family(current));
```
## 3) Command context (`ExtensionCommandContext`)
Command handlers additionally get:
+2 -2
View File
@@ -17,7 +17,7 @@ Chord names are case-insensitive and use the same notation shown in the UI, such
Set an action to an empty array to disable it:
```yaml
app.stt.toggle: []
app.history.search: []
```
## Common action IDs
@@ -40,7 +40,7 @@ app.stt.toggle: []
| `app.clipboard.copyLine` | `Alt+Shift+L` | Copy the current line |
| `app.clipboard.copyPrompt` | `Alt+Shift+C` | Copy the whole prompt |
| `app.clipboard.pasteImage` | `Ctrl+V` (`Alt+V` fallback on Windows) | Paste from the clipboard (image preferred, text fallback) |
| `app.stt.toggle` | `Alt+H` | Toggle speech-to-text recording |
| `app.stt.toggle` | Unbound (hold `Space`) | Toggle speech-to-text. By default there is no key chord — hold the space bar to record (push-to-talk) and release to transcribe; bind a chord here for a press-to-toggle alternative |
On Windows Terminal, `Ctrl+V` may be handled by the terminal paste command before `omp` sees it; use the `Alt+V` fallback when clipboard image paste appears to do nothing. When the clipboard holds no image, `app.clipboard.pasteImage` pastes the clipboard text instead, so hosts that deliver only this chord (VS Code's integrated terminal when configured to forward `Ctrl+V`, Windows clipboard history via `Win+V`) work for both payload kinds. Windows Terminal also swallows `Ctrl+Enter`, so the follow-up shortcut also binds `Ctrl+Q` — the same chord GitHub Copilot CLI uses. If your existing `keybindings.yml` already assigns `Ctrl+Q` to another action, that user remap wins and follow-up keeps `Ctrl+Enter` unless you explicitly bind `app.message.followUp`.
+27
View File
@@ -641,6 +641,33 @@ providers:
type: openai-models-list
```
The built-in vLLM provider can be pointed at a non-default endpoint without declaring a custom discovery type. OMP uses vLLM's `/v1/models` metadata and preserves vLLM's `max_model_len` field as the discovered context window.
```yaml
providers:
vllm:
baseUrl: http://192.168.5.3:8085/v1
auth: none
```
For multiple vLLM endpoints, use arbitrary provider IDs with the generic OpenAI-compatible discovery path. Set `auth: none` for local no-auth servers or `apiKey` for authenticated ones. Generic discovery reads `max_model_len` first and then `context_length` as a generic OpenAI-compatible fallback.
```yaml
providers:
vllm-fast:
baseUrl: http://host-a:8000/v1
auth: none
api: openai-completions
discovery:
type: openai-models-list
vllm-long:
baseUrl: http://host-b:8000/v1
auth: none
api: openai-completions
discovery:
type: openai-models-list
```
### Hosted proxy with env-based key
```yaml
+31 -3
View File
@@ -45,8 +45,9 @@ There is no envelope beyond the object shape itself.
6. Host URI requests/cancellations (`host_uri_request`, `host_uri_cancel`)
7. Extension errors (`{ type: "extension_error", extensionPath, event, error }`)
8. Available-commands updates (`{ type: "available_commands_update", commands }`), emitted at startup and whenever command metadata changes
9. Subagent frames (`subagent_lifecycle`, `subagent_progress`, `subagent_event`), gated by `set_subagent_subscription`
10. Builtin slash-command side channels (`command_output`, `session_info_update`, `config_update`)
9. Prompt lifecycle hints (`{ type: "prompt_result", id?, agentInvoked }`) for scheduled prompts that later resolve without invoking the agent
10. Subagent frames (`subagent_lifecycle`, `subagent_progress`, `subagent_event`), gated by `set_subagent_subscription`
11. Builtin slash-command side channels (`command_output`, `session_info_update`, `config_update`)
### Inbound frame categories (stdin)
@@ -67,6 +68,8 @@ Important edge behavior from runtime:
- Unknown command responses are emitted with `id: undefined` (even if the request had an `id`).
- Parse/handler exceptions in the input loop emit `command: "parse"` with `id: undefined`.
- `prompt` and `abort_and_prompt` return immediate success, then may emit a later error response with the **same** id if async prompt scheduling fails.
- `prompt` success responses may include `data.agentInvoked`. `false` means the prompt completed locally without an agent turn; `true` means the prompt produced agent lifecycle events; omitted means the host must rely on session events for completion.
- `abort_and_prompt` does not currently emit `data.agentInvoked` or `prompt_result`; hosts should treat it as the legacy abort-then-schedule path and rely on session events or same-id scheduling errors.
## Command Schema (canonical)
@@ -153,6 +156,30 @@ All command results use `RpcResponse`:
Data payloads are command-specific and defined in `rpc-types.ts`.
### `prompt` payload
`prompt` is acknowledged after the command is accepted, not after a model turn finishes:
```json
{
"id": "req_1",
"type": "response",
"command": "prompt",
"success": true,
"data": { "agentInvoked": false }
}
```
`data.agentInvoked: false` is a completion signal for local-only prompts, including slash commands that produce output without starting an agent turn. `data.agentInvoked: true` means the prompt produced agent lifecycle events; those events can be emitted before or after the prompt response depending on the command path. Older runtimes may omit `data`; hosts should then rely on `agent_end`, custom message completion, or `prompt_result`.
`prompt_result` is emitted when a prompt was accepted immediately but later resolves as local-only:
```json
{ "type": "prompt_result", "id": "req_1", "agentInvoked": false }
```
Local-only slash commands may emit `command_output` frames before completing via `data.agentInvoked: false` or a later `prompt_result`. They do not emit `agent_end`.
### `get_state` payload
```json
@@ -344,7 +371,8 @@ This is the most important operational behavior.
That means:
- command acceptance != run completion
- final completion is observed via `agent_end`
- agent turns complete via `agent_end`
- local-only prompts complete via `data.agentInvoked: false` on the response or via a later `prompt_result`
### While streaming
+3
View File
@@ -502,6 +502,9 @@ memory:
| `compaction.remoteEnabled` | boolean | `true` | Allow remote compaction service. |
| `compaction.autoContinue` | boolean | `true` | Continue automatically after compaction. |
| `memory.backend` | enum | `off` | `off`, `local`, `hindsight`, `mnemopi`. Each backend has its own `hindsight.*` / `mnemopi.*` / `memories.*` tuning keys. |
| `autolearn.enabled` | boolean | `false` | Experimental: after the agent stops, nudge it to capture lessons to memory and create/enhance isolated managed skills under `~/.omp/agent/managed-skills`. Enables the `manage_skill` tool (and `learn` when a memory backend is active). |
| `autolearn.autoContinue` | boolean | `false` | When `autolearn.enabled`, auto-run one capture turn at stop (uses extra tokens). Off = a passive reminder rides your next turn. |
| `autolearn.minToolCalls` | number | `5` | Only nudge after a turn that used at least this many tools. |
`compaction` has additional tuning keys (idle compaction, supersede/drop heuristics) visible in `omp config list`. See [Compaction](./compaction.md) for the full strategy reference.
+6 -6
View File
@@ -134,11 +134,11 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod
- `completion(prompt, opts?)` for oneshot, stateless model calls (see _Oneshot completion helper_ below)
- `agent(prompt, opts?)` for a single subagent call, plus `parallel()` / `pipeline()` bounded-pool helpers (see _Subagent helper_ below)
- JS helpers that touch the host/runtime boundary are async and `await`able; pure text helpers (`sort`, `uniq`, `counter`) return synchronously but may still be safely awaited.
- JS helper signatures use a trailing options object rather than Python keyword arguments:
- `await read(path, { offset?, limit? })`
- `await tree(path = ".", { maxDepth?, hidden? })`
- `sort(text, { reverse?, unique? })`, `uniq(text, { count? })`, `counter(items, { limit?, reverse? })`
- `await agent(prompt, { agentType?, model?, label?, schema? })`
- JS helper options may be passed either positionally in the Python order or as a trailing options object. `null` and `undefined` skip positional slots:
- `await read(path, offset?, limit?)` or `await read(path, { offset?, limit? })`
- `await tree(path = ".", maxDepth?, showHidden?)` or `await tree(path, { maxDepth?, showHidden? })`
- `sort(text, reverse?, unique?)`, `uniq(text, count?)`, `counter(items, limit?, reverse?)`
- `await agent(prompt, agentType?, model?, label?, schema?)` or `await agent(prompt, { agentType?, model?, label?, schema? })`
- `await parallel([() => agent("a"), () => agent("b")])`
- `await pipeline(items, stage1, stage2)`
- `display(value)` behavior:
@@ -192,7 +192,7 @@ Both runtimes expose `completion()` — a single stateless completion against a
Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state.
- Signatures:
- JS: `await agent(prompt, { agentType?, model?, label?, schema? })`
- JS: `await agent(prompt, agentType?, model?, label?, schema?)` or `await agent(prompt, { agentType?, model?, label?, schema? })`
- Python: `agent(prompt, *, agent_type="task", model=None, label=None, schema=None)`
- `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work.
- `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply.
+5 -4
View File
@@ -62,7 +62,7 @@ Read-only snapshot path:
7. If every watched job is already non-running, `#buildResult(...)` returns immediately without waiting.
8. Otherwise the tool waits on `Promise.race(...)` across:
- every watched running job's `job.promise`,
- a timeout promise for `async.pollWaitDuration`,
- a timeout promise for the poll wait window — `manager.nextPollWaitMs(ownerId)` when `async.pollWaitDuration` is `smart`, otherwise the fixed duration,
- the tool-call abort signal when present.
9. Before waiting, it calls `manager.watchJobs(watchedJobIds)`. This suppresses automatic completion delivery for those ids while they are being watched.
10. If `onUpdate` exists, a 500 ms interval sends progress snapshots from `#snapshotJobs(...)`; one snapshot is emitted immediately before entering the race.
@@ -106,9 +106,10 @@ Lifecycle and exact state names:
- Cancelling a job does not synchronously await teardown; it flips state, aborts, and returns control to the manager/job promise.
## Limits & Caps
- Poll wait duration comes from `async.pollWaitDuration` in `packages/coding-agent/src/config/settings-schema.ts`:
- allowed values: `5s`, `10s`, `30s`, `1m`, `5m`
- default: `30s`
- Poll wait duration comes from `async.pollWaitDuration` ("Max Poll Time") in `packages/coding-agent/src/config/settings-schema.ts`:
- allowed values: `5s`, `10s`, `30s`, `1m`, `5m`, `smart`
- default: `smart`
- fixed values block for exactly that long; `smart` uses the adaptive ladder `POLL_WAIT_LADDER_MS = [5s, 10s, 30s, 1m, 5m]` in `packages/coding-agent/src/async/job-manager.ts`, climbing one rung per back-to-back poll and resetting to the 5s floor after `POLL_ESCALATION_RESET_MS = 60_000` ms without polling. Per-owner state is driven by `nextPollWaitMs(...)` / `recordPollWaitEnd(...)`.
- Progress update cadence while polling: `PROGRESS_INTERVAL_MS = 500` in `packages/coding-agent/src/tools/job.ts`.
- Async job retention default: `DEFAULT_RETENTION_MS = 5 * 60 * 1000` in `packages/coding-agent/src/async/job-manager.ts`.
- Manager fallback max-running limit: `DEFAULT_MAX_RUNNING_JOBS = 15` in `packages/coding-agent/src/async/job-manager.ts`.
+46 -18
View File
@@ -41,23 +41,34 @@ selection, transcript persists after exit). The engine maintains one ledger:
- **`windowTopRow` (W)** — the frame row mapped to grid row 0. The visible
window is frame rows `[W, W + height)`, repainted in place with relative
cursor moves.
- **commit boundary (B)** — reported by the component tree per frame
(`NativeScrollbackLiveRegion`): `B = commitSafeEnd ?? liveRegionStart ??
frame.length`. Rows below B may still re-layout and must not enter history.
- **commit boundary** — reported by the component tree per frame
(`NativeScrollbackLiveRegion`) as two nested ends:
- **byte-stable end (B)** — `commitSafeEnd ?? liveRegionStart ?? frame.length`.
Rows below B are asserted never to re-layout and stay under the
committed-prefix audit.
- **durable end (D)** — `max(B, snapshotSafeEnd ?? B)`. Rows in `[B, D)` may
still drift bytes later (a streaming markdown table re-aligning columns) but
are *durable* — their current snapshot is permanent content, so dropping them
when they scroll off is forbidden. They commit **audit-exempt**: later drift
becomes a frozen stale row in history, never a re-anchor.
Per ordinary frame: `W = max(C, L − height)`, `C' = max(C, min(B, W))`, and the
Per ordinary frame: `W = max(C, L − height)`, `C' = max(C, min(D, W))`, and the
only bytes that ever touch history are the **chunk** `frame[C, C')` written at
the scrollback seam. Scrollback therefore equals `frame[0..C)` — every row
exactly once, in order, with its content at commit time. There is nothing to
guess, nothing to defer, and nothing to reconcile: the scroll position is
irrelevant because ordinary updates never rewrite anything a scrolled reader
could be looking at.
the scrollback seam. The engine also tracks **`auditRows` (A ≤ C)** — the
byte-stable leading prefix `[0, A)`; the committed-prefix audit (§2) samples only
that prefix, so the durable suffix `[A, C)` drifting never triggers a re-anchor.
Scrollback therefore equals `frame[0..C)` — every row exactly once, in order,
with its content at commit time. There is nothing to guess, nothing to defer,
and nothing to reconcile: the scroll position is irrelevant because ordinary
updates never rewrite anything a scrolled reader could be looking at.
### What this costs (the accepted tradeoffs)
- A block that has scrolled past the window top cannot reflow in place. Blocks
stay in the live region (below B) until they are final; a late mutation of
committed content is ignored (the stale committed copy stays in history).
- A block that has scrolled past the window top cannot reflow in place. A
byte-stable block stays in the live region (below B) until final; a durable
block (below D) commits its scroll-off snapshot, so a late layout change of an
already-committed row is a frozen stale row in history (duplication never loss),
not a dropped row.
- A component tree that reports **no seam** gets shell semantics: whatever
scrolls off is final. Shrinking such a frame into its committed prefix
re-anchors the window and leaves the stale copy in history (§3).
@@ -122,17 +133,25 @@ of history:
- `getNativeScrollbackLiveRegionStart()` — first row that may still mutate
(everything below it, including root chrome rendered after it, stays in the
window).
- `getNativeScrollbackCommitSafeEnd()` — optional deeper boundary: the
append-only prefix of the live region (a streaming assistant message's
settled rows). Without it, a single live block taller than the window would
hold its head out of history until it finalizes.
- `getNativeScrollbackCommitSafeEnd()` — optional **byte-stable** deeper boundary
(B): the append-only prefix of the live region (a streaming assistant message's
settled rows), asserted never to re-layout, so it stays under the audit.
- `getNativeScrollbackSnapshotSafeEnd()` — optional **durable** deeper boundary
(D ≥ B): rows whose current snapshot is permanent but may still drift bytes
(a streaming markdown table whose columns keep re-aligning). They commit on
scroll-off (never dropped) but **audit-exempt** — drift after commit freezes a
stale row in history rather than re-anchoring the audit and spraying duplicate
snapshots. Without it, a commit-stable block that perpetually re-lays-out an
interior row (a table taller than the window) had no byte-stable prefix past
the table head, so its scrolled-off rows were committed nowhere and repainted
nowhere — silent content loss as the reply streamed.
`TranscriptContainer` implements this for the coding agent: finalized blocks
freeze (their render is snapshotted, so their content can never drift after
the engine may have committed it), still-mutating blocks
(`isTranscriptBlockFinalized?.() === false`) anchor the live region, and
`deriveLiveCommitState` derives the commit-safe end of the first live block
from two independent signals:
`deriveLiveCommitState` derives the byte-stable commit-safe end of the first
live block from two independent signals:
- **append-only detection** — a block observed growing without visibly
rewriting an interior row commits its full body; a rewrite suspends this
@@ -156,6 +175,15 @@ from two independent signals:
one-off re-layouts before any promotion never arm it, and the append-only
path commits the full block regardless.
The byte-stable end gates audited commits; the **durable snapshot end** is the
separate floor that guarantees no loss. `TranscriptContainer` reports the whole
body of a still-live **commit-stable** block (`isTranscriptBlockCommitStable?.()
!== false`) as the snapshot-safe end, so its scrolled-off rows always reach
history even while its interior re-lays-out. Provisional blocks
(`isTranscriptBlockCommitStable?.() === false`: a collapsing tool/edit preview
whose head is a throwaway tail window) report no snapshot-safe end, so their
head is correctly dropped rather than stranded as stale history.
Freezing is unconditional — it is the engine's required guarantee, not a
per-terminal optimization.
+30 -16
View File
@@ -24,18 +24,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "15.12.5",
"@oh-my-pi/omp-stats": "15.12.5",
"@oh-my-pi/pi-agent-core": "15.12.5",
"@oh-my-pi/pi-ai": "15.12.5",
"@oh-my-pi/pi-catalog": "15.12.5",
"@oh-my-pi/pi-coding-agent": "15.12.5",
"@oh-my-pi/pi-mnemopi": "15.12.5",
"@oh-my-pi/pi-natives": "15.12.5",
"@oh-my-pi/pi-tui": "15.12.5",
"@oh-my-pi/pi-utils": "15.12.5",
"@oh-my-pi/pi-wire": "15.12.5",
"@oh-my-pi/snapcompact": "15.12.5",
"@oh-my-pi/hashline": "15.13.0",
"@oh-my-pi/omp-stats": "15.13.0",
"@oh-my-pi/pi-agent-core": "15.13.0",
"@oh-my-pi/pi-ai": "15.13.0",
"@oh-my-pi/pi-catalog": "15.13.0",
"@oh-my-pi/pi-coding-agent": "15.13.0",
"@oh-my-pi/pi-mnemopi": "15.13.0",
"@oh-my-pi/pi-natives": "15.13.0",
"@oh-my-pi/pi-tui": "15.13.0",
"@oh-my-pi/pi-utils": "15.13.0",
"@oh-my-pi/pi-wire": "15.13.0",
"@oh-my-pi/snapcompact": "15.13.0",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
@@ -103,7 +103,8 @@
"build": "bun run --workspaces --if-present build",
"build:native": "bun --cwd=packages/natives run build",
"test": "bun run --parallel test:ts test:rs",
"test:ts": "GITHUB_ACTIONS= bun run --workspaces --if-present test -- --only-failures",
"test:ts": "GITHUB_ACTIONS= bun run --workspaces --if-present test -- --only-failures && bun run test:scripts",
"test:scripts": "bun test scripts/ci-concurrency.test.ts",
"test:rs": "bun scripts/run-rs-task.ts test:rs",
"check": "bun run --parallel check:ts check:rs",
"check:ts": "bun run check:tools && bun run --workspaces --if-present check",
@@ -117,8 +118,8 @@
"fmt:ts": "bun run fmt:tools && bun run --workspaces --if-present fmt",
"fmt:tools": "biome format --write . --no-errors-on-unmatched",
"fmt:rs": "bun scripts/run-rs-task.ts fmt:rs",
"fix": "bun run --parallel fix:ts fix:rs",
"fix:all": "bun run --parallel fix:ts:all fix:rs",
"fix": "bun run --parallel fix:ts fix:rs fix:changelogs",
"fix:all": "bun run --parallel fix:ts:all fix:rs fix:changelogs",
"fix:ts": "bun run fix:tools && bun run --workspaces --if-present fix",
"fix:ts:all": "bun run fix:tools:all && bun run --workspaces --if-present fix",
"fix:tools": "biome check --write --unsafe --changed --no-errors-on-unmatched .",
@@ -127,7 +128,15 @@
"fix:rs": "bun scripts/run-rs-task.ts fix:rs",
"ci:check:full": "bun run check:ts",
"ci:build:native": "bun scripts/ci-build-native.ts",
"ci:test:full": "bun run test",
"ci:test:full": "bun run ci:test:ts && bun run test:rs",
"ci:test:ts": "bun scripts/ci-test-ts.ts all",
"ci:test:ts:workspace": "bun scripts/ci-test-ts.ts workspace",
"ci:test:ts:native": "bun scripts/ci-test-ts.ts native",
"ci:test:coding-agent:singleton": "bun scripts/ci-test-ts.ts coding-agent-singleton",
"ci:test:coding-agent:ui": "bun scripts/ci-test-ts.ts coding-agent-ui",
"ci:test:coding-agent:runtime": "bun scripts/ci-test-ts.ts coding-agent-runtime",
"ci:test:coding-agent:native": "bun scripts/ci-test-ts.ts coding-agent-native",
"ci:test:coding-agent:heavy": "bun scripts/ci-test-ts.ts coding-agent-heavy",
"ci:test:smoke": "bun packages/coding-agent/src/cli.ts --version && bun packages/coding-agent/src/cli.ts --help && bun packages/coding-agent/src/cli.ts stats --help && bun packages/coding-agent/src/cli.ts --smoke-test",
"ci:test:install-methods": "bash scripts/install-tests/run-ci.sh",
"ci:release:build-binaries": "bun scripts/ci-release-build-binaries.ts",
@@ -178,5 +187,10 @@
},
"lint-staged": {
"*.{js,ts,jsx,tsx,json,jsonc,css}": "biome check --write --no-errors-on-unmatched"
},
"dependencies": {
"sherpa-onnx": "1.12.37",
"sherpa-onnx-darwin-arm64": "1.12.37",
"sherpa-onnx-node": "1.12.37"
}
}
+82 -161
View File
@@ -2,111 +2,68 @@
## [Unreleased]
## [15.12.4] - 2026-06-13
### Fixed
- Fixed remote compaction input trimming to use unlimited context when `model.contextWindow` is unset
## [15.12.1] - 2026-06-12
### Breaking Changes
- Changed `pruneSupersededToolResults` to allow `supersedeKey` to be omitted so useless-result pruning can run without read-style supersede grouping
### Added
- Added `pruneUseless` controls to `PruneConfig` and `SupersedePruneConfig` so callers can toggle compaction of `toolResult` entries marked `useless`
- Added the ability to disable useless-result pruning by setting `pruneUseless` to false
- Tools can flag a result contextually useless (`AgentToolResult.useless`; overridable via `AfterToolCallResult.useless`): the agent loop copies the flag onto the persisted `ToolResultMessage` (errors always win), and compaction consumes it — the cache-aware supersede pass and the threshold prune blank flagged results to the exact `USELESS_NOTICE` placeholder (bypassing the protect window, skipping results smaller than the notice), shake collects them inside the protect-recent window, and `serializeConversation` drops the whole tool call/result pair from summarizer input
### Changed
- Changed `pruneSupersededToolResults` to allow omitted `supersedeKey` when `pruneUseless` is enabled, so useless-result pruning can run without read-style supersede grouping
## [15.11.4] - 2026-06-12
### Added
- Added `hasSteeringMessages` to `AgentLoopConfig` (wired by `Agent` to its steering queue): a peek used by the immediate-interrupt poll during tool execution, so the loop can detect queued steering without dequeuing and the queue keeps owning its messages until the injection boundary
- The agent loop now re-samples after a non-terminal stop (`stopReason: "stop"` with `stopDetails: { type: "pause_turn" }`, emitted by the Codex providers for `end_turn: false` commentary-only responses): the assistant message is committed to history and the model is called again without ending the turn. Consecutive pause continuations without an intervening tool call are capped at 8 to bound a backend that never stops pausing.
### Changed
- Changed steering handling so queued steering messages are now dequeued only at injection boundaries, with immediate mid-batch interrupt polling using `hasSteeringMessages`. Consumers constructing `AgentLoopConfig` directly with only `getSteeringMessages` no longer get mid-batch interrupts — steering degrades to boundary-only delivery until they also supply `hasSteeringMessages`
- Compaction, handoff, short-summary, and branch-summarization helpers now accept an `ApiKey` (static string or resolver) instead of a pre-resolved string, so a 401 mid-compaction force-refreshes and rotates the credential through the central auth-retry policy before any model-level fallback. The remote OpenAI compaction request is wrapped in `withAuth` and its HTTP failures now carry `.status`, so the retry classifier actually fires on remote-compaction 401s.
- `transformProviderContext` now receives the dispatch model as a second argument (`(context, model) => Context`), so per-request transforms can gate on model capabilities (vision input, provider, API family). Existing single-argument implementations keep working unchanged.
- Remote-compaction and summarization failures now throw pi-ai's typed `ProviderHttpError` instead of mutating plain `Error`s with a `.status` property; the generic `requestRemoteCompaction` error now carries `.status` (and response headers) too.
### Fixed
- Fixed a regression where steering messages could be injected into history during an aborted in-flight tool batch, leaving them hidden from queue consumers for post-abort continue
## [15.11.2] - 2026-06-11
### Added
- `AgentTool.concurrency` now also accepts a per-call resolver function `(args) => "shared" | "exclusive"`, letting tools pick the scheduling mode from the call's arguments (a throwing resolver falls back to `"exclusive"`)
### Fixed
- Fixed whitespace-only error tool results so Anthropic requests no longer 400 with `tool_result: content cannot be empty if is_error is true` and wedge the session on every subsequent turn
## [15.11.0] - 2026-06-10
### Breaking Changes
- Removed `compaction/index.ts` re-export of snapcompact helpers, so snapcompact utilities are no longer available from the agent compaction barrel and should be imported from `@oh-my-pi/snapcompact`
- Removed the `convertToLlm` alias export from `compaction/messages` — it duplicated `defaultConvertToLlm` under a second name. Import `defaultConvertToLlm` (array form) or the new `convertMessageToLlm` (single-message form) instead
### Added
- Added repetition-loop detection to the streaming agent loop for Gemini-family providers. A runaway run of a repeated text or thinking unit is detected mid-stream from a bounded rolling tail (O(1) per delta), the provider request is aborted, the repeated tail is collapsed to a single representative copy, and the turn ends gracefully with an `error` stop reason. Legitimate all-numeric/whitespace/punctuation runs (hexdumps, zero-fills, numeric tables) are not misclassified as loops ([#2549](https://github.com/can1357/oh-my-pi/pull/2549) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)).
- Added `pruneUseless` controls to `PruneConfig` and `SupersedePruneConfig` so callers can toggle compaction of `toolResult` entries marked `useless`
- Added the ability to disable useless-result pruning by setting `pruneUseless` to false
- Tools can flag a result contextually useless (`AgentToolResult.useless`; overridable via `AfterToolCallResult.useless`): the agent loop copies the flag onto the persisted `ToolResultMessage` (errors always win), and compaction consumes it — the cache-aware supersede pass and the threshold prune blank flagged results to the exact `USELESS_NOTICE` placeholder (bypassing the protect window, skipping results smaller than the notice), shake collects them inside the protect-recent window, and `serializeConversation` drops the whole tool call/result pair from summarizer input
- Added `hasSteeringMessages` to `AgentLoopConfig` (wired by `Agent` to its steering queue): a peek used by the immediate-interrupt poll during tool execution, so the loop can detect queued steering without dequeuing and the queue keeps owning its messages until the injection boundary
- The agent loop now re-samples after a non-terminal stop (`stopReason: "stop"` with `stopDetails: { type: "pause_turn" }`, emitted by the Codex providers for `end_turn: false` commentary-only responses): the assistant message is committed to history and the model is called again without ending the turn. Consecutive pause continuations without an intervening tool call are capped at 8 to bound a backend that never stops pausing.
- `AgentTool.concurrency` now also accepts a per-call resolver function `(args) => "shared" | "exclusive"`, letting tools pick the scheduling mode from the call's arguments (a throwing resolver falls back to `"exclusive"`)
- Added `convertMessageToLlm()`: the single-message core transformer behind `defaultConvertToLlm()`. Embedders with app-specific message roles should handle their own roles and delegate every core role (`user`/`developer`/`assistant`/`toolResult`/`custom`/`hookMessage`/`branchSummary`/`compactionSummary`) to it instead of duplicating the conversion — a duplicated `compactionSummary` case is how snapcompact frames once silently dropped off provider requests
- Added `pruneSupersededToolResults()` and the opt-in `PruneConfig.supersedeKey` hook so harnesses can prune stale tool results superseded by a newer read of the same file; superseded results are pruned ahead of age-based victims during overflow pruning and replaced with a `[Superseded by a newer read of this file]` placeholder. Without the new config, `pruneToolOutputs()` behavior is unchanged.
- Added `readToolSupersedeKey()` implementing the read-tool path/selector grammar (selector-free reads supersede range reads of the same file; URL-scheme paths exempt). Pruning honors prompt-cache economics: per-turn prunes only fire when the post-candidate suffix is small or the cache is cold (idle gap).
- Added the `snapcompact` compaction strategy via `@oh-my-pi/snapcompact`: instead of an LLM summary, discarded history is printed onto dense bitmap frames and re-attached to the compaction summary message as image blocks. `CompactionSummaryMessage` gains an optional `images` field, `estimateTokens()` charges per attached frame, and frames persist under `preserveData.snapcompact` with an 8-frame middle-out eviction budget.
- Snapcompact frames are now rendered in a provider-aware shape (`SNAPCOMPACT_SHAPES` + `resolveSnapcompactShape(api)`), following the snapcompact 200k-token monolithic evals: Anthropic-family and unknown APIs get `8x8r-bw` (unscii-8 square cells, black ink, every line printed twice with the copy on a pale highlight band — read at F1 parity with raw text at ~2x lower cost and the most refusal-robust), Google gets `8x8r-sent` (sentence-hue ink, ~2.9x cheaper), and OpenAI gets `6x6u-sent` (unscii Lanczos-stretched to 6x6 cells — OpenAI bills a flat ~2.9k tokens per image, so frame count is the only cost lever) with `detail: "original"` on the frame images. `snapcompactCompact()` accepts `model`/`shape` options, frames persist their shape metadata, mixed-shape archives (provider switches, legacy 5x8 frames) are flagged in the reading instructions, and `snapcompactGeometry()`/`renderSnapcompactFrame()` now take a shape
### Changed
- Compaction and branch-summary file lists are now a single `<files>` tag instead of `<read-files>`/`<modified-files>`: paths render as the grouped, prefix-folded directory tree the find/search tools emit (`# dir/` headers, bare basenames), each annotated `(Read)`, `(Write)`, or `(RW)` — modified files that were also read get `(RW)`. Legacy tags in summaries written by earlier versions are still stripped and self-heal on the next compaction
### Fixed
- Fixed queued steering messages being drained into an externally aborted run: interrupting mid-tool execution (e.g. Enter with a pending steer) dequeued the steer into the dying run — it landed in history without a response and the post-abort resume saw an empty queue, so the agent stopped instead of continuing. Steering/follow-up/aside queue polls are now skipped once the run's abort signal fires, leaving the queue intact for `Agent.continue()`.
- Fixed `<read-files>` compaction lists recording the same file once per line-range/raw selector (`src/foo.ts:50-200`, `:raw`, `:1-50:raw`, …): read-tool selectors are now stripped before tracking, so reads dedupe to the base path and match their write/edit path when splitting read-only vs modified lists. Selector-polluted lists stored by earlier compactions self-heal on the next compaction. `readToolSupersedeKey()` now shares the same splitter (`splitReadSelector()`), gaining the `..` range alias and `L`-prefix forms it previously missed.
- Fixed `estimateTokens()` undercounting thinking-heavy assistant messages on replay: `thinkingSignature` payloads (OpenAI Responses encrypted reasoning items, Anthropic signed thinking blocks, etc.) and `redactedThinking.data` are now charged alongside the visible thinking text, so the local estimate tracks provider-reported usage instead of straddling the threshold on every turn ([#2275](https://github.com/can1357/oh-my-pi/issues/2275)).
## [15.10.12] - 2026-06-10
### Added
- Added `AgentLoopConfig.getDisableReasoning` so callers can override `disableReasoning` per LLM call, mirroring `getReasoning`.
- Added `transformProviderContext` to `AgentOptions`/`AgentLoopConfig`: an optional hook applied to the assembled provider context after conversion, normalization, and append-only handling, but before telemetry capture and provider send.
### Fixed
- Fixed `Agent` runs so explicit reasoning disablement is forwarded to provider stream options and re-resolved per continuation, keeping mid-run thinking-off changes in sync with the next provider request.
## [15.10.11] - 2026-06-10
### Changed
- Editorial pass over the compaction prompts: fixed garbled grammar and missing articles, RFC-keyed prohibitions, deduped restated instructions; parsed markers (`<read-files>`/`<modified-files>`/`<previous-summary>`) and all output-format headings left byte-identical
- Catalog imports moved to the new `@oh-my-pi/pi-catalog` package: subpath imports (`calculateCost`, Codex wire constants) plus catalog values previously taken from the `@oh-my-pi/pi-ai` root (`getBundledModel`, `clampThinkingLevelForModel`), which pi-ai no longer re-exports; type-only `Model`/`Api`/`Effort` imports from pi-ai are unchanged
## [15.10.8] - 2026-06-09
### Added
- Added optional `fetch` overrides to `SummaryOptions` and `compact`/`generateSummary` so remote compaction can use custom HTTP clients
- Added optional `fetch` option to `ProxyStreamOptions` to control the HTTP request used by `streamProxy`
- Added optional `fetch` overrides to `requestOpenAiRemoteCompaction` and `requestRemoteCompaction` for injectable HTTP transport
- Added the upstream provider that served a request (`AssistantMessage.upstreamProvider`, e.g. OpenRouter's routed provider) as a `pi.gen_ai.response.upstream_provider` chat-span telemetry attribute, alongside the existing response id and time-to-first-chunk.
- Added a non-interrupting "aside" message channel to the agent loop (`AgentLoopConfig.getAsideMessages` / `Agent.setAsideMessageProvider`). Asides are drained at each step boundary (after a tool batch, before the next model call) and at the yield check, so passive notifications (e.g. background-job completions, late LSP diagnostics) reach the model *between requests* without waiting for the agent to stop and without aborting in-flight tools the way steering does.
- Added optional `promptCacheKey` support to `AgentOptions` and `Agent` via a new `promptCacheKey` property so providers can receive a caller-provided prompt cache key
- Added optional `ApiKeyResolveContext` parameter to `getApiKey` in `AgentOptions` and `AgentLoopConfig` so key resolvers can receive retry context
- Added `getReadToolPath(context)` to `@oh-my-pi/pi-agent-core/compaction/tool-protection` to extract a paired `read` tool call's `path` for embedders building read-targeted protection matchers
- Added `getReadToolPath(context)` to `@oh-my-pi/pi-agent-core/compaction/tool-protection`: the shared primitive that extracts a paired `read` tool call's `path` argument, so embedders can build their own read-targeted compaction protection matchers (e.g. plan-file reads) the same way `isSkillReadToolResult` does.
- Added optional `AgentTool.matcherDigest(args)` hook: tools whose streamed arguments encode content in a wire grammar (patch formats, escaped strings) can expose the real content they introduce, so stream-content matchers (e.g. TTSR rules) run against plain source text instead of the wire format.
- Added `shake` compaction primitives (`collectShakeRegions`, `applyShakeRegion`, `applyShakeRegions`, `summarizeShakeRegions`, `DEFAULT_SHAKE_CONFIG`, `AGGRESSIVE_SHAKE_CONFIG`, plus the `ShakeRegion`/`ShakeConfig`/`ShakeSummaryItem`/`ShakeSummaryComplete`/`ProtectedToolMatcher` types) under `@oh-my-pi/pi-agent-core/compaction`. These detect heavy context regions — whole tool-call results plus large fenced/XML blocks — and either elide them with placeholders or extractively compress them through an injected completion backend (no LLM summary cut-point). The compressor is provider-agnostic: callers wire it to a local on-device model. Pure detection/mutation; no I/O.
## [15.10.5] - 2026-06-08
### Changed
### Removed
- Removed the `maxToolCallsPerTurn` option from `AgentOptions` and `AgentLoopConfig`, so assistant turns are no longer capped after a configured number of completed tool calls
- Changed `pruneSupersededToolResults` to allow omitted `supersedeKey` when `pruneUseless` is enabled, so useless-result pruning can run without read-style supersede grouping
- Changed steering handling so queued steering messages are now dequeued only at injection boundaries, with immediate mid-batch interrupt polling using `hasSteeringMessages`. Consumers constructing `AgentLoopConfig` directly with only `getSteeringMessages` no longer get mid-batch interrupts — steering degrades to boundary-only delivery until they also supply `hasSteeringMessages`
- Compaction, handoff, short-summary, and branch-summarization helpers now accept an `ApiKey` (static string or resolver) instead of a pre-resolved string, so a 401 mid-compaction force-refreshes and rotates the credential through the central auth-retry policy before any model-level fallback. The remote OpenAI compaction request is wrapped in `withAuth` and its HTTP failures now carry `.status`, so the retry classifier actually fires on remote-compaction 401s.
- `transformProviderContext` now receives the dispatch model as a second argument (`(context, model) => Context`), so per-request transforms can gate on model capabilities (vision input, provider, API family). Existing single-argument implementations keep working unchanged.
- Remote-compaction and summarization failures now throw pi-ai's typed `ProviderHttpError` instead of mutating plain `Error`s with a `.status` property; the generic `requestRemoteCompaction` error now carries `.status` (and response headers) too.
- Compaction and branch-summary file lists are now a single `<files>` tag instead of `<read-files>`/`<modified-files>`: paths render as the grouped, prefix-folded directory tree the find/search tools emit (`# dir/` headers, bare basenames), each annotated `(Read)`, `(Write)`, or `(RW)` — modified files that were also read get `(RW)`. Legacy tags in summaries written by earlier versions are still stripped and self-heal on the next compaction
- Editorial pass over the compaction prompts: fixed garbled grammar and missing articles, RFC-keyed prohibitions, deduped restated instructions; parsed markers (`<read-files>`/`<modified-files>`/`<previous-summary>`) and all output-format headings left byte-identical
- Catalog imports moved to the new `@oh-my-pi/pi-catalog` package: subpath imports (`calculateCost`, Codex wire constants) plus catalog values previously taken from the `@oh-my-pi/pi-ai` root (`getBundledModel`, `clampThinkingLevelForModel`), which pi-ai no longer re-exports; type-only `Model`/`Api`/`Effort` imports from pi-ai are unchanged
- Changed core custom and hook messages to convert to `developer` messages for provider context.
- Enabled streaming API calls to re-resolve credentials through the `getApiKey` callback when retries occur after authentication-related errors
- `Agent.abort(reason?)` now forwards `reason` to the underlying `AbortController`, and the synthesized aborted assistant message carries that reason on `errorMessage` (string or non-`AbortError` `Error` message) instead of always defaulting to `"Request was aborted"`. Bare `abort()` is unchanged.
- Changed `Agent.appendMessage`, `popMessage`, `clearMessages`, and `reset` to mutate `state.messages` and `state.pendingToolCalls` in place instead of allocating a fresh array/Set on every transition. Subscribers that capture `state.messages` by reference now observe updates without needing to re-read `state` after each event. The public type signature is unchanged (always `AgentMessage[]` / `Set<string>`).
### Fixed
- Fixed repetition loop handling to collapse repeated `thinking` blocks to a single representative copy when a loop is detected
- Fixed repetition-loop detection to ignore repeats that contain only digits, whitespace, or punctuation so legitimate numeric outputs no longer stop with a repetition-loop error
- Fixed false-positive repetition-loop checks across `text` and `thinking` stream boundaries by tracking loop detection per block type
- Fixed dynamic forced tool choices from queue hooks being filtered against the active per-turn tool set before provider dispatch. ([#1701](https://github.com/can1357/oh-my-pi/issues/1701))
- Fixed remote compaction input trimming to use unlimited context when `model.contextWindow` is unset
- Fixed a regression where steering messages could be injected into history during an aborted in-flight tool batch, leaving them hidden from queue consumers for post-abort continue
- Fixed whitespace-only error tool results so Anthropic requests no longer 400 with `tool_result: content cannot be empty if is_error is true` and wedge the session on every subsequent turn
- Fixed queued steering messages being drained into an externally aborted run: interrupting mid-tool execution (e.g. Enter with a pending steer) dequeued the steer into the dying run — it landed in history without a response and the post-abort resume saw an empty queue, so the agent stopped instead of continuing. Steering/follow-up/aside queue polls are now skipped once the run's abort signal fires, leaving the queue intact for `Agent.continue()`.
- Fixed `<read-files>` compaction lists recording the same file once per line-range/raw selector (`src/foo.ts:50-200`, `:raw`, `:1-50:raw`, …): read-tool selectors are now stripped before tracking, so reads dedupe to the base path and match their write/edit path when splitting read-only vs modified lists. Selector-polluted lists stored by earlier compactions self-heal on the next compaction. `readToolSupersedeKey()` now shares the same splitter (`splitReadSelector()`), gaining the `..` range alias and `L`-prefix forms it previously missed.
- Fixed `estimateTokens()` undercounting thinking-heavy assistant messages on replay: `thinkingSignature` payloads (OpenAI Responses encrypted reasoning items, Anthropic signed thinking blocks, etc.) and `redactedThinking.data` are now charged alongside the visible thinking text, so the local estimate tracks provider-reported usage instead of straddling the threshold on every turn ([#2275](https://github.com/can1357/oh-my-pi/issues/2275)).
- Fixed `Agent` runs so explicit reasoning disablement is forwarded to provider stream options and re-resolved per continuation, keeping mid-run thinking-off changes in sync with the next provider request.
- Fixed stalled aborted assistant responses so the run now stops without waiting for provider iterator cleanup and returns the aborted message promptly
- Fixed `afterToolCall` handling so it now runs for completed tool executions even after a run is aborted so tool post-processing still applies
- Fixed `agentLoopDetailed().detailed()` so run telemetry and coverage are captured before `stream.result()` resolves.
@@ -117,96 +74,64 @@
- Fixed tool-call completion so assistant messages on abort keep only completed tool-call blocks and continue processing tool calls when a length stop still included results
- Fixed deliberate aborts (TTSR rule matches, user-interrupt labels) so a mid-stream tool-call block that never reached `toolcall_end` is retained on the aborted assistant message and paired with a placeholder result labeled by the abort reason, instead of being dropped; anonymous aborts (bare `abort()`) still drop incomplete tool calls whose partial arguments are unsafe to replay
- Fixed runs that stopped with reason `length` after returning tool results so execution continues to handle additional tool calls
## [15.10.3] - 2026-06-08
### Added
- Added a non-interrupting "aside" message channel to the agent loop (`AgentLoopConfig.getAsideMessages` / `Agent.setAsideMessageProvider`). Asides are drained at each step boundary (after a tool batch, before the next model call) and at the yield check, so passive notifications (e.g. background-job completions, late LSP diagnostics) reach the model *between requests* without waiting for the agent to stop and without aborting in-flight tools the way steering does.
### Changed
- Changed core custom and hook messages to convert to `developer` messages for provider context.
### Fixed
- Fixed the compaction spinner freezing (only repainting on a terminal resize) when compacting very large codex/OpenAI contexts. `buildOpenAiNativeHistory` re-collected the full known/custom tool-call id sets on every history-bearing message, rescanning the entire growing native history each time — O(N²) in history items — which blocked the event loop for seconds and starved the loader's animation timer and render scheduler. The sets are now maintained incrementally (linear), so building the compaction request no longer monopolizes the main thread.
### Removed
- Removed the now-dead `<turn-aborted>` marker from the OpenAI compaction output user-message filter, since `transformMessages` no longer emits that note.
- Removed stale synthetic user-message tag filters from OpenAI remote compaction output preservation; developer messages are now dropped by role instead.
- Tool executions now receive the active turn `AbortSignal` unconditionally.
## [15.10.2] - 2026-06-08
### Fixed
- Fixed proxy stream silently returning a zero-token success response when the server disconnects without sending a `done` or `error` terminal SSE event. The stream now throws an error, surfacing the disconnect as an `error` event with `stopReason: "error"` and resolving `finalResultPromise`, instead of defaulting to `stopReason: "stop"` with empty content and leaving `stream.result()` callers hanging indefinitely.
## [15.10.1] - 2026-06-07
### Added
- Added optional `promptCacheKey` support to `AgentOptions` and `Agent` via a new `promptCacheKey` property so providers can receive a caller-provided prompt cache key
- Added optional `ApiKeyResolveContext` parameter to `getApiKey` in `AgentOptions` and `AgentLoopConfig` so key resolvers can receive retry context
### Changed
- Enabled streaming API calls to re-resolve credentials through the `getApiKey` callback when retries occur after authentication-related errors
- `Agent.abort(reason?)` now forwards `reason` to the underlying `AbortController`, and the synthesized aborted assistant message carries that reason on `errorMessage` (string or non-`AbortError` `Error` message) instead of always defaulting to `"Request was aborted"`. Bare `abort()` is unchanged.
### Fixed
- Fixed handling of short-lived API keys so that expired tokens are retried with a refreshed value during 401/usage-limit failures
- Ensured fallback API key resolution uses the initially configured static `apiKey` when `getApiKey` is present
- Wrapped oneshot LLM completions (`instrumentedCompleteSimple`: handoff, compaction/branch summaries) in an `EventLoopKeepalive`. These run outside the agent `#runLoop`, so without the keepalive Bun's event loop stopped servicing timers while parked on the completion promise — freezing host spinners (e.g. the `/handoff` loader) until an unrelated terminal resize poked the loop into rendering again.
## [15.9.5] - 2026-06-05
### Fixed
- Surfaced Anthropic stream failures whose message starts with `Output blocked by conten` as normal assistant error lifecycle events, so interactive clients render content-filter blocks instead of silently dropping the streaming bubble at `agent_end`.
## [15.8.3] - 2026-06-03
### Added
- Added `getReadToolPath(context)` to `@oh-my-pi/pi-agent-core/compaction/tool-protection` to extract a paired `read` tool call's `path` for embedders building read-targeted protection matchers
- Added `getReadToolPath(context)` to `@oh-my-pi/pi-agent-core/compaction/tool-protection`: the shared primitive that extracts a paired `read` tool call's `path` argument, so embedders can build their own read-targeted compaction protection matchers (e.g. plan-file reads) the same way `isSkillReadToolResult` does.
## [15.8.2] - 2026-06-03
### Added
- Added optional `AgentTool.matcherDigest(args)` hook: tools whose streamed arguments encode content in a wire grammar (patch formats, escaped strings) can expose the real content they introduce, so stream-content matchers (e.g. TTSR rules) run against plain source text instead of the wire format.
### Fixed
- Fixed the agent loop wedging the model when a `write`/`edit` tool call is truncated by `stop_reason: length` (e.g. an OpenCode Zen / Claude-3.5-Haiku turn that emits >~1000 lines of code, blowing past the 8K `max_tokens` output cap). The skipped tool result now surfaces an actionable hint — naming `stop_reason: length` and telling the model to split the payload into multiple smaller calls — instead of the generic "Tool call was not executed because the assistant ended its turn" placeholder, which left the auto-continue loop re-emitting the same oversized payload until the user gave up. Tools are still NOT executed when the arguments are truncated. ([#1785](https://github.com/can1357/oh-my-pi/issues/1785))
## [15.8.0] - 2026-06-02
### Fixed
- Engaged GPT-5 Harmony leak detection on the committed assistant message (openai-codex only). `detectHarmonyLeakInAssistantMessage` now runs on the streamed `done`/`error` result and the trailing fallback, so a leaked final response is aborted-and-retried by the existing mitigation instead of being committed as-is. Tool-argument (`tool_arg`) scanning is gated on the trailing-garbage `T` co-signal and only fires when a caller supplies a parse boundary via `detectHarmonyLeakInAssistantMessage`'s new optional `toolArgParseEnd` resolver. The agent loop passes none — it cannot bound a streamed tool DSL — so that surface stays inert and a legitimate codex tool call whose content legitimately carries `to=functions.*` next to a channel word or non-Latin script (e.g. editing the harmony fixtures) is never hard-aborted.
## [15.7.4] - 2026-05-31
- Fixed tool-output pruning and shake protection for `read`: ordinary file/URL reads are now eligible for compaction, while `read` calls whose `path` starts with `skill://` remain protected like native `skill` results.
### Removed
- Removed the `maxToolCallsPerTurn` option from `AgentOptions` and `AgentLoopConfig`, so assistant turns are no longer capped after a configured number of completed tool calls
- Removed the now-dead `<turn-aborted>` marker from the OpenAI compaction output user-message filter, since `transformMessages` no longer emits that note.
- Removed stale synthetic user-message tag filters from OpenAI remote compaction output preservation; developer messages are now dropped by role instead.
- Tool executions now receive the active turn `AbortSignal` unconditionally.
- Removed the local-model `summarizeShakeRegions` compressor and related shake-summary prompt/types; shake now only provides mechanical artifact-backed elision primitives.
## [15.13.0] - 2026-06-14
## [15.12.6] - 2026-06-14
## [15.12.4] - 2026-06-13
## [15.12.1] - 2026-06-12
## [15.11.4] - 2026-06-12
## [15.11.2] - 2026-06-11
## [15.11.0] - 2026-06-10
## [15.10.12] - 2026-06-10
## [15.10.11] - 2026-06-10
## [15.10.8] - 2026-06-09
## [15.10.5] - 2026-06-08
## [15.10.3] - 2026-06-08
## [15.10.2] - 2026-06-08
## [15.10.1] - 2026-06-07
## [15.9.5] - 2026-06-05
## [15.8.3] - 2026-06-03
## [15.8.2] - 2026-06-03
## [15.8.0] - 2026-06-02
## [15.7.4] - 2026-05-31
## [15.7.3] - 2026-05-31
### Added
- Added `shake` compaction primitives (`collectShakeRegions`, `applyShakeRegion`, `applyShakeRegions`, `summarizeShakeRegions`, `DEFAULT_SHAKE_CONFIG`, `AGGRESSIVE_SHAKE_CONFIG`, plus the `ShakeRegion`/`ShakeConfig`/`ShakeSummaryItem`/`ShakeSummaryComplete`/`ProtectedToolMatcher` types) under `@oh-my-pi/pi-agent-core/compaction`. These detect heavy context regions — whole tool-call results plus large fenced/XML blocks — and either elide them with placeholders or extractively compress them through an injected completion backend (no LLM summary cut-point). The compressor is provider-agnostic: callers wire it to a local on-device model. Pure detection/mutation; no I/O.
### Fixed
- Fixed tool-output pruning and shake protection for `read`: ordinary file/URL reads are now eligible for compaction, while `read` calls whose `path` starts with `skill://` remain protected like native `skill` results.
## [15.5.15] - 2026-05-30
### Added
@@ -229,10 +154,6 @@
- Fixed compaction summarizer throws losing the provider's HTTP status. `generateSummary`, `generateHandoff`, `generateShortSummary`, and `generateTurnPrefixSummary` now route their `stopReason === "error"` throws through a `createSummarizationError` helper that copies `AssistantMessage.errorStatus` onto the thrown `Error` as `.status`, letting downstream consumers (e.g. `AgentSession.#isCompactionAuthFailure` in `@oh-my-pi/pi-coding-agent`) branch on real provider 401/403s without regex-scraping the message body.
### Changed
- Changed `Agent.appendMessage`, `popMessage`, `clearMessages`, and `reset` to mutate `state.messages` and `state.pendingToolCalls` in place instead of allocating a fresh array/Set on every transition. Subscribers that capture `state.messages` by reference now observe updates without needing to re-read `state` after each event. The public type signature is unchanged (always `AgentMessage[]` / `Set<string>`).
## [15.5.0] - 2026-05-26
### Added
@@ -734,4 +655,4 @@ Initial release under @oh-my-pi scope. See previous releases at [badlogic/pi-mon
- `Agent` constructor now has all options optional (empty options use defaults).
- `queueMessage()` is now synchronous (no longer returns a Promise).
- `queueMessage()` is now synchronous (no longer returns a Promise).
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "15.12.5",
"version": "15.13.0",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+180 -2
View File
@@ -15,7 +15,7 @@ import {
validateToolArguments,
zodToWireSchema,
} from "@oh-my-pi/pi-ai";
import { sanitizeText } from "@oh-my-pi/pi-utils";
import { logger, sanitizeText } from "@oh-my-pi/pi-utils";
import {
createHarmonyAuditEvent,
detectHarmonyLeakInAssistantMessage,
@@ -708,6 +708,7 @@ async function runLoopBody(
});
}
stream.push({ type: "turn_end", message, toolResults });
stream.push(buildAgentEndEvent(newMessages, telemetry, stepCounter.count));
stream.end(newMessages);
return;
@@ -917,6 +918,10 @@ async function streamAssistantResponse(
? AbortSignal.any([signal, harmonyAbortController.signal])
: harmonyAbortController.signal
: signal;
const repetitionAbortController = new AbortController();
const finalRequestSignal = requestSignal
? AbortSignal.any([requestSignal, repetitionAbortController.signal])
: repetitionAbortController.signal;
const effectiveTemperature =
harmonyRetryAttempt > 0 && config.temperature !== undefined ? config.temperature + 0.05 : config.temperature;
const effectiveToolChoice = dynamicToolChoice ?? config.toolChoice;
@@ -984,7 +989,7 @@ async function streamAssistantResponse(
reasoning: effectiveReasoning,
disableReasoning: effectiveDisableReasoning,
temperature: effectiveTemperature,
signal: requestSignal,
signal: finalRequestSignal,
onResponse: captureOnResponse,
});
@@ -1013,6 +1018,56 @@ async function streamAssistantResponse(
return aborted;
};
const finishRepetitionStream = async (
kind: "text" | "thinking",
pattern: string,
count: number,
): Promise<AssistantMessage> => {
repetitionAbortController.abort();
try {
const cleanup = responseIterator.return?.();
if (cleanup) void cleanup.catch(() => {});
} catch {
// ignore
}
if (partialMessage) {
truncateRepetition(partialMessage, kind, pattern);
partialMessage.stopReason = "error";
partialMessage.errorMessage = `Repetition loop detected: assistant repeated "${pattern.trim()}" ${count} times consecutively.`;
}
const finalMsg = snapshotAssistantMessage(
partialMessage ?? {
role: "assistant",
content: [],
api: config.model.api,
provider: config.model.provider,
model: config.model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "error",
errorMessage: `Repetition loop detected.`,
timestamp: Date.now(),
},
);
if (addedPartial) {
context.messages[context.messages.length - 1] = finalMsg;
} else {
context.messages.push(finalMsg);
}
if (!addedPartial) {
stream.push({ type: "message_start", message: snapshotAssistantMessage(finalMsg) });
}
stream.push({ type: "message_end", message: snapshotAssistantMessage(finalMsg) });
await finishChat(finalMsg);
return finalMsg;
};
// Set up a single abort race: register the abort listener once for the whole
// stream and reuse the same race promise for every iterator.next() instead of
// allocating Promise.withResolvers and add/removeEventListener per event.
@@ -1029,6 +1084,14 @@ async function streamAssistantResponse(
detachAbortListener = () => requestSignal.removeEventListener("abort", onAbort);
}
// Rolling tail of streamed text/thinking used for repetition-loop detection.
// Bounded to REPETITION_WINDOW chars and reset when the active block kind
// switches (text <-> thinking) so detection stays O(1) per delta and never
// miscounts a repeated unit across a thinking/answer boundary.
let repetitionTail = "";
let repetitionKind: "text" | "thinking" | undefined;
const isGeminiModel = config.model.provider.includes("google") || config.model.provider.includes("gemini");
try {
while (true) {
let next: IteratorResult<AssistantMessageEvent>;
@@ -1113,6 +1176,27 @@ async function streamAssistantResponse(
assistantMessageEvent: snapshotAssistantMessageEvent(event),
message: snapshotAssistantMessage(partialMessage),
});
if (isGeminiModel && (event.type === "text_delta" || event.type === "thinking_delta")) {
const kind = event.type === "text_delta" ? "text" : "thinking";
if (repetitionKind !== kind) {
repetitionKind = kind;
repetitionTail = "";
}
repetitionTail += event.delta;
if (repetitionTail.length > REPETITION_WINDOW) {
repetitionTail = repetitionTail.slice(-REPETITION_WINDOW);
}
const repetition = detectRepetition(repetitionTail);
if (repetition) {
const [pattern, count] = repetition;
logger.warn("Repetition loop detected during assistant stream, aborting.", {
pattern,
count,
});
return await finishRepetitionStream(kind, pattern, count);
}
}
}
break;
}
@@ -1719,3 +1803,97 @@ function createSkippedToolResult(): AgentToolResult<any> {
details: {},
};
}
const REPETITION_WINDOW = 250;
const REPETITION_MIN_REPEATED_CHARS = 180;
function detectRepetition(text: string): [pattern: string, count: number] | null {
if (text.length < REPETITION_MIN_REPEATED_CHARS) return null;
const windowSize = Math.min(text.length, REPETITION_WINDOW);
const searchSpace = text.slice(-windowSize);
for (let len = 2; len <= 60; len++) {
if (searchSpace.length < len * 4) continue;
const pattern = searchSpace.slice(-len);
// Only treat a repeated unit as a pathological loop when it carries real
// linguistic content (a letter or a pictographic emoji). Runs made purely of
// digits, whitespace or punctuation are legitimate in tabular / hex / numeric
// output (e.g. "00 00 00", "0, 0, 0", "| -- | -- |") and must not trip.
if (!/[\p{L}\p{Extended_Pictographic}]/u.test(pattern)) continue;
let count = 0;
let pos = searchSpace.length;
while (pos >= len) {
const chunk = searchSpace.slice(pos - len, pos);
if (chunk === pattern) {
count++;
pos -= len;
} else {
break;
}
}
if (count >= 4 && len * count >= REPETITION_MIN_REPEATED_CHARS) {
return [pattern, count];
}
}
return null;
}
function truncateRepetition(message: AssistantMessage, kind: "text" | "thinking", pattern: string): void {
// A repetition loop streams into a single growing block (real providers) or a run
// of same-kind blocks (some transports), always at the tail of the message. Gather
// that trailing contiguous run and collapse its repeated copies down to one, so the
// committed transcript keeps a representative sample instead of the full runaway.
const matches = (block: AssistantContentBlock): boolean =>
kind === "text" ? block.type === "text" : block.type === "thinking";
const readBlock = (block: AssistantContentBlock): string =>
block.type === "text" ? block.text : block.type === "thinking" ? block.thinking : "";
const clearThinkingReplayAnchors = (block: AssistantContentBlock): void => {
if (block.type !== "thinking") return;
block.thinkingSignature = undefined;
block.itemId = undefined;
};
const writeBlock = (block: AssistantContentBlock, value: string): void => {
if (block.type === "text") {
block.text = value;
} else if (block.type === "thinking") {
block.thinking = value;
clearThinkingReplayAnchors(block);
}
};
const trailing: AssistantContentBlock[] = [];
for (let i = message.content.length - 1; i >= 0; i--) {
const block = message.content[i];
if (!matches(block)) break;
trailing.unshift(block);
}
if (trailing.length === 0) return;
if (kind === "thinking") {
for (const block of trailing) clearThinkingReplayAnchors(block);
}
let joined = "";
for (const block of trailing) joined += readBlock(block);
let kept = joined;
while (kept.length >= pattern.length * 2 && kept.slice(kept.length - pattern.length * 2) === pattern + pattern) {
kept = kept.slice(0, kept.length - pattern.length);
}
let remainingToRemove = joined.length - kept.length;
for (let i = trailing.length - 1; i >= 0 && remainingToRemove > 0; i--) {
const block = trailing[i];
const value = readBlock(block);
if (value.length <= remainingToRemove) {
remainingToRemove -= value.length;
writeBlock(block, "");
} else {
writeBlock(block, value.slice(0, value.length - remainingToRemove));
remainingToRemove = 0;
}
}
}
+9 -4
View File
@@ -657,8 +657,8 @@ export class Agent {
}
// State mutators
setSystemPrompt(v: string[]) {
this.#state.systemPrompt = v;
setSystemPrompt(v: string[] | string) {
this.#state.systemPrompt = typeof v === "string" ? [v] : v;
}
setModel(m: Model) {
@@ -974,8 +974,13 @@ export class Agent {
}
: undefined;
const getToolChoice = () =>
this.#getToolChoice?.() ?? refreshToolChoiceForActiveTools(options?.toolChoice, this.#state.tools);
const getToolChoice = () => {
const queuedToolChoice = this.#getToolChoice?.();
if (queuedToolChoice !== undefined) {
return refreshToolChoiceForActiveTools(queuedToolChoice, this.#state.tools);
}
return refreshToolChoiceForActiveTools(options?.toolChoice, this.#state.tools);
};
const config: AgentLoopConfig = {
model,
+121
View File
@@ -1710,4 +1710,125 @@ describe("agentLoopContinue with AgentMessage", () => {
expect(toolEnd.result.content).toEqual([{ type: "text", text: "Tool failed with no output." }]);
}
});
it("should detect repetition loops during assistant stream and abort gracefully", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [
{
content: Array.from({ length: 80 }, () => "🌊 "),
},
],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const messages = await stream.result();
expect(messages.length).toBe(2);
expect(messages[1].role).toBe("assistant");
const assistantMsg = messages[1] as AssistantMessage;
expect(assistantMsg.stopReason).toBe("error");
expect(assistantMsg.errorMessage).toContain("Repetition loop detected");
let text = "";
for (const block of assistantMsg.content) {
if (block.type === "text") text += block.text;
}
expect(text).toBe("🌊 ");
});
it("detects and truncates repetition loops inside a thinking stream", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [
{
content: Array.from({ length: 80 }, (_, index) => ({
type: "thinking" as const,
thinking: "🌊 ",
thinkingSignature: `signature-${index}`,
itemId: `rs_${index}`,
})),
},
],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("Hello")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const assistantMsg = (await stream.result())[1] as AssistantMessage;
expect(assistantMsg.stopReason).toBe("error");
expect(assistantMsg.errorMessage).toContain("Repetition loop detected");
// A looping thinking stream must be both detected AND collapsed to a single
// representative copy — not committed to the transcript in full.
let thinking = "";
for (const block of assistantMsg.content) {
if (block.type === "thinking") thinking += block.thinking;
}
expect(thinking).toBe("🌊 ");
for (const block of assistantMsg.content) {
if (block.type === "thinking") {
expect(block.thinkingSignature).toBeUndefined();
expect(block.itemId).toBeUndefined();
}
}
});
it("does not flag short requested repetitive text as a loop", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
const repeated = "🌊 ".repeat(26);
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [{ content: [repeated] }],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("print 26 wave emoji")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const assistantMsg = (await stream.result())[1] as AssistantMessage;
expect(assistantMsg.stopReason).not.toBe("error");
let text = "";
for (const block of assistantMsg.content) {
if (block.type === "text") text += block.text;
}
expect(text).toBe(repeated);
});
it("does not flag legitimate repetitive numeric output as a loop", async () => {
const context: AgentContext = { systemPrompt: ["You are helpful."], messages: [], tools: [] };
// A hexdump of zero-filled memory is highly repetitive but legitimate; the
// detector must not classify pure digit/whitespace runs as a loop.
const hexdump = "00 ".repeat(80);
const mock = createMockModel({
provider: "google-gemini-cli",
responses: [{ content: [hexdump] }],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const stream = agentLoop([createUserMessage("dump the zero page")], context, config, undefined, mock.stream);
for await (const _event of stream) {
// drain the stream to completion
}
const assistantMsg = (await stream.result())[1] as AssistantMessage;
expect(assistantMsg.stopReason).not.toBe("error");
let text = "";
for (const block of assistantMsg.content) {
if (block.type === "text") text += block.text;
}
expect(text).toBe(hexdump);
});
});
+32
View File
@@ -341,6 +341,38 @@ describe("Agent", () => {
]);
});
it("drops queued forced toolChoice when the queued tool is not active", async () => {
const toolSchema = z.object({ value: z.string() });
type Details = { value: string };
const betaTool: AgentTool<typeof toolSchema, Details> = {
name: "beta",
label: "Beta",
description: "Beta tool",
parameters: toolSchema,
async execute(_toolCallId, params) {
return { content: [{ type: "text", text: `beta:${params.value}` }], details: { value: params.value } };
},
};
const mock = createMockModel({ responses: [{ content: ["done"] }] });
const agent = new Agent({
initialState: {
model: mock.model,
tools: [betaTool],
messages: [],
},
streamFn: mock.stream,
getToolChoice: () => ({ type: "function", name: "alpha" }),
});
await agent.prompt("refresh tools");
expect(mock.calls).toHaveLength(1);
expect(mock.calls[0]?.context.tools?.map(tool => tool.name)).toEqual(["beta"]);
expect(mock.calls[0]?.options?.toolChoice).toBeUndefined();
});
it("re-reads thinking level for each model call within a run", async () => {
const toolSchema = z.object({ value: z.string() });
type Details = { value: string };
+26 -2
View File
@@ -2,6 +2,31 @@
## [Unreleased]
## [15.13.0] - 2026-06-14
### Fixed
- Fixed OpenAI Responses/Realtime SSE stream handler crashing with "Error Code undefined: undefined" when parsing error events with nested error details by falling back to the nested error object fields.
- Fixed OpenAI-compatible providers that reject forced `tool_choice` on thinking-required models by downgrading unsupported forced choices to `auto` while keeping tools available ([#2546](https://github.com/can1357/oh-my-pi/issues/2546)).
- Fixed GitHub Copilot Anthropic transport (`api.githubcopilot.com/v1/messages`) returning `400 tools.0.custom.eager_input_streaming: Extra inputs are not permitted` on every tool-bearing turn by stopping the emission of the per-tool `eager_input_streaming` flag and the `fine-grained-tool-streaming-2025-05-14` beta header on the Copilot transport — the proxy whitelists neither ([#2558](https://github.com/can1357/oh-my-pi/issues/2558)).
- Disabled Bun's native ~300s pre-response `fetch` timeout in every streaming provider (OpenAI completions/responses, Azure responses, Anthropic, Codex SSE, Bedrock, Gemini CLI, Ollama). The configurable first-event/idle/SDK watchdogs (`PI_STREAM_FIRST_EVENT_TIMEOUT_MS`, `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS`, `compat.streamIdleTimeoutMs`) were silently capped by Bun's hidden ceiling, so cold large-context streams (e.g. self-hosted vLLM at multi-hundred-K prompts) died at exactly 300s with `TimeoutError: The operation timed out.` Direct callers of `./providers/{amazon-bedrock,google-gemini-cli,ollama,openai-codex-responses}` (which bypass `register-builtins`' iterator-level watchdog) now install a pre-response `AbortSignal.timeout(firstEventTimeoutMs)` alongside the disable, so a stalled upstream still fails within the configured budget instead of hanging forever ([#2422](https://github.com/can1357/oh-my-pi/issues/2422))
- Fixed Gemini / Antigravity streams (Google Cloud Code Assist API) creating a trailing empty text block and emitting redundant `text_start`/`text_delta`/`text_end` events at the end of the turn when the final SSE chunk contains an empty text part (`text: ""`). The parser now ignores empty text parts, preserving the active transcript block state and ensuring proper nesting and rendering of subsequent background jobs or new turns.
- Preserved terminal Google `thoughtSignature`s by still extracting and applying the signature on the active block even when the text part is empty or undefined.
- Stopped Gemini Antigravity sessions (`gemini-3*` / Claude under Cloud Code Assist) from leaking system rule reminders and personality preambles into the final response, by appending an explicit 'do not output rule checks' instruction to the injected system parts.
- Fixed Gemini / Antigravity streams (Google Cloud Code Assist API) letting a `functionCall` part's own `thoughtSignature` clobber the preceding text or thinking block's signature on `think → tool` and `text → tool` turns. A signed function-call part has `text: undefined`, so it fell into the terminal-signature branch while the prior block was still active; that branch now skips function-call parts, leaving the tool call's signature on the tool call where it belongs and preventing corrupted signatures on same-model replay.
- Fixed MiniMax-M3 OpenAI-compatible streams rendering reasoning twice when the same chunk carried both `<think>…</think>` content and structured `reasoning_content`; structured reasoning now wins and cumulative MiniMax reasoning snapshots are collapsed to deltas using a per-signature snapshot tracker that survives the `</think>`-to-text block transition (so post-answer cumulative snapshots don't reinstate a duplicate thinking block). ([#2433](https://github.com/can1357/oh-my-pi/issues/2433))
## [15.12.6] - 2026-06-14
### Changed
- Bumped Z.AI (GLM Coding Plan) API key validation probe to glm-5.2.
### Fixed
- Fixed tool schema conversion for non-Cloud Code Assist Google Gemini models by normalizing parameters with `normalizeSchemaForGoogle` to prevent un-normalized schema properties (such as `additionalProperties: false` or type arrays) from causing Gemini API errors.
- Fixed OpenAI-family request builders dropping forced named `tool_choice` directives when the named tool is absent from the serialized `tools` array, preventing spec-strict providers from rejecting self-inconsistent requests. ([#1701](https://github.com/can1357/oh-my-pi/issues/1701))
## [15.12.4] - 2026-06-13
### Added
@@ -12,7 +37,6 @@
### Changed
- Replaced the OpenAI SDK client usage in `openai-completions`, `openai-responses`, `azure-openai-responses`, and `openai-codex-responses` with the new internal `postOpenAIStream` OpenAI-wire JSON/SSE transport
- Bumped Z.AI (GLM Coding Plan) API key validation probe to glm-5.2.
### Fixed
@@ -3393,4 +3417,4 @@ _Dedicated to Peter's shoulder ([@steipete](https://twitter.com/steipete))_
## [0.9.4] - 2025-11-26
Initial release with multi-provider LLM support.
Initial release with multi-provider LLM support.
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "15.12.5",
"version": "15.13.0",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+1 -1
View File
@@ -450,7 +450,7 @@ export type AuthStorageOptions = {
*
* Examples:
* - `"local ~/.omp/agent/agent.db"`
* - `"broker http://can.internal:8765"`
* - `"broker http://omp.internal:8765"`
*/
sourceLabel?: string;
/**
+1 -1
View File
@@ -18,7 +18,7 @@ export interface ProviderDetailsContext {
authMode?: string;
/**
* Human-readable description of the active credential, e.g.
* `"broker http://can.internal:8765 · oauth #5 (foo@bar.com)"`.
* `"broker http://omp.internal:8765 · oauth #5 (foo@bar.com)"`.
* Rendered as a `Source` field; omitted when undefined.
*/
credentialSource?: string;
+19 -1
View File
@@ -31,6 +31,7 @@ import type {
import { normalizeToolCallId, resolveCacheRetention } from "../utils";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump } from "../utils/http-inspector";
import { getStreamFirstEventTimeoutMs } from "../utils/idle-iterator";
import { parseStreamingJson, parseStreamingJsonThrottled } from "../utils/json-parse";
import { toolWireSchema } from "../utils/schema/wire";
import { invalidateAwsCredentialCache, resolveAwsCredentials } from "./aws-credentials";
@@ -282,12 +283,29 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
requestHeaders = { ...baseHeaders, ...signed };
}
// Bun's native fetch ceiling is disabled below (`timeout: false`) so
// configurable watchdogs govern slow-prefill streams (issue #2422).
// Direct callers that bypass `register-builtins` (which installs the
// iterator-level first-event watchdog) still need a pre-response
// timer, otherwise a Bedrock/proxy that accepts the POST and never
// sends headers would hang forever.
const firstEventTimeoutMs = options.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs();
const preResponseWatchdog =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0
? AbortSignal.timeout(firstEventTimeoutMs)
: undefined;
const fetchSignal = preResponseWatchdog
? options.signal
? AbortSignal.any([options.signal, preResponseWatchdog])
: preResponseWatchdog
: options.signal;
const response = await fetchWithRetry(url, {
method: "POST",
headers: requestHeaders,
body,
signal: options.signal,
signal: fetchSignal,
fetch: options.fetch,
timeout: false,
});
if (!response.ok) {
@@ -57,6 +57,8 @@ export type AnthropicFetchOptions = RequestInit & {
cert?: string;
key?: string;
};
/** Bun extension: see {@link FetchWithRetryOptions.timeout} — `false` disables Bun's native fetch TTFT timeout (issue #2422). */
timeout?: number | false;
};
export interface AnthropicClientOptions {
+14 -7
View File
@@ -2305,16 +2305,22 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
const baseUrl = resolveAnthropicBaseUrl(model, apiKey);
const foundryCustomHeaders = resolveAnthropicCustomHeaders(model);
const tlsFetchOptions = buildClaudeCodeTlsFetchOptions(model, baseUrl);
// Disable Bun's native ~300s pre-response fetch timeout (issue #2422).
// `AnthropicMessagesClient` already arms its own DEFAULT_TIMEOUT_MS timer
// per request, so the native ceiling can only short-circuit slow-prefill
// streams before the configured watchdog gets to govern them.
const fetchOptions: AnthropicFetchOptions = { ...(tlsFetchOptions ?? {}), timeout: false };
const baseFetch = args.fetch ?? fetch;
// Only OAuth requests inject the CC billing header; no API-key request can ever
// contain it, so there is no need to install the rewriter for those.
const cchFetch = oauthToken ? wrapFetchForCch(baseFetch) : baseFetch;
if (model.provider === "github-copilot") {
const copilotApiKey = parseGitHubCopilotApiKey(apiKey).accessToken;
// The GitHub Copilot Anthropic proxy doesn't accept Anthropic beta
// features (and the catalog already forces `supportsEagerToolInputStreaming
// = false` for this host, so `needsFineGrainedToolStreamingBeta` is true
// whenever tools are present). Forward only caller-supplied betas.
const betaFeatures = [...extraBetas];
if (needsFineGrainedToolStreamingBeta) {
betaFeatures.push(fineGrainedToolStreamingBeta);
}
const defaultHeaders = mergeHeaders(
{
Accept: stream ? "text/event-stream" : "application/json",
@@ -2337,7 +2343,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
maxRetries: 5,
defaultHeaders,
fetch: cchFetch,
...(tlsFetchOptions ? { fetchOptions: tlsFetchOptions } : {}),
fetchOptions,
};
}
@@ -2372,6 +2378,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
maxRetries: 5,
defaultHeaders,
fetch: cchFetch,
fetchOptions,
};
}
@@ -2388,7 +2395,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
maxRetries: 5,
defaultHeaders,
fetch: cchFetch,
...(tlsFetchOptions ? { fetchOptions: tlsFetchOptions } : {}),
fetchOptions,
};
}
// OpenCode Zen's Anthropic-compatible gateway accepts bearer auth only;
@@ -2402,7 +2409,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
maxRetries: 5,
defaultHeaders,
fetch: cchFetch,
...(tlsFetchOptions ? { fetchOptions: tlsFetchOptions } : {}),
fetchOptions,
};
}
@@ -2421,7 +2428,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
maxRetries: 5,
defaultHeaders,
fetch: cchFetch,
...(tlsFetchOptions ? { fetchOptions: tlsFetchOptions } : {}),
fetchOptions,
};
}
@@ -336,7 +336,15 @@ function buildParams(
if (context.tools) {
params.tools = convertTools(context.tools);
if (options?.toolChoice) {
params.tool_choice = mapToOpenAIResponsesToolChoice(options.toolChoice);
const toolChoice = mapToOpenAIResponsesToolChoice(options.toolChoice);
if (
toolChoice &&
(typeof toolChoice === "string" ||
toolChoice.type !== "function" ||
context.tools.some(tool => tool.name === toolChoice.name))
) {
params.tool_choice = toolChoice;
}
}
}
+35 -3
View File
@@ -7,6 +7,7 @@ import { createHash, randomBytes, randomUUID } from "node:crypto";
import { scheduler } from "node:timers/promises";
import { calculateCost } from "@oh-my-pi/pi-catalog/models";
import {
ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION,
ANTIGRAVITY_SYSTEM_INSTRUCTION,
getAntigravityUserAgent,
getGeminiCliHeaders,
@@ -27,6 +28,7 @@ import type {
import { normalizeSystemPrompts } from "../utils";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump } from "../utils/http-inspector";
import { getStreamFirstEventTimeoutMs } from "../utils/idle-iterator";
// Refresh is the sole responsibility of AuthStorage (broker-aware, single-flighted);
// the stream provider trusts the access token threaded through `options.apiKey`.
import { normalizeSchemaForCCA } from "../utils/schema";
@@ -101,6 +103,7 @@ const ANTIGRAVITY_SANDBOX_ENDPOINT = "https://daily-cloudcode-pa.sandbox.googlea
const ANTIGRAVITY_ENDPOINT_FALLBACKS = [ANTIGRAVITY_DAILY_ENDPOINT, ANTIGRAVITY_SANDBOX_ENDPOINT] as const;
export {
ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION,
ANTIGRAVITY_SYSTEM_INSTRUCTION,
getAntigravityUserAgent,
getGeminiCliHeaders,
@@ -365,17 +368,34 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
headers: requestHeaders,
};
// Direct callers that skip `register-builtins` (which installs the
// iterator-level watchdog) need a pre-response timer alongside
// `timeout: false`; otherwise a stalled Cloud Code Assist proxy
// would hang forever. Floor matches the lazy wrapper's 5min default.
const firstEventTimeoutMs =
options?.streamFirstEventTimeoutMs ?? getStreamFirstEventTimeoutMs(undefined, 300_000);
const preResponseWatchdog =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0
? AbortSignal.timeout(firstEventTimeoutMs)
: undefined;
const callerSignal = options?.signal;
const fetchSignal = preResponseWatchdog
? callerSignal
? AbortSignal.any([callerSignal, preResponseWatchdog])
: preResponseWatchdog
: callerSignal;
const response = await fetchWithRetry(
attempt => `${endpoints[Math.min(attempt, endpoints.length - 1)]}/v1internal:streamGenerateContent?alt=sse`,
{
method: "POST",
headers: requestHeaders,
body: requestBodyJson,
signal: options?.signal,
signal: fetchSignal,
maxAttempts: MAX_RETRIES + 1,
defaultDelayMs: attempt => BASE_DELAY_MS * 2 ** attempt,
maxDelayMs: options?.maxRetryDelayMs ?? RATE_LIMIT_BUDGET_MS,
fetch: options?.fetch,
timeout: false,
},
);
if (!response.ok) {
@@ -447,7 +467,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
const candidate = responseData.candidates?.[0];
if (candidate?.content?.parts) {
for (const part of candidate.content.parts) {
if (part.text !== undefined) {
if (part.text !== undefined && part.text !== "") {
const isThinking = isThinkingPart(part);
if (
!currentBlock ||
@@ -484,6 +504,18 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
partial: output,
});
}
} else if (part.text === "" && part.thoughtSignature && currentBlock && !part.functionCall) {
if (currentBlock.type === "thinking") {
currentBlock.thinkingSignature = retainThoughtSignature(
currentBlock.thinkingSignature,
part.thoughtSignature,
);
} else {
currentBlock.textSignature = retainThoughtSignature(
currentBlock.textSignature,
part.thoughtSignature,
);
}
}
if (part.functionCall) {
@@ -849,10 +881,10 @@ export function buildRequest(
if (isAntigravity && shouldInjectAntigravitySystemInstruction(model.id)) {
const existingParts = request.systemInstruction?.parts ?? [];
request.systemInstruction = {
role: "user",
parts: [
{ text: ANTIGRAVITY_SYSTEM_INSTRUCTION },
{ text: `Please ignore following [ignore]${ANTIGRAVITY_SYSTEM_INSTRUCTION}[/ignore]` },
{ text: ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION },
...existingParts,
],
};
+14 -2
View File
@@ -372,7 +372,7 @@ export function convertTools(
description: tool.description || "",
...(useParameters
? { parameters: normalizeSchemaForCCA(toolWireSchema(tool)) }
: { parametersJsonSchema: toolWireSchema(tool) }),
: { parametersJsonSchema: normalizeSchemaForGoogle(toolWireSchema(tool)) }),
})),
},
];
@@ -609,7 +609,7 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
const candidate = chunk.candidates?.[0];
if (candidate?.content?.parts) {
for (const part of candidate.content.parts) {
if (part.text !== undefined) {
if (part.text !== undefined && part.text !== "") {
if (!firstTokenSeen) {
firstTokenSeen = true;
onFirstToken?.();
@@ -650,6 +650,18 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
partial: output,
});
}
} else if (part.text === "" && part.thoughtSignature && currentBlock && !part.functionCall) {
if (currentBlock.type === "thinking") {
currentBlock.thinkingSignature = retainThoughtSignature(
currentBlock.thinkingSignature,
part.thoughtSignature,
);
} else if (retainTextSignature) {
currentBlock.textSignature = retainThoughtSignature(
currentBlock.textSignature,
part.thoughtSignature,
);
}
}
if (part.functionCall) {
+19 -1
View File
@@ -18,6 +18,7 @@ import type {
import { normalizeSystemPrompts } from "../utils";
import { AssistantMessageEventStream } from "../utils/event-stream";
import { type CapturedHttpErrorResponse, finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
import { getOpenAIStreamFirstEventTimeoutMs, getOpenAIStreamIdleTimeoutMs } from "../utils/idle-iterator";
import { parseStreamingJson } from "../utils/json-parse";
import { toolWireSchema } from "../utils/schema/wire";
import {
@@ -525,6 +526,22 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
url: `${baseUrl}/api/chat`,
body,
};
// Direct callers that bypass `register-builtins` (which installs
// the iterator-level watchdog) need a pre-response timer alongside
// `timeout: false`; otherwise an Ollama server that accepts the
// POST and never streams headers would hang forever (issue #2422).
const idleTimeoutMs = options.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
const firstEventTimeoutMs =
options.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
const preResponseWatchdog =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0
? AbortSignal.timeout(firstEventTimeoutMs)
: undefined;
const fetchSignal = preResponseWatchdog
? options.signal
? AbortSignal.any([options.signal, preResponseWatchdog])
: preResponseWatchdog
: options.signal;
const response = await fetchWithRetry(`${baseUrl}/api/chat`, {
method: "POST",
headers: {
@@ -534,9 +551,10 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
"Content-Type": "application/json",
},
body: JSON.stringify(body),
signal: options.signal,
signal: fetchSignal,
defaultDelayMs: OLLAMA_RETRY_DELAYS_MS,
fetch: options.fetch,
timeout: false,
});
if (!response.ok) {
capturedErrorResponse = await captureHttpErrorResponse(response);
@@ -272,6 +272,7 @@ interface CodexRequestSetup {
requestSignal: AbortSignal;
wrapCodexSseStream: (source: AsyncGenerator<Record<string, unknown>>) => AsyncGenerator<Record<string, unknown>>;
requestAbortController: AbortController;
firstEventTimeoutMs: number | undefined;
websocketIdleTimeoutMs: number | undefined;
websocketFirstEventTimeoutMs: number | undefined;
}
@@ -554,13 +555,16 @@ export function normalizeCodexToolChoice(
if (!choice) return undefined;
if (typeof choice === "string") return choice;
const allowFreeform = model ? supportsFreeformApplyPatchCodex(model) : false;
const mapName = (name: string): Record<string, string> => {
const mapName = (name: string): Record<string, string> | undefined => {
const directTool = tools.find(tool => tool.name === name);
const customTool = allowFreeform
? tools.find(tool => tool.customFormat && (tool.name === name || tool.customWireName === name))
: undefined;
const offeredTool = customTool ?? directTool;
if (!offeredTool) return undefined;
return customTool
? { type: "custom", name: customTool.customWireName ?? customTool.name }
: { type: "function", name };
: { type: "function", name: offeredTool.name };
};
if (choice.type === "function") {
if ("function" in choice && choice.function?.name) {
@@ -687,6 +691,7 @@ function createRequestSetup(options: OpenAICodexResponsesOptions | undefined): C
requestAbortController,
requestSignal,
wrapCodexSseStream,
firstEventTimeoutMs,
websocketIdleTimeoutMs,
websocketFirstEventTimeoutMs,
};
@@ -983,6 +988,7 @@ async function openCodexSseTransport(
state,
requestContext.responsesLite,
requestSetup.requestSignal,
requestSetup.firstEventTimeoutMs,
event => options?.onSseEvent?.(event, model),
options?.fetch,
),
@@ -3016,7 +3022,8 @@ async function openCodexSseEventStream(
body: RequestBody,
state: CodexWebSocketSessionState | undefined,
responsesLite: boolean,
signal?: AbortSignal,
signal: AbortSignal | undefined,
firstEventTimeoutMs: number | undefined,
onSseEvent?: OpenAICodexResponsesOptions["onSseEvent"],
fetchOverride?: FetchImpl,
): Promise<AsyncGenerator<Record<string, unknown>>> {
@@ -3028,15 +3035,31 @@ async function openCodexSseEventStream(
sentTurnStateHeader: headers.has(X_CODEX_TURN_STATE_HEADER),
sentModelsEtagHeader: headers.has(X_MODELS_ETAG_HEADER),
});
// `wrapCodexSseStream` arms a first-event watchdog only after this fetch
// resolves (it wraps the SSE generator). With `timeout: false` disabling
// Bun's native 300s ceiling, a stalled pre-response request needs its own
// watchdog — combine the caller signal with a fresh
// `AbortSignal.timeout(firstEventTimeoutMs)` so headers must arrive
// within the configured budget (issue #2422).
const preResponseWatchdog =
firstEventTimeoutMs !== undefined && firstEventTimeoutMs > 0
? AbortSignal.timeout(firstEventTimeoutMs)
: undefined;
const fetchSignal = preResponseWatchdog
? signal
? AbortSignal.any([signal, preResponseWatchdog])
: preResponseWatchdog
: signal;
const response = await fetchWithRetry(url, {
method: "POST",
headers,
body: JSON.stringify(body),
signal,
signal: fetchSignal,
maxAttempts: CODEX_MAX_RETRIES + 1,
defaultDelayMs: attempt => CODEX_RETRY_DELAY_MS * (attempt + 1),
maxDelayMs: CODEX_RATE_LIMIT_BUDGET_MS,
fetch: fetchOverride,
timeout: false,
});
logCodexDebug("codex response", {
url: response.url,
@@ -699,6 +699,14 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
if (!firstTokenTime) firstTokenTime = Date.now();
appendText(output, stream, text);
};
// Tracks the last full cumulative reasoning snapshot per signature (the
// reasoning field name) so dedup survives block transitions. Required
// for MiniMax-M3: once `</think>` and visible text arrive, currentBlock
// flips to "text", but later chunks keep carrying the same cumulative
// `reasoning_content` snapshot. Without an external tracker the guard
// below misses and the snapshot gets re-emitted as a fresh thinking
// block after the answer has started.
const lastCumulativeReasoningBySignature = new Map<string, string>();
const appendThinkingDelta = (
thinking: string,
signature?: string,
@@ -706,13 +714,13 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
): void => {
if (!thinking) return;
let emittedThinking = thinking;
if (
source === "cumulative" &&
currentBlock?.type === "thinking" &&
(signature === undefined || currentBlock.thinkingSignature === signature) &&
thinking.startsWith(currentBlock.thinking)
) {
emittedThinking = thinking.slice(currentBlock.thinking.length);
if (source === "cumulative") {
const key = signature ?? "";
const lastSnapshot = lastCumulativeReasoningBySignature.get(key) ?? "";
if (thinking.startsWith(lastSnapshot)) {
emittedThinking = thinking.slice(lastSnapshot.length);
}
lastCumulativeReasoningBySignature.set(key, thinking);
if (!emittedThinking) return;
}
if (!firstTokenTime) firstTokenTime = Date.now();
@@ -1217,6 +1225,11 @@ async function createRequestSetup(
};
}
function getForcedCompletionsToolName(toolChoice: OpenAICompletionsParams["tool_choice"]): string | undefined {
if (typeof toolChoice !== "object" || toolChoice === null || !("function" in toolChoice)) return undefined;
return toolChoice.function.name;
}
function buildParams(
model: Model<"openai-completions">,
context: Context,
@@ -1228,6 +1241,7 @@ function buildParams(
Boolean(options?.reasoning) && !options?.disableReasoning && Boolean(model.reasoning);
const forcedToolChoiceSuppressesThinking =
compat.disableReasoningOnForcedToolChoice &&
compat.supportsForcedToolChoice &&
isForcedToolChoice(mapToOpenAICompletionsToolChoice(options?.toolChoice));
if (compat.whenThinking && thinkingEnabledForRequest && !forcedToolChoiceSuppressesThinking) {
compat = compat.whenThinking; // precomputed at model build — pointer swap, no allocation
@@ -1329,6 +1343,12 @@ function buildParams(
if (options?.toolChoice && compat.supportsToolChoice) {
params.tool_choice = mapToOpenAICompletionsToolChoice(options.toolChoice);
}
if (isForcedToolChoice(params.tool_choice) && !compat.supportsForcedToolChoice) {
// Some thinking-required OpenAI-compatible models reject forced
// `tool_choice` while still accepting tools with the default auto
// selector. Keep the tool available and let the model choose it.
params.tool_choice = "auto";
}
if (params.tool_choice === "none" && (!Array.isArray(params.tools) || params.tools.length === 0)) {
// `tool_choice: "none"` with no tools to gate is redundant and also
@@ -1342,6 +1362,19 @@ function buildParams(
delete params.tool_choice;
}
const forcedToolName = getForcedCompletionsToolName(params.tool_choice);
if (
forcedToolName !== undefined &&
(!Array.isArray(params.tools) ||
!params.tools.some(tool => tool.type === "function" && tool.function.name === forcedToolName))
) {
// A forced named tool_choice is only valid when the same request offers
// that function in `tools`. Active-tool filtering normally enforces this
// before provider dispatch; this guard keeps raw provider callers from
// emitting a self-inconsistent OpenAI-compatible payload.
delete params.tool_choice;
}
if (supportsReasoningParams && compat.thinkingFormat === "zai" && model.reasoning) {
// Z.ai uses binary thinking: { type: "enabled" | "disabled" }
// Must explicitly disable since z.ai defaults to thinking enabled.
@@ -934,7 +934,10 @@ export async function processResponsesStream<TApi extends Api>(
// reaches the SDK stream), actively releasing the connection.
break;
} else if (event.type === "error") {
throw new Error(`Error Code ${event.code}: ${event.message}`);
const err = (event as any).error ?? event;
const code = err.code ?? "unknown";
const message = err.message ?? "no message";
throw new Error(`Error Code ${code}: ${message}`);
} else if (event.type === "response.failed") {
populateResponsesUsageFromResponse(output, event.response?.usage);
const error = event.response?.error ?? (event.response as any)?.status_details?.error;
@@ -836,13 +836,18 @@ export function mapOpenAIResponsesToolChoiceForTools(
model: Model<"openai-responses">,
): OpenAIResponsesToolChoice {
const mapped = mapToOpenAIResponsesToolChoice(choice);
if (!mapped || typeof mapped === "string" || mapped.type !== "function" || !supportsFreeformApplyPatch(model)) {
if (!mapped || typeof mapped === "string" || mapped.type !== "function") {
return mapped;
}
const customTool = tools.find(
tool => tool.customFormat && (tool.name === mapped.name || tool.customWireName === mapped.name),
);
const directTool = tools.find(tool => tool.name === mapped.name);
const customTool = supportsFreeformApplyPatch(model)
? tools.find(tool => tool.customFormat && (tool.name === mapped.name || tool.customWireName === mapped.name))
: undefined;
const offeredTool = customTool ?? directTool;
if (!offeredTool) {
return undefined;
}
return customTool ? { type: "custom", name: customTool.customWireName ?? customTool.name } : mapped;
}
+8 -3
View File
@@ -40,9 +40,10 @@ function resolveClientId(): string {
/**
* Resolve callback-server options from `GITLAB_REDIRECT_URI`. When set, the
* exact string is advertised to GitLab (strict matching), random-port fallback
* is disabled, and the local listener is bound to the URI's loopback host/port
* so the browser callback lands on us. Non-loopback URIs bind a random local
* port — only the paste-code path can complete in that case.
* is disabled, and HTTP loopback URIs bind the listener to the URI's host/port
* so the browser callback lands on us. HTTPS loopback URIs are rejected because
* the local callback server is plaintext HTTP. Non-loopback URIs bind a random
* local port — only the paste-code path can complete in that case.
*/
function resolveCallbackOptions(): OAuthCallbackFlowOptions {
const raw = process.env.GITLAB_REDIRECT_URI?.trim();
@@ -65,6 +66,10 @@ function resolveCallbackOptions(): OAuthCallbackFlowOptions {
}
const isLoopback = parsed.hostname === "localhost" || parsed.hostname === "127.0.0.1" || parsed.hostname === "[::1]";
if (isLoopback && parsed.protocol !== "http:") {
throw new Error(`GITLAB_REDIRECT_URI loopback callbacks must use http://, got: ${raw}`);
}
const port = parsed.port ? Number.parseInt(parsed.port, 10) : parsed.protocol === "https:" ? 443 : 80;
return {
+4
View File
@@ -79,6 +79,10 @@ export async function postOpenAIStream<TEvent>(init: OpenAIStreamRequestInit): P
signal: init.signal,
fetch: init.fetch,
maxAttempts: init.maxAttempts ?? DEFAULT_MAX_ATTEMPTS,
// Bun's native fetch enforces a hard ~300s pre-response timeout (issue #2422).
// Cold large-context streams legitimately exceed it; the caller's
// `firstEventTimeoutMs`/`AbortSignal` already govern stuck requests.
timeout: false,
});
if (!response.ok) {
throw await captureOpenAIHttpError(response);
@@ -1,6 +1,7 @@
import { describe, expect, it } from "bun:test";
import { streamGoogle } from "@oh-my-pi/pi-ai/providers/google";
import { streamGoogleGeminiCli } from "@oh-my-pi/pi-ai/providers/google-gemini-cli";
import { streamGoogleVertex } from "@oh-my-pi/pi-ai/providers/google-vertex";
import type { AssistantMessageEvent, Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -54,6 +55,19 @@ const genaiModel: Model<"google-generative-ai"> = buildModel({
maxTokens: 32_000,
});
const vertexModel: Model<"google-vertex"> = buildModel({
id: "gemini-3-flash",
name: "Gemini 3 Flash (Vertex)",
api: "google-vertex",
provider: "google",
baseUrl: "",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 32_000,
});
const cliModel: Model<"google-gemini-cli"> = buildModel({
id: "gemini-3-flash",
name: "Gemini 3 Flash (CCA)",
@@ -100,6 +114,100 @@ describe("Google empty-response retry (public + Vertex path)", () => {
expect(result.stopReason).toBe("error");
expect(result.errorMessage).toContain("empty response");
});
it("filters out empty text parts at stream end but preserves terminal thought signatures", async () => {
const chunks = [
{ candidates: [{ content: { parts: [{ text: "Hello" }] } }] },
{
candidates: [
{ content: { parts: [{ text: "", thoughtSignature: "terminal-sig" }] }, finishReason: "STOP" },
],
},
];
const fetchMock: FetchImpl = async input => {
const url = input instanceof Request ? input.url : input.toString();
if (url.includes("oauth2.googleapis.com/token") || url.includes("metadata.google.internal")) {
return new Response(JSON.stringify({ access_token: "token", expires_in: 3600 }));
}
return sse(...chunks);
};
const stream = streamGoogleVertex(vertexModel, context, {
project: "project",
location: "location",
fetch: fetchMock,
});
const { events } = await drain(stream);
const result = await stream.result();
expect(result.stopReason).toBe("stop");
expect(result.content).toHaveLength(1);
expect(result.content[0]).toEqual({
type: "text",
text: "Hello",
textSignature: "terminal-sig",
});
const textStartEvents = events.filter(e => e.type === "text_start");
expect(textStartEvents).toHaveLength(1);
expect(textStartEvents[0].contentIndex).toBe(0);
const textDeltaEvents = events.filter(e => e.type === "text_delta");
expect(textDeltaEvents).toHaveLength(1);
expect(textDeltaEvents[0].delta).toBe("Hello");
const textEndEvents = events.filter(e => e.type === "text_end");
expect(textEndEvents).toHaveLength(1);
expect(textEndEvents[0].content).toBe("Hello");
});
it("does not coalesce function-call thought signatures into the preceding Vertex text block", async () => {
const chunks = [
{ candidates: [{ content: { parts: [{ text: "Hello" }] } }] },
{
candidates: [
{
content: {
parts: [
{
functionCall: { name: "lookup", args: { q: "x" }, id: "call_1" },
thoughtSignature: "function-call-sig",
},
],
},
finishReason: "STOP",
},
],
},
];
const fetchMock: FetchImpl = async input => {
const url = input instanceof Request ? input.url : input.toString();
if (url.includes("oauth2.googleapis.com/token") || url.includes("metadata.google.internal")) {
return new Response(JSON.stringify({ access_token: "token", expires_in: 3600 }));
}
return sse(...chunks);
};
const stream = streamGoogleVertex(vertexModel, context, {
project: "project",
location: "location",
fetch: fetchMock,
});
const result = await stream.result();
expect(result.stopReason).toBe("toolUse");
expect(result.content).toHaveLength(2);
expect(result.content[0]).toEqual({ type: "text", text: "Hello" });
expect(result.content[1]).toMatchObject({
type: "toolCall",
id: "call_1",
name: "lookup",
arguments: { q: "x" },
thoughtSignature: "function-call-sig",
});
});
});
describe("Google empty-response retry (Cloud Code Assist path)", () => {
@@ -126,4 +234,50 @@ describe("Google empty-response retry (Cloud Code Assist path)", () => {
expect(textOf(result)).toBe("Done.");
void events;
});
it("does not coalesce function-call thought signatures into the preceding text block", async () => {
const chunks = [
{ response: { candidates: [{ content: { parts: [{ text: "Done" }] } }] } },
{
response: {
candidates: [
{
content: {
parts: [
{
functionCall: { name: "lookup", args: { q: "x" }, id: "call_1" },
thoughtSignature: "function-call-sig",
},
],
},
finishReason: "STOP",
},
],
},
},
];
const fetchMock: FetchImpl = async () => {
const response = sse(...chunks);
Object.defineProperty(response, "url", { value: "https://example.com/v1internal:streamGenerateContent" });
return response;
};
const stream = streamGoogleGeminiCli(cliModel, context, {
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
fetch: fetchMock,
});
const result = await stream.result();
expect(result.stopReason).toBe("toolUse");
expect(result.content).toHaveLength(2);
expect(result.content[0]).toEqual({ type: "text", text: "Done" });
expect(result.content[1]).toMatchObject({
type: "toolCall",
id: "call_1",
name: "lookup",
arguments: { q: "x" },
thoughtSignature: "function-call-sig",
});
});
});
@@ -1,6 +1,7 @@
import { describe, expect, it } from "bun:test";
import * as geminiCliProvider from "@oh-my-pi/pi-ai/providers/google-gemini-cli";
import {
ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION,
ANTIGRAVITY_SYSTEM_INSTRUCTION,
buildRequest,
parseGeminiCliCredentials,
@@ -8,7 +9,7 @@ import {
streamGoogleGeminiCli,
} from "@oh-my-pi/pi-ai/providers/google-gemini-cli";
import { getOAuthApiKey } from "@oh-my-pi/pi-ai/registry/oauth";
import type { Context, FetchImpl, Model, TJsonSchema } from "@oh-my-pi/pi-ai/types";
import type { AssistantMessageEvent, Context, FetchImpl, Model, TJsonSchema } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import type { ModelSpec } from "@oh-my-pi/pi-catalog/types";
@@ -222,8 +223,10 @@ describe("Google Gemini CLI alignment", () => {
const parts = payload.request.systemInstruction?.parts ?? [];
// The antigravity identity header must be injected as the first part.
expect(parts[0]?.text).toBe(ANTIGRAVITY_SYSTEM_INSTRUCTION);
expect(parts[1]?.text).toBe(`Please ignore following [ignore]${ANTIGRAVITY_SYSTEM_INSTRUCTION}[/ignore]`);
expect(parts[2]?.text).toBe(ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION);
// The user-supplied system prompt must appear after the injected parts.
expect(parts.some(p => p.text === "my instructions")).toBe(true);
expect(parts.slice(3).some(p => p.text === "my instructions")).toBe(true);
}
});
it("adds anthropic-beta for Antigravity Claude reasoning models without relying on id suffix", async () => {
@@ -252,6 +255,130 @@ describe("Google Gemini CLI alignment", () => {
expect(requestHeaders!.get("Client-Metadata")).toBeNull();
});
it("filters out empty text parts at stream end but preserves terminal thought signatures", async () => {
const sseChunks = [
'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":"Hello"}]}}]}}\n\n',
'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":"","thoughtSignature":"terminal-sig"}]},"finishReason":"STOP"}]}}\n\n',
];
const fetchMock: FetchImpl = async () => {
const stream = new ReadableStream({
async start(controller) {
const encoder = new TextEncoder();
for (const chunk of sseChunks) {
controller.enqueue(encoder.encode(chunk));
await Bun.sleep(5);
}
controller.close();
},
});
return new Response(stream, {
status: 200,
headers: { "Content-Type": "text/event-stream" },
});
};
const model: Model<"google-gemini-cli"> = buildModel({
...createModel("google-antigravity"),
id: "gemini-3.5-flash",
name: "Gemini 3.5 Flash",
reasoning: true,
} as ModelSpec<"google-gemini-cli">);
const events: AssistantMessageEvent[] = [];
const stream = streamGoogleGeminiCli(model, createContext(), {
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
fetch: fetchMock,
});
for await (const event of stream) {
events.push(event);
}
const result = await stream.result();
expect(result.stopReason).toBe("stop");
expect(result.content).toHaveLength(1);
expect(result.content[0]).toEqual({
type: "text",
text: "Hello",
textSignature: "terminal-sig",
});
const textStartEvents = events.filter(e => e.type === "text_start");
expect(textStartEvents).toHaveLength(1);
expect(textStartEvents[0].contentIndex).toBe(0);
const textDeltaEvents = events.filter(e => e.type === "text_delta");
expect(textDeltaEvents).toHaveLength(1);
expect(textDeltaEvents[0].delta).toBe("Hello");
const textEndEvents = events.filter(e => e.type === "text_end");
expect(textEndEvents).toHaveLength(1);
expect(textEndEvents[0].content).toBe("Hello");
});
it("keeps a text block's own thoughtSignature when a following function call carries its own", async () => {
// A functionCall part with `text: undefined` must NOT pollute the preceding text/thinking
// block via the terminal-signature branch; its signature belongs on the tool call alone.
const sseChunks = [
'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"text":"Hello","thoughtSignature":"text-sig"}]}}]}}\n\n',
'data: {"response":{"candidates":[{"content":{"role":"model","parts":[{"functionCall":{"name":"get_weather","args":{"city":"SF"}},"thoughtSignature":"toolcall-sig"}]},"finishReason":"STOP"}]}}\n\n',
];
const fetchMock: FetchImpl = async () => {
const stream = new ReadableStream({
async start(controller) {
const encoder = new TextEncoder();
for (const chunk of sseChunks) {
controller.enqueue(encoder.encode(chunk));
await Bun.sleep(5);
}
controller.close();
},
});
return new Response(stream, {
status: 200,
headers: { "Content-Type": "text/event-stream" },
});
};
const model: Model<"google-gemini-cli"> = buildModel({
...createModel("google-antigravity"),
id: "gemini-3.5-flash",
name: "Gemini 3.5 Flash",
reasoning: true,
} as ModelSpec<"google-gemini-cli">);
const events: AssistantMessageEvent[] = [];
const stream = streamGoogleGeminiCli(model, createContext(), {
apiKey: JSON.stringify({ token: "token", projectId: "proj-123" }),
fetch: fetchMock,
});
for await (const event of stream) {
events.push(event);
}
const result = await stream.result();
expect(result.stopReason).toBe("toolUse");
expect(result.content).toHaveLength(2);
// The text block keeps its OWN signature — the function call's signature must NOT migrate onto it.
expect(result.content[0]).toEqual({
type: "text",
text: "Hello",
textSignature: "text-sig",
});
// The function call's signature is captured on the tool call itself, by the functionCall branch.
const toolCall = result.content[1];
expect(toolCall.type).toBe("toolCall");
if (toolCall.type === "toolCall") {
expect(toolCall.name).toBe("get_weather");
expect(toolCall.thoughtSignature).toBe("toolcall-sig");
}
expect(events.filter(e => e.type === "toolcall_start")).toHaveLength(1);
});
describe("retry guardrails", () => {
it("does not treat explicit HTTP failures as network retry errors", async () => {
let fetchCalls = 0;
+28 -2
View File
@@ -172,7 +172,7 @@ describe("Cloud Code Assist Claude tool schema conversion", () => {
expect(claudeDeclaration.parametersJsonSchema).toBeUndefined();
expect(
(geminiDeclaration.parametersJsonSchema as { properties?: Record<string, unknown> })?.properties?.lines,
).toEqual((parameters as { properties: { lines: unknown } }).properties.lines);
).toEqual(normalizeSchemaForGoogle((parameters as { properties: { lines: unknown } }).properties.lines));
});
it("collapses mixed anyOf with shared metadata for edit-style lines fields", () => {
@@ -285,7 +285,7 @@ describe("Cloud Code Assist Claude tool schema conversion", () => {
expect(JSON.stringify(claudeDeclaration.parameters)).not.toContain('"anyOf"');
expect(
(geminiDeclaration.parametersJsonSchema as { properties?: Record<string, unknown> })?.properties?.value,
).toEqual((parameters as { properties: { value: unknown } }).properties.value);
).toEqual(normalizeSchemaForGoogle((parameters as { properties: { value: unknown } }).properties.value));
});
it("falls back to minimal object schema when non-null unresolved unions remain for CCA Claude", () => {
@@ -347,6 +347,32 @@ describe("Cloud Code Assist Claude tool schema conversion", () => {
},
});
});
it("normalizes schemas for gemini models using normalizeSchemaForGoogle", () => {
const parameters = {
type: "object",
properties: {
value: {
type: "string",
},
},
additionalProperties: false,
} as unknown as TJsonSchema;
const tools: Tool[] = [{ name: "test_tool", description: "Test tool", parameters }];
const model = createModel("gemini-3.5-flash");
const result = convertTools(tools, model);
const declaration = result?.[0]?.functionDeclarations[0] as Record<string, unknown>;
expect(declaration.parametersJsonSchema).toEqual({
type: "object",
properties: {
value: {
type: "string",
},
},
});
});
});
/**
+68
View File
@@ -123,4 +123,72 @@ describe("issue #1203 - MiniMax Coding Plan CN think tags", () => {
{ type: "text", text: "Hello!" },
]);
});
it("dedupes MiniMax-M3 cumulative reasoning snapshots after answer text has started", async () => {
const model = getBundledModel("minimax-code-cn", "MiniMax-M3") as Model<"openai-completions">;
const fetchMock = createMockFetch([
{
id: "chatcmpl-minimax-cn",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [
{
index: 0,
delta: {
role: "assistant",
content: "<think>The user just",
reasoning_content: "The user just",
},
},
],
},
{
id: "chatcmpl-minimax-cn",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [
{
index: 0,
delta: {
content: " said hi.</think>Hello!",
reasoning_content: "The user just said hi.",
},
},
],
},
{
// Visible text continues, yet the host keeps echoing the same
// cumulative reasoning snapshot. currentBlock is now "text", so the
// old (block-scoped) dedup would re-emit the entire snapshot as a
// second thinking block.
id: "chatcmpl-minimax-cn",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [
{
index: 0,
delta: {
content: " How can I help?",
reasoning_content: "The user just said hi.",
},
},
],
},
stopChunk(model),
"[DONE]",
]);
const result = await streamOpenAICompletions(model, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
}).result();
expect(result.content).toEqual([
{ type: "thinking", thinking: "The user just said hi.", thinkingSignature: "reasoning_content" },
{ type: "text", text: "Hello! How can I help?" },
]);
});
});
+211
View File
@@ -0,0 +1,211 @@
import { describe, expect, it } from "bun:test";
import { streamAzureOpenAIResponses } from "@oh-my-pi/pi-ai/providers/azure-openai-responses";
import { streamOpenAICodexResponses } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import { streamOpenAIResponses } from "@oh-my-pi/pi-ai/providers/openai-responses";
import type { Context, Model, Tool, ToolChoice } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { z } from "zod/v4";
const completionsModel: Model<"openai-completions"> = buildModel({
id: "gpt-4o-mini-test",
name: "GPT-4o Mini Test",
api: "openai-completions",
provider: "openai",
baseUrl: "https://example.test/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: 4096,
});
const responsesModel: Model<"openai-responses"> = buildModel({
id: "gpt-5-mini-test",
name: "GPT-5 Mini Test",
api: "openai-responses",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 400000,
maxTokens: 128000,
});
const azureModel: Model<"azure-openai-responses"> = buildModel({
id: "gpt-5-mini-test",
name: "GPT-5 Mini Test",
api: "azure-openai-responses",
provider: "azure",
baseUrl: "https://example.openai.azure.com/openai/v1",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 400000,
maxTokens: 128000,
});
const codexModel: Model<"openai-codex-responses"> = buildModel({
id: "gpt-5-codex-test",
name: "GPT-5 Codex Test",
api: "openai-codex-responses",
provider: "openai-codex",
baseUrl: "https://chatgpt.com/backend-api",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 272000,
maxTokens: 128000,
});
const forkAgentTool: Tool = {
name: "fork_agent",
description: "Fork a subagent",
parameters: z.object({ prompt: z.string() }),
};
const searchTool: Tool = {
name: "dataforseo_search",
description: "Search via DataForSEO",
parameters: z.object({ query: z.string() }),
};
const todoTool: Tool = {
name: "todo",
description: "Manage a phased task list",
parameters: z.object({ ops: z.array(z.object({ op: z.string() })) }),
};
const absentTodoContext: Context = {
messages: [{ role: "user", content: "do the thing", timestamp: Date.now() }],
tools: [forkAgentTool, searchTool],
};
const presentTodoContext: Context = {
messages: [{ role: "user", content: "list everything", timestamp: Date.now() }],
tools: [forkAgentTool, todoTool],
};
const forcedTodoChoice: ToolChoice = { type: "tool", name: "todo" };
function createAbortedSignal(): AbortSignal {
const controller = new AbortController();
controller.abort();
return controller.signal;
}
function createCodexToken(accountId: string): string {
const header = Buffer.from(JSON.stringify({ alg: "none", typ: "JWT" })).toString("base64url");
const payload = Buffer.from(
JSON.stringify({ "https://api.openai.com/auth": { chatgpt_account_id: accountId } }),
).toString("base64url");
return `${header}.${payload}.signature`;
}
function captureCompletionsPayload(context: Context): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<Record<string, unknown>>();
streamOpenAICompletions(completionsModel, context, {
apiKey: "test-key",
toolChoice: forcedTodoChoice,
signal: createAbortedSignal(),
onPayload: payload => resolve(payload as Record<string, unknown>),
});
return promise;
}
function captureResponsesPayload(context: Context): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<Record<string, unknown>>();
streamOpenAIResponses(responsesModel, context, {
apiKey: "test-key",
toolChoice: forcedTodoChoice,
signal: createAbortedSignal(),
onPayload: payload => resolve(payload as Record<string, unknown>),
});
return promise;
}
function captureAzurePayload(context: Context): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<Record<string, unknown>>();
streamAzureOpenAIResponses(azureModel, context, {
apiKey: "test-key",
azureBaseUrl: azureModel.baseUrl,
azureApiVersion: "v1",
toolChoice: forcedTodoChoice,
signal: createAbortedSignal(),
onPayload: payload => resolve(payload as Record<string, unknown>),
});
return promise;
}
function captureCodexPayload(context: Context): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<Record<string, unknown>>();
streamOpenAICodexResponses(codexModel, context, {
apiKey: createCodexToken("acct_test"),
toolChoice: forcedTodoChoice,
signal: createAbortedSignal(),
onPayload: payload => resolve(payload as Record<string, unknown>),
});
return promise;
}
function completionToolNames(payload: Record<string, unknown>): Array<string | undefined> {
const tools = payload.tools as Array<{ function?: { name?: string } }> | undefined;
return tools?.map(tool => tool.function?.name) ?? [];
}
function responsesToolNames(payload: Record<string, unknown>): Array<string | undefined> {
const tools = payload.tools as Array<{ name?: string }> | undefined;
return tools?.map(tool => tool.name) ?? [];
}
describe("issue #1701 forced tool_choice guards", () => {
it("drops OpenAI Completions forced tool_choice when the named tool is absent", async () => {
const payload = await captureCompletionsPayload(absentTodoContext);
expect(completionToolNames(payload)).toEqual(["fork_agent", "dataforseo_search"]);
expect(payload.tool_choice).toBeUndefined();
});
it("keeps OpenAI Completions forced tool_choice when the named tool is present", async () => {
const payload = await captureCompletionsPayload(presentTodoContext);
expect(completionToolNames(payload)).toEqual(["fork_agent", "todo"]);
expect(payload.tool_choice).toEqual({ type: "function", function: { name: "todo" } });
});
it("drops OpenAI Responses forced tool_choice when the named tool is absent", async () => {
const payload = await captureResponsesPayload(absentTodoContext);
expect(responsesToolNames(payload)).toEqual(["fork_agent", "dataforseo_search"]);
expect(payload.tool_choice).toBeUndefined();
});
it("keeps OpenAI Responses forced tool_choice when the named tool is present", async () => {
const payload = await captureResponsesPayload(presentTodoContext);
expect(responsesToolNames(payload)).toEqual(["fork_agent", "todo"]);
expect(payload.tool_choice).toEqual({ type: "function", name: "todo" });
});
it("drops Azure Responses forced tool_choice when the named tool is absent", async () => {
const payload = await captureAzurePayload(absentTodoContext);
expect(responsesToolNames(payload)).toEqual(["fork_agent", "dataforseo_search"]);
expect(payload.tool_choice).toBeUndefined();
});
it("drops Codex Responses forced tool_choice when the named tool is absent", async () => {
const payload = await captureCodexPayload(absentTodoContext);
expect(responsesToolNames(payload)).toEqual(["fork_agent", "dataforseo_search"]);
expect(payload.tool_choice).toBeUndefined();
});
it("keeps Codex Responses forced tool_choice when the named tool is present", async () => {
const payload = await captureCodexPayload(presentTodoContext);
expect(responsesToolNames(payload)).toEqual(["fork_agent", "todo"]);
expect(payload.tool_choice).toEqual({ type: "function", name: "todo" });
});
});
+16
View File
@@ -123,6 +123,22 @@ describe("gitlab-duo OAuth env overrides (issue #2424)", () => {
).rejects.toThrow(/Invalid GITLAB_REDIRECT_URI/);
});
it("rejects HTTPS loopback GITLAB_REDIRECT_URI before opening browser auth", async () => {
process.env.GITLAB_REDIRECT_URI = "https://localhost:8443/callback";
const onAuth = vi.fn();
await expect(
loginGitLabDuo({
onAuth,
onManualCodeInput: async () => "x",
onPrompt: async () => "",
signal: AbortSignal.timeout(1_000),
}),
).rejects.toThrow(/loopback callbacks must use http:\/\//);
expect(onAuth).not.toHaveBeenCalled();
});
it("threads GITLAB_CLIENT_ID through the refresh request", async () => {
process.env.GITLAB_CLIENT_ID = "rotation-client";
+22 -5
View File
@@ -21,9 +21,12 @@ function abortedSignal(): AbortSignal {
return controller.signal;
}
async function capturePayload(opts: Parameters<typeof streamOpenAICompletions>[2]): Promise<Record<string, unknown>> {
async function capturePayload(
model: Model<"openai-completions">,
opts: Parameters<typeof streamOpenAICompletions>[2],
): Promise<Record<string, unknown>> {
const { promise, resolve } = Promise.withResolvers<unknown>();
streamOpenAICompletions(getBundledModel("opencode-go", "deepseek-v4-pro"), context, {
streamOpenAICompletions(model, context, {
...opts,
apiKey: "test-key",
signal: abortedSignal(),
@@ -32,14 +35,28 @@ async function capturePayload(opts: Parameters<typeof streamOpenAICompletions>[2
return (await promise) as Record<string, unknown>;
}
describe("issue #945 — OpenCode Go DeepSeek tool_choice is disabled", () => {
describe("OpenCode Go tool_choice compatibility", () => {
it("marks deepseek-v4-pro as not supporting tool_choice via compat override", () => {
const model = getBundledModel("opencode-go", "deepseek-v4-pro") as Model<"openai-completions">;
expect(model.compat?.supportsToolChoice).toBe(false);
});
it("omits tool_choice from payload but preserves tools and reasoning_effort", async () => {
const body = await capturePayload({ reasoning: "high", toolChoice: "auto" });
it("marks mimo-v2.5-pro as not supporting tool_choice via compat override", () => {
const model = getBundledModel("opencode-go", "mimo-v2.5-pro") as Model<"openai-completions">;
expect(model.compat?.supportsToolChoice).toBe(false);
});
it("omits tool_choice from MiMo title-style payloads while preserving tools", async () => {
const model = getBundledModel("opencode-go", "mimo-v2.5-pro") as Model<"openai-completions">;
const body = await capturePayload(model, { reasoning: "high", toolChoice: { type: "tool", name: "echo" } });
expect(body.tools).toBeDefined();
expect(body.tool_choice).toBeUndefined();
expect(body.reasoning_effort).toBe("high");
});
it("omits tool_choice from DeepSeek payloads but preserves tools and reasoning_effort", async () => {
const model = getBundledModel("opencode-go", "deepseek-v4-pro") as Model<"openai-completions">;
const body = await capturePayload(model, { reasoning: "high", toolChoice: "auto" });
expect(body.tools).toBeDefined();
expect(body.tool_choice).toBeUndefined();
expect(body.reasoning_effort).toBe("high");
@@ -33,6 +33,7 @@ const compat: ResolvedOpenAICompat = {
reasoningEffortMap: {},
supportsUsageInStreaming: true,
supportsToolChoice: true,
supportsForcedToolChoice: true,
disableReasoningOnForcedToolChoice: false,
disableReasoningOnToolChoice: false,
maxTokensField: "max_completion_tokens",
@@ -11,6 +11,7 @@ import type {
Model,
ModelSpec,
OpenAICompat,
Tool,
ToolResultMessage,
} from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
@@ -134,6 +135,7 @@ describe("openai-completions compatibility", () => {
reasoningEffortMap: {},
supportsUsageInStreaming: true,
supportsToolChoice: true,
supportsForcedToolChoice: true,
disableReasoningOnForcedToolChoice: false,
disableReasoningOnToolChoice: false,
maxTokensField: "max_completion_tokens",
@@ -1029,6 +1031,16 @@ describe("kimi model detection via detectCompat", () => {
timestamp: Date.now(),
};
const readTool: Tool = {
name: "read",
description: "Read a file",
parameters: {
type: "object",
properties: { path: { type: "string" } },
required: ["path"],
},
};
const { promise, resolve } = Promise.withResolvers<unknown>();
const fetchMock = createMockFetch(["[DONE]"]);
streamOpenAICompletions(
@@ -1046,6 +1058,7 @@ describe("kimi model detection via detectCompat", () => {
timestamp: Date.now(),
},
],
tools: [readTool],
},
{
apiKey: "test-key",
@@ -1073,6 +1086,96 @@ describe("kimi model detection via detectCompat", () => {
expect(payload.reasoning_effort).toBeUndefined();
});
it("downgrades unsupported forced tool_choice without suppressing thinking", async () => {
const model: Model<"openai-completions"> = buildModel({
...gpt4oMiniSpec,
api: "openai-completions",
provider: "opencode-go",
baseUrl: "https://opencode.ai/zen/go/v1",
id: "kimi-k2.7-code",
reasoning: true,
compat: { supportsForcedToolChoice: false },
} as ModelSpec<"openai-completions">);
const priorAssistant: AssistantMessage = {
role: "assistant",
content: [
{
type: "thinking",
thinking: "Plan first, then call the tool.",
thinkingSignature: "reasoning_content",
},
{
type: "toolCall",
id: "call_abc123",
name: "read",
arguments: { path: "README.md" },
},
],
api: model.api,
provider: model.provider,
model: model.id,
usage: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
},
stopReason: "toolUse",
timestamp: Date.now(),
};
const readTool: Tool = {
name: "read",
description: "Read a file",
parameters: {
type: "object",
properties: { path: { type: "string" } },
required: ["path"],
},
};
const { promise, resolve } = Promise.withResolvers<unknown>();
const fetchMock = createMockFetch(["[DONE]"]);
streamOpenAICompletions(
model,
{
messages: [
{ role: "user", content: "Summarize the README", timestamp: Date.now() },
priorAssistant,
{
role: "toolResult",
toolCallId: "call_abc123",
toolName: "read",
content: [{ type: "text", text: "# Hello\n" }],
isError: false,
timestamp: Date.now(),
},
],
tools: [readTool],
},
{
apiKey: "test-key",
fetch: fetchMock,
reasoning: "high",
toolChoice: { type: "tool", name: "read" },
signal: createAbortedSignal(),
onPayload: payload => resolve(payload),
},
);
const payload = (await promise) as {
messages: Array<Record<string, unknown>>;
reasoning_effort?: unknown;
tool_choice?: unknown;
};
const assistant = payload.messages.find(m => m.role === "assistant");
expect(assistant).toBeDefined();
expect(Reflect.get(assistant as object, "reasoning_content")).toBe("Plan first, then call the tool.");
expect(payload.reasoning_effort).toBe("high");
expect(payload.tool_choice).toBe("auto");
});
// #1484 follow-up: DeepSeek V4 on opencode-go exhibits the same gateway
// invariant as Kimi (same Zen gateway). DeepSeek emits reasoning under the
// `reasoning` signature, so the pre-fix code wrote both `reasoning` and
@@ -22,6 +22,7 @@ const compat: ResolvedOpenAICompat = {
reasoningEffortMap: {},
supportsUsageInStreaming: true,
supportsToolChoice: true,
supportsForcedToolChoice: true,
disableReasoningOnForcedToolChoice: false,
disableReasoningOnToolChoice: false,
maxTokensField: "max_completion_tokens",
@@ -334,6 +334,28 @@ describe("processResponsesStream: lost output_item.added recovery", () => {
).rejects.toThrow("incomplete: content_filter");
});
test("handles nested error object in error events", async () => {
const output = makeOutput();
const stream = { push: () => {}, end: () => {} } as never;
await expect(
processResponsesStream(
makeStream([
{
type: "error",
error: {
code: "context_length_exceeded",
message: "Your input exceeds the context window limit",
},
},
]),
output,
stream,
makeModel(),
),
).rejects.toThrow("Error Code context_length_exceeded: Your input exceeds the context window limit");
});
test("preserves premiumRequests across usage population", async () => {
const output = makeOutput();
output.usage.premiumRequests = 3;
+58 -99
View File
@@ -4,115 +4,18 @@
### Added
- Added `modelFamilyToken(modelId)` to `@oh-my-pi/pi-catalog/identity`: a coarse vendor-lineage token (`anthropic`/`openai`/`gemini`/`kimi`/…) for "are two models the same family?" comparisons, backed by `parseKnownModel` canonical-id normalization. Opaque and comparison-only; kind/variant collapsed onto the vendor token ([#2406](https://github.com/can1357/oh-my-pi/issues/2406))
- Added GLM-5.2 to the bundled zai (GLM Coding Plan) catalog as the selectable 1M served model.
### Changed
- Pinned zai `glm-5.2` to 1M context during catalog generation so endpoint discovery and older fallbacks cannot regress it to 200k.
## [15.12.4] - 2026-06-13
### Added
- Added bundled Fireworks models `deepseek-v4-flash`, `kimi-k2.7-code`, `minimax-m2.5`, `minimax-m3`, `nemotron-3-ultra-nvfp4`, `qwen3.6-plus`, and `qwen3.7-plus`
- Changed
### Changed
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
- Changed the `github-copilot` model context window to `524288` tokens
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
### Fixed
- Filled missing `contextWindow` and `maxTokens` in generated `models.json` for proxy/reseller variants by inheriting limits from canonical-family and segment-reference models
- Ignored zero-cost `x-ai` subscription entries as reference sources when backfilling limits so inflated values are not propagated
- Fixed the model cache opening with `PRAGMA journal_mode=WAL` before `PRAGMA busy_timeout`, so concurrent omp startups could crash inside `getDb()` on `SQLITE_BUSY` during WAL recovery instead of waiting through the transient lock. The busy handler is now installed before the first lock-taking statement ([#2421](https://github.com/can1357/oh-my-pi/issues/2421)).
## [15.11.8] - 2026-06-12
### Fixed
- Fixed Antigravity `gemini-3.1-pro --thinking high` failing with `Cloud Code Assist API error (400): Request contains an invalid argument.` — the upstream `gemini-3.1-pro-high` deployment rejects every `streamGenerateContent` request on both CCA endpoints while discovery still advertises it. High effort now routes to `gemini-pro-agent` (the same "Gemini 3.1 Pro (High)" model, verified accepting the identical request body), and the model-cache fingerprint version was bumped (`merge-v2` → `merge-v3`) so existing fresh caches refetch discovery and pick up the corrected routing immediately.
## [15.11.7] - 2026-06-12
### Added
- Added effort-tier variant collapsing (`variant-collapse`): providers that expose one logical model as several effort/thinking-suffixed upstream ids (Antigravity CCA `gemini-3.5-flash-extra-low`/`-low`/`gemini-3-flash-agent`, `gemini-3[.1]-pro-low|high`, `claude-*[-thinking]` pairs, `gpt-oss-120b-medium`) collapse into one logical entry carrying per-effort upstream routing in `thinking.effortRouting` (plus `thinking.suppressWhenOff` for Cloud Code Assist ids whose baked server default re-applies when `thinkingConfig` is omitted). Request-time code resolves the outbound id via `resolveWireModelId(model, effort)`; selection, caching, and usage attribution key on the logical id.
- Added the automatic `X`/`X-thinking` pair rule (`deriveThinkingPairFamilies`): any provider's live bare/thinking twin collapses into the bare id, routing thinking-enabled requests to the `-thinking` backing id (trailing or infix token, so `kimi-k2-thinking-turbo` pairs with `kimi-k2-turbo`). Gated on same api and compatible pricing — all-zero cost rows count as unknown, while twins that both carry real, differing prices remain separate SKUs.
- Added `collapseBuiltModelVariants` and wired collapsing at every materialization point — Antigravity discovery, the catalog generator, and the model-manager merge — so stale sources (old static beside collapsed dynamic results, mixed cache rows) converge on logical entries instead of unioning raw tier ids back into the catalog.
- Added `thinking.requiresEffort`, baked for reasoning-only upstreams — Gemini 3.x (levels only, no off), Gemini 2.5 Pro (thinkingBudget floors at 128, rejects 0), OpenAI o-series, MiniMax M2, and thinking-variant SKUs (`*-thinking`/`*-reasoner`/`*-reasoning`, with a negation-aware token grammar so `non-thinking` ids never match). Identity derivation bakes it for new entries and `fillThinkingWireDefaults` backfills explicit/cached metadata; `minimumSupportedEffort` exposes the canonical floor. Pair-collapsed twins drop member flags (their off routes to the bare SKU), while identity re-flags pairs whose logical id is itself mandatory
### Changed
- Changed model display names to drop model-extrinsic decorations: gateway author prefixes (`OpenAI: …`, `Google: …`), `(latest)` alias markers, `(Antigravity)` provider attribution, price tiers (`($$$$)`), and promo/lifecycle tags (`(20% off)`, `(retires …)`). `cleanModelName` is applied in `buildModel` (covers live discovery and stale caches) and as a catalog-generator pass; Antigravity discovery no longer appends `(Antigravity)` to display names. Variant tags that map to distinct wire ids (`(Thinking)`, `(free)`, `(Fast)`, dates, regions) are preserved.
- Changed the `google-antigravity` default model from `gemini-3-pro-high` to `gemini-3.1-pro`
- Changed `gemini-2.5-flash-thinking` handling from discovery-denylist to collapsing into `gemini-2.5-flash` (thinking-enabled requests route to the `-thinking` backing id)
- Bumped the model cache schema to v5 so rows predating effort-tier variant collapsing (raw `-low`/`-high`/`-thinking` member ids) are invalidated
### Fixed
- Fixed catalog generation to apply effort-tier variant collapsing before provider grouping to ensure collapsed model families are consistently materialized without being impacted by in-loop mutation
- Fixed Kimi K2.6 OpenAI-compatible compat metadata to use a 300s stream watchdog floor, covering Fire Pass router ids as well as public `kimi-k2.6` ids so long reasoning starts do not hit the generic first-event timeout ([#2366](https://github.com/can1357/oh-my-pi/issues/2366)).
## [15.11.4] - 2026-06-12
### Fixed
- Fixed MiniMax M2-family and OpenAI gpt-oss model metadata so OpenAI-compatible catalog entries declare only `low|medium|high` thinking efforts. Their upstreams reject `minimal`, `xhigh`, and Fireworks' `minimal → none` wire mapping, so `fireworks/minimax-m2.7` as the smol auto-thinking classifier model 400ed on every turn. OpenAI-compatible provider effort maps (`Groq qwen/qwen3-32b`, DeepSeek-family, OpenRouter Anthropic adaptive, Fireworks `minimal → none`) now bake into `thinking.effortMap` in catalog metadata instead of `buildOpenAICompat`, and request builders read that field directly. Regenerated `models.json` now makes `disableReasoning` choose `low` for those families while leaving GLM-5.x and other Fireworks models on the existing `minimal → none` path ([#2315](https://github.com/can1357/oh-my-pi/issues/2315)).
### Added
- Added `requiresJuiceZeroHack` Responses-API compat flag, resolved by `buildOpenAIResponsesCompat` from GPT-5-family model names and overridable via sparse model `compat` config. Replaces the request-time `model.name.startsWith("gpt-5")` sniff that gated the trailing `# Juice: 0 !important` no-reasoning developer item.
## [15.11.3] - 2026-06-11
### Added
- Added `requestModelId` on `Model` to represent the upstream model id used when a catalog entry is a local variant
- Added synthetic GitHub Copilot long-context model variants with `-1m` suffixes when tiered token pricing is advertised
### Changed
- Changed GitHub Copilot discovery to request `X-GitHub-Api-Version: 2026-06-01` from `api.githubcopilot.com`
- Changed GitHub Copilot discovery to cap base model `contextWindow` to the default token tier and keep long-context access as the separate `-1m` model entry
- Changed Copilot model mapping to omit non-chat `/models` entries and enable image input for models whose capabilities indicate vision support
### Fixed
- Fixed long-context variant pricing to use `billing.token_prices.long_context` rates instead of default model pricing
- Fixed `mapModel` handling in OpenAI-compatible discovery so returning `null` now skips a model entry rather than falling back to defaults
- Fixed model ID precedence so a real upstream Copilot model id is kept when it conflicts with a synthesized `-1m` variant
## [15.11.1] - 2026-06-11
### Fixed
- Fixed NVIDIA NIM Qwen turns failing with `400 Validation: Unsupported parameter(s): enable_thinking`. NIM's chat-completions schema is `additionalProperties: false` and exposes thinking via the vLLM convention `chat_template_kwargs.enable_thinking`; `buildOpenAICompat` was sending top-level `enable_thinking` for every `qwen/*` id regardless of host. Registered `nvidia` as a known host (`integrate.api.nvidia.com`) and routed NVIDIA-hosted Qwen models to `thinkingFormat: "qwen-chat-template"` ([#2299](https://github.com/can1357/oh-my-pi/issues/2299)).
- Fixed Moonshot/Kimi native OpenAI-compatible request metadata so Kimi K2 uses `max_tokens` and omits OpenAI-only `store`, restoring first-turn output with `MOONSHOT_API_KEY` ([#2289](https://github.com/can1357/oh-my-pi/issues/2289)).
## [15.11.0] - 2026-06-10
### Fixed
- Fixed `buildModel` so malformed explicit thinking metadata without `efforts` is treated as sparse input and inferred instead of crashing during model resolution ([#2251](https://github.com/can1357/oh-my-pi/issues/2251)).
## [15.10.12] - 2026-06-10
### Added
- Added `grok-composer-2.5-fast` (Cursor "Composer 2.5 Fast") to the xAI Grok OAuth (SuperGrok) catalog: non-reasoning, text-only, 200K context.
### Changed
- Set every xAI Grok OAuth (SuperGrok) curated model's max output tokens to mirror its context window (`grok-build`, `grok-4.3`, `grok-4.20-0309-{reasoning,non-reasoning}`, `grok-4.20-multi-agent-0309`, `grok-composer-2.5-fast`), replacing the `8888` `UNK_MAX_TOKENS` placeholder (and a stale `30000` on three grok-4.x entries). xAI's OAuth `/v1/models` reports no per-request output limit, so the curated catalog now owns `maxTokens` like `contextWindow`, deterministic on both the static-seed and online-overlay paths; the `openai-responses` wire still clamps the actual request to `OPENAI_MAX_OUTPUT_TOKENS` (64k).
### Fixed
- Excluded zero-cost `xai-oauth` subscription entries from the model reference indexes (`buildModelReferenceIndex`, `createReferenceResolver`), so their zero pricing and context-window-sized `maxTokens` cannot outrank paid/public Grok references when resolving custom-provider model identities.
## [15.10.11] - 2026-06-10
### Added
- Added `hostMatchesUrl`, `modelMatchesHost`, and endpoint-shape helpers in the new `hosts` module for consistent provider/baseUrl matching
- `buildModel(spec)` (`build.ts`) is now the single Model constructor: it materializes the fully-resolved compat record and canonical thinking metadata exactly once (compat first, thinking derived from identity + resolved compat), so `Model.compat` is a required, complete `CompatOf<TApi>` (`ResolvedOpenAICompat`/`ResolvedOpenAIResponsesCompat`/`ResolvedAnthropicCompat`) and request-path code reads fields with zero URL parsing and zero per-request allocation. Sparse user/config overrides live on the new `ModelSpec<TApi>` input shape and survive on `Model.compatConfig` for introspection.
- Added `ResolvedAnthropicCompat.supportsSamplingParams` (Opus 4.7+/Fable/Mythos reject `temperature`/`top_p`/`top_k` with a 400), baked at build time from model identity so the request path stops re-parsing model ids.
@@ -124,6 +27,21 @@
### Changed
- Changed catalog metadata to update a model’s per-token pricing to input 0.09 and output 0.18
- Changed the same cataloged model’s maximum token limit from 384000 to 65536
- Pinned zai `glm-5.2` to 1M context during catalog generation so endpoint discovery and older fallbacks cannot regress it to 200k.
- Replaced the hand-maintained `zhipu-coding-plan` GLM reasoning allowlist and vision regex with a `parseGlmModel` family classifier in `identity/classify.ts` (variant + vision + version), surfaced as `isReasoningGlmModelId` / `isGlmVisionModelId`. Discovery now derives reasoning/vision capability from the GLM family instead of a per-id list, so newly-bumped integers (`glm-5.3`, `glm-6`, …) are covered automatically while `-flash`/`-preview` and the vision `…v` shape stay correctly classified.
- Model `contextWindow`/`maxTokens` are now `number | null`; discovery emits `null` when a provider reports no limit, replacing the `222222`/`8888` (`UNK_CONTEXT_WINDOW`/`UNK_MAX_TOKENS`) sentinels (now removed). Bundled `models.json` unknown limits are `null`.
- Changed the `github-copilot` model context window to `524288` tokens
- Changed Fireworks model discovery to source the control-plane `List Models` API (`GET /v1/accounts/fireworks/models?filter=supports_serverless=true`) instead of the OpenAI-compatible `/v1/models` inference listing. The inference endpoint returns a sparse, account-specific subset that omits on-demand serverless models (e.g. `kimi-k2.7-code`), so newly published serverless models stayed invisible in the picker until hand-added to the bundled catalog. The control-plane catalog enumerates every serverless model with capability metadata (`supportsServerless`/`supportsTools`/`supportsImageInput`/`contextLength`/`displayName`), paginated and filtered to tool-capable `READY` entries, then merged with bundled/models.dev references — the Kimi K2 max-output clamp and DeepSeek V4 thinking-toggle strip are preserved, and unbundled models default to reasoning so `buildModel` derives the Fireworks effort map. New serverless releases now surface automatically with no catalog edits.
- Changed model display names to drop model-extrinsic decorations: gateway author prefixes (`OpenAI: …`, `Google: …`), `(latest)` alias markers, `(Antigravity)` provider attribution, price tiers (`($$$$)`), and promo/lifecycle tags (`(20% off)`, `(retires …)`). `cleanModelName` is applied in `buildModel` (covers live discovery and stale caches) and as a catalog-generator pass; Antigravity discovery no longer appends `(Antigravity)` to display names. Variant tags that map to distinct wire ids (`(Thinking)`, `(free)`, `(Fast)`, dates, regions) are preserved.
- Changed the `google-antigravity` default model from `gemini-3-pro-high` to `gemini-3.1-pro`
- Changed `gemini-2.5-flash-thinking` handling from discovery-denylist to collapsing into `gemini-2.5-flash` (thinking-enabled requests route to the `-thinking` backing id)
- Bumped the model cache schema to v5 so rows predating effort-tier variant collapsing (raw `-low`/`-high`/`-thinking` member ids) are invalidated
- Changed GitHub Copilot discovery to request `X-GitHub-Api-Version: 2026-06-01` from `api.githubcopilot.com`
- Changed GitHub Copilot discovery to cap base model `contextWindow` to the default token tier and keep long-context access as the separate `-1m` model entry
- Changed Copilot model mapping to omit non-chat `/models` entries and enable image input for models whose capabilities indicate vision support
- Set every xAI Grok OAuth (SuperGrok) curated model's max output tokens to mirror its context window (`grok-build`, `grok-4.3`, `grok-4.20-0309-{reasoning,non-reasoning}`, `grok-4.20-multi-agent-0309`, `grok-composer-2.5-fast`), replacing the `8888` `UNK_MAX_TOKENS` placeholder (and a stale `30000` on three grok-4.x entries). xAI's OAuth `/v1/models` reports no per-request output limit, so the curated catalog now owns `maxTokens` like `contextWindow`, deterministic on both the static-seed and online-overlay paths; the `openai-responses` wire still clamps the actual request to `OPENAI_MAX_OUTPUT_TOKENS` (64k).
- Changed OpenAI compatibility detection to use shared host classifiers (`modelMatchesHost`/`hostMatchesUrl`) with normalized matching instead of raw URL substring checks
- Changed `hostMatchesUrl`/`modelMatchesHost` usage in compatibility detection to reduce mismatches across case variants and provider alias hosts
- Provider catalog entries now carry the runtime API-key env fallback as an ordered `envVars` list; `catalogDiscovery.envVars` became an optional generation-time override (only `cursor` and `vercel-ai-gateway` differ) and `PROVIDER_DESCRIPTORS` materializes the resolved list for `generate-models.ts`.
@@ -134,6 +52,25 @@
### Fixed
- Fixed MiniMax-M3 catalog context for `minimax` and `minimax-cn` to report the documented 1M long-context tier instead of the upstream 512K pricing boundary ([#2576](https://github.com/can1357/oh-my-pi/issues/2576)).
- Fixed OpenCode Go MiMo catalog metadata so title generation and other tool-enabled calls omit unsupported `tool_choice` instead of triggering provider 400s ([#2509](https://github.com/can1357/oh-my-pi/issues/2509)).
- Fixed OpenCode Go `kimi-k2.7-code` catalog metadata so resolve-gate requests use automatic tool selection instead of Moonshot-rejected forced `tool_choice` ([#2546](https://github.com/can1357/oh-my-pi/issues/2546)).
- Fixed Anthropic compat for the `github-copilot` host so `supportsEagerToolInputStreaming` defaults to `false` there, matching the Copilot proxy which rejects the per-tool `eager_input_streaming` field ([#2558](https://github.com/can1357/oh-my-pi/issues/2558)).
- Scoped vLLM model cache validity to the discovery base URL so changed endpoints refetch immediately, and bounded built-in vLLM discovery requests with a timeout.
- Filled missing `contextWindow` and `maxTokens` in generated `models.json` for proxy/reseller variants by inheriting limits from canonical-family and segment-reference models
- Ignored zero-cost `x-ai` subscription entries as reference sources when backfilling limits so inflated values are not propagated
- Fixed the model cache opening with `PRAGMA journal_mode=WAL` before `PRAGMA busy_timeout`, so concurrent omp startups could crash inside `getDb()` on `SQLITE_BUSY` during WAL recovery instead of waiting through the transient lock. The busy handler is now installed before the first lock-taking statement ([#2421](https://github.com/can1357/oh-my-pi/issues/2421)).
- Fixed Antigravity `gemini-3.1-pro --thinking high` failing with `Cloud Code Assist API error (400): Request contains an invalid argument.` — the upstream `gemini-3.1-pro-high` deployment rejects every `streamGenerateContent` request on both CCA endpoints while discovery still advertises it. High effort now routes to `gemini-pro-agent` (the same "Gemini 3.1 Pro (High)" model, verified accepting the identical request body), and the model-cache fingerprint version was bumped (`merge-v2` → `merge-v3`) so existing fresh caches refetch discovery and pick up the corrected routing immediately.
- Fixed catalog generation to apply effort-tier variant collapsing before provider grouping to ensure collapsed model families are consistently materialized without being impacted by in-loop mutation
- Fixed Kimi K2.6 OpenAI-compatible compat metadata to use a 300s stream watchdog floor, covering Fire Pass router ids as well as public `kimi-k2.6` ids so long reasoning starts do not hit the generic first-event timeout ([#2366](https://github.com/can1357/oh-my-pi/issues/2366)).
- Fixed MiniMax M2-family and OpenAI gpt-oss model metadata so OpenAI-compatible catalog entries declare only `low|medium|high` thinking efforts. Their upstreams reject `minimal`, `xhigh`, and Fireworks' `minimal → none` wire mapping, so `fireworks/minimax-m2.7` as the smol auto-thinking classifier model 400ed on every turn. OpenAI-compatible provider effort maps (`Groq qwen/qwen3-32b`, DeepSeek-family, OpenRouter Anthropic adaptive, Fireworks `minimal → none`) now bake into `thinking.effortMap` in catalog metadata instead of `buildOpenAICompat`, and request builders read that field directly. Regenerated `models.json` now makes `disableReasoning` choose `low` for those families while leaving GLM-5.x and other Fireworks models on the existing `minimal → none` path ([#2315](https://github.com/can1357/oh-my-pi/issues/2315)).
- Fixed long-context variant pricing to use `billing.token_prices.long_context` rates instead of default model pricing
- Fixed `mapModel` handling in OpenAI-compatible discovery so returning `null` now skips a model entry rather than falling back to defaults
- Fixed model ID precedence so a real upstream Copilot model id is kept when it conflicts with a synthesized `-1m` variant
- Fixed NVIDIA NIM Qwen turns failing with `400 Validation: Unsupported parameter(s): enable_thinking`. NIM's chat-completions schema is `additionalProperties: false` and exposes thinking via the vLLM convention `chat_template_kwargs.enable_thinking`; `buildOpenAICompat` was sending top-level `enable_thinking` for every `qwen/*` id regardless of host. Registered `nvidia` as a known host (`integrate.api.nvidia.com`) and routed NVIDIA-hosted Qwen models to `thinkingFormat: "qwen-chat-template"` ([#2299](https://github.com/can1357/oh-my-pi/issues/2299)).
- Fixed Moonshot/Kimi native OpenAI-compatible request metadata so Kimi K2 uses `max_tokens` and omits OpenAI-only `store`, restoring first-turn output with `MOONSHOT_API_KEY` ([#2289](https://github.com/can1357/oh-my-pi/issues/2289)).
- Fixed `buildModel` so malformed explicit thinking metadata without `efforts` is treated as sparse input and inferred instead of crashing during model resolution ([#2251](https://github.com/can1357/oh-my-pi/issues/2251)).
- Excluded zero-cost `xai-oauth` subscription entries from the model reference indexes (`buildModelReferenceIndex`, `createReferenceResolver`), so their zero pricing and context-window-sized `maxTokens` cannot outrank paid/public Grok references when resolving custom-provider model identities.
- Fixed Anthropic official-endpoint detection to require strict HTTPS hostname matching so non-official or lookalike URLs are no longer treated as official Anthropic hosts
- Fixed Ollama Cloud dynamic discovery so same-id matches from other providers no longer supply context-window or max-output-token limits for discovered models.
- Wired `@oh-my-pi/pi-catalog` into the release publish package list, tarball install smoke test, and root `bun generate-models` script.
@@ -142,4 +79,26 @@
### Removed
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
- Removed the runtime enrichment layer: `enrichModelThinking` (and its non-enumerable memo-slot cache), `refreshModelThinking`, `modelOmitsReasoningEffort`, and the `model-thinking` re-exports of generator-only policies. Thinking metadata is resolved exactly once inside `buildModel`; runtime helpers (`getSupportedEfforts`, `clampThinkingLevelForModel`, `requireSupportedEffort`, the effort mappers) are pure field reads.
## [15.13.0] - 2026-06-14
## [15.12.6] - 2026-06-14
## [15.12.4] - 2026-06-13
## [15.11.8] - 2026-06-12
## [15.11.7] - 2026-06-12
## [15.11.4] - 2026-06-12
## [15.11.3] - 2026-06-11
## [15.11.1] - 2026-06-11
## [15.11.0] - 2026-06-10
## [15.10.12] - 2026-06-10
## [15.10.11] - 2026-06-10
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "15.12.5",
"version": "15.13.0",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+50 -11
View File
@@ -6,6 +6,7 @@
import { buildCompat } from "../src/build";
import {
type AnthropicModel,
bareModelId,
isFableOrMythos,
type OpenAIModel,
type OpenAIVariant,
@@ -14,6 +15,7 @@ import {
semverEqual,
} from "../src/identity/classify";
import { buildCanonicalModelIndex, buildCanonicalReferenceData } from "../src/identity/equivalence";
import { isMimoModelIdOrName } from "../src/identity/family";
import { getLongestModelLikeIdSegment } from "../src/identity/id";
import { buildModelReferenceIndex, resolveModelReference } from "../src/identity/reference";
import { resolveModelThinking } from "../src/model-thinking";
@@ -91,26 +93,46 @@ export function rebakeModelThinking(model: ModelSpec<Api>): void {
/**
* Link OpenAI model variants to their context promotion targets.
*
* When a model's context is exhausted, the agent can promote to a sibling
* model with a larger context window on the same provider:
* - `codex-spark` variants promote to `gpt-5.5`.
* - `gpt-5.5` (270K input) promotes to `gpt-5.4` (1M input).
* When a model's context is exhausted, the agent can promote to a sibling model
* on the same provider:
* - `codex-spark` variants promote to the full `gpt-5.5`.
* - every `gpt-5.5` flavor (base, `-pro`, `-instant`, dated snapshots, and
* namespaced ids like `openai/gpt-5.5`) promotes to its `gpt-5.4` sibling.
*
* The sibling is resolved by parsed version + matching provider/api, not a
* hardcoded bare id, so namespaced (`openrouter/openai/gpt-5.4`), dotted
* (`amazon-bedrock` `openai.gpt-5.4`), and dated (`gpt-5.4-2026-03-05`) ids all
* link. The runtime still gates on the target actually being larger
* (`#resolveContextPromotionTarget`), so an equal/smaller sibling is a harmless
* no-op rather than a counterproductive switch.
*/
export function linkOpenAIPromotionTargets(models: ModelSpec<Api>[]): void {
for (const candidate of models) {
const parsedCandidate = parseKnownModel(candidate.id);
if (parsedCandidate.family !== "openai") continue;
let targetId: string | undefined;
let targetVersion: string | undefined;
if (parsedCandidate.variant === "codex-spark") {
targetId = "gpt-5.5";
} else if (parsedCandidate.variant === "base" && semverEqual(parsedCandidate.version, "5.5")) {
targetId = "gpt-5.4";
targetVersion = "5.5";
} else if (semverEqual(parsedCandidate.version, "5.5")) {
targetVersion = "5.4";
} else {
continue;
}
const fallback = models.find(
model => model.provider === candidate.provider && model.api === candidate.api && model.id === targetId,
);
// Prefer the plainest sibling id (shortest bare segment) so the base model
// wins over `-pro`/`-mini`/`-nano` siblings that parse to the same version.
let fallback: ModelSpec<Api> | undefined;
let fallbackBareLength = Number.POSITIVE_INFINITY;
for (const model of models) {
if (model === candidate) continue;
if (model.provider !== candidate.provider || model.api !== candidate.api) continue;
const parsed = parseKnownModel(model.id);
if (parsed.family !== "openai" || !semverEqual(parsed.version, targetVersion)) continue;
const bareLength = bareModelId(model.id).length;
if (bareLength < fallbackBareLength) {
fallback = model;
fallbackBareLength = bareLength;
}
}
if (!fallback) continue;
candidate.contextPromotionTarget = `${fallback.provider}/${fallback.id}`;
}
@@ -190,6 +212,11 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
model.contextWindow = 1_000_000;
model.maxTokens = 131_072;
}
// MiniMax-M3: 512K is the standard pricing tier boundary, not the
// model ceiling. Pin the long-context providers to the documented 1M tier.
if ((model.provider === "minimax" || model.provider === "minimax-cn") && model.id === "MiniMax-M3") {
model.contextWindow = 1_000_000;
}
if (
model.api === "openai-completions" &&
@@ -204,6 +231,18 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
};
delete model.compat.thinkingFormat;
}
if (model.api === "openai-completions" && model.provider === "opencode-go" && isMimoModelIdOrName(model.id)) {
model.compat = {
...(model.compat ?? {}),
supportsToolChoice: false,
};
}
if (model.api === "openai-completions" && model.provider === "opencode-go" && model.id === "kimi-k2.7-code") {
model.compat = {
...(model.compat ?? {}),
supportsForcedToolChoice: false,
};
}
if (
model.api === "openai-completions" &&
model.provider === "opencode-go" &&
+7 -1
View File
@@ -34,11 +34,17 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res
const official = isOfficialAnthropicApiUrl(baseUrl);
// Z.AI's Anthropic-compatible proxy lives at `api.z.ai/api/anthropic`.
const isZai = modelMatchesHost(spec, "zai");
// GitHub Copilot's Anthropic-compatible proxy (api.githubcopilot.com/v1/messages)
// rejects the per-tool `eager_input_streaming` field with
// `tools.0.custom.eager_input_streaming: Extra inputs are not permitted` and
// doesn't whitelist the `fine-grained-tool-streaming-2025-05-14` beta either
// (issue #2558), so eager tool-input streaming is unavailable on this host.
const isCopilot = modelMatchesHost(spec, "githubCopilot");
const compat: ResolvedAnthropicCompat = {
officialEndpoint: official,
disableStrictTools: false,
disableAdaptiveThinking: false,
supportsEagerToolInputStreaming: true,
supportsEagerToolInputStreaming: !isCopilot,
// Long cache retention is only sent to the official API by default;
// proxies opt in explicitly via `compat.supportsLongCacheRetention: true`.
supportsLongCacheRetention: official,
+1
View File
@@ -217,6 +217,7 @@ export function buildOpenAICompat(spec: ModelSpec<"openai-completions">): Resolv
disableReasoningOnForcedToolChoice: isKimiModel || isAnthropicModel,
disableReasoningOnToolChoice: isDeepseekFamily && Boolean(spec.reasoning) && !isOpenRouter,
supportsToolChoice: !isDirectDeepseekReasoning,
supportsForcedToolChoice: true,
maxTokensField: useMaxTokens ? "max_tokens" : "max_completion_tokens",
requiresToolResultName: isMistral,
requiresAssistantAfterToolResult: false,
+59 -6
View File
@@ -14,6 +14,7 @@ export type SemVer = {
export type GeminiKind = "pro" | "flash";
export type AnthropicKind = "opus" | "sonnet" | "fable" | "mythos";
export type OpenAIVariant = "base" | "codex" | "codex-max" | "codex-mini" | "codex-spark" | "mini" | "max" | "nano";
export type GlmVariant = "base" | "air" | "turbo" | "flash" | "flashx" | "preview";
export interface GeminiModel {
family: "gemini";
@@ -33,6 +34,15 @@ export interface OpenAIModel {
version: SemVer;
}
export interface GlmModel {
family: "glm";
/** Suffix variant (`-air`, `-turbo`, `-flash`, `-flashx`, `-preview`); `base` when none. */
variant: GlmVariant;
/** Vision SKU — the `v` that attaches directly to the version (`glm-4v`, `glm-4.5v`). */
vision: boolean;
version: SemVer;
}
export interface UnknownModel {
family: "unknown";
id: string;
@@ -55,8 +65,26 @@ export function parseKnownModel(modelId: string): ParsedModel {
);
}
/**
* Wrap a parse function in a per-id memo cache. Caches the `null` result too, so
* repeated misses (the common case — ids of other families) stay O(1) and never
* re-run the regex/semver work.
*/
function parser<T>(parse: (modelId: string) => T | null): (modelId: string) => T | null {
const cache = new Map<string, T | null>();
return modelId => {
const hit = cache.get(modelId);
if (hit !== undefined || cache.has(modelId)) {
return hit ?? null;
}
const result = parse(modelId);
cache.set(modelId, result);
return result;
};
}
const GEMINI_SUFFIX = "-preview";
export function parseGeminiModel(modelId: string): GeminiModel | null {
export const parseGeminiModel = parser((modelId): GeminiModel | null => {
if (modelId.endsWith(GEMINI_SUFFIX)) {
modelId = modelId.slice(0, -GEMINI_SUFFIX.length);
}
@@ -69,9 +97,9 @@ export function parseGeminiModel(modelId: string): GeminiModel | null {
return null;
}
return { family: "gemini", kind: match[2] as GeminiKind, version };
}
});
export function parseAnthropicModel(modelId: string): AnthropicModel | null {
export const parseAnthropicModel = parser((modelId): AnthropicModel | null => {
const match = /claude-(opus|sonnet|fable|mythos)-(\d{1,2}(?:[.-]\d{1,2}){0,2})\b/.exec(modelId);
if (!match) {
return null;
@@ -81,9 +109,9 @@ export function parseAnthropicModel(modelId: string): AnthropicModel | null {
return null;
}
return { family: "anthropic", kind: match[1] as AnthropicKind, version };
}
});
export function parseOpenAIModel(modelId: string): OpenAIModel | null {
export const parseOpenAIModel = parser((modelId): OpenAIModel | null => {
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))?\b/.exec(modelId);
if (!match) {
return null;
@@ -93,7 +121,32 @@ export function parseOpenAIModel(modelId: string): OpenAIModel | null {
return null;
}
return { family: "openai", variant: (match[2] as OpenAIVariant | undefined) ?? "base", version };
}
});
/**
* Parse a GLM (Zhipu / Z.AI) model id into family + variant + vision + version.
* Shape: `glm-<version>[v][-<variant>]` — e.g. `glm-4.5`, `glm-4.5-air`,
* `glm-5-turbo`, `glm-4.5v`, `glm-5-preview`. The `v` (vision) attaches to the
* version; other variants are `-` suffixes. Standalone like `parseAnthropicModel`
* is used in family.ts — GLM needs no global thinking policy, so it stays out of
* `parseKnownModel`.
*/
export const parseGlmModel = parser((modelId): GlmModel | null => {
const match = /glm-(\d{1,2}(?:\.\d+)?)(v)?(?:-(air|turbo|flashx|flash|preview))?\b/.exec(modelId);
if (!match) {
return null;
}
const version = parseSemVer(match[1]);
if (!version) {
return null;
}
return {
family: "glm",
variant: (match[3] as GlmVariant | undefined) ?? "base",
vision: match[2] === "v",
version,
};
});
export function isFableOrMythos(kind: AnthropicKind): boolean {
return kind === "fable" || kind === "mythos";
+54 -1
View File
@@ -7,7 +7,14 @@
* here.
*/
import { bareModelId, isFableOrMythos, parseAnthropicModel, semverGte } from "./classify";
import {
bareModelId,
isFableOrMythos,
parseAnthropicModel,
parseGlmModel,
parseKnownModel,
semverGte,
} from "./classify";
/** Kimi family ids in any namespace form (`moonshotai/kimi-*`, `kimi-k2.6`, `vendor/kimi.x`). */
export function isKimiModelId(modelId: string): boolean {
@@ -71,6 +78,52 @@ export function isOpenAIGptOssModelId(modelId: string): boolean {
return /(^|\/)gpt-oss[-:]/i.test(modelId);
}
/**
* Reasoning-capable GLM coding SKUs: glm-4.5 and up on the base / `-air` /
* `-turbo` lines. Excludes the vision (`…v`) shape, the non-reasoning
* `-flash`/`-flashx`/`-preview` variants, and pre-4.5 ids. Matching the family
* keeps newly-bumped integers (`glm-5.3`, `glm-6`, …) covered without a per-id
* allowlist.
*/
export function isReasoningGlmModelId(modelId: string): boolean {
const glm = parseGlmModel(bareModelId(modelId));
if (!glm || glm.vision) {
return false;
}
if (glm.variant !== "base" && glm.variant !== "air" && glm.variant !== "turbo") {
return false;
}
return semverGte(glm.version, "4.5");
}
/** GLM vision SKUs — the `v` that attaches to the version (`glm-4v`, `glm-4.5v`). */
export function isGlmVisionModelId(modelId: string): boolean {
return parseGlmModel(bareModelId(modelId))?.vision === true;
}
/**
* Coarse vendor-lineage token for "are two models the same family?" checks
* (e.g. picking a cross-family reviewer). All Claude point releases share a token,
* Claude and GPT differ; namespace prefixes and aggregator mirrors fold onto the
* lineage via {@link parseKnownModel}'s `bareModelId` normalization. Opaque and
* comparison-only — not a stable key to persist, since the vocabulary tracks new
* releases. Returns `""` for ids it cannot classify; callers fall back to the provider.
*
* Vendor-only by design: a model's kind/variant (opus vs sonnet, codex vs base) is
* collapsed onto the single vendor token; use {@link parseKnownModel} for finer breakdowns.
*/
export function modelFamilyToken(modelId: string): string {
const parsed = parseKnownModel(modelId);
if (parsed.family !== "unknown") return parsed.family;
if (isKimiModelId(modelId)) return "kimi";
if (isQwenModelId(modelId)) return "qwen";
if (isMinimaxM2FamilyModelId(modelId)) return "minimax";
if (isOpenAIGptOssModelId(modelId)) return "gpt-oss";
if (isDeepseekModelIdOrName(modelId)) return "deepseek";
if (isMimoModelIdOrName(modelId)) return "mimo";
if (parseGlmModel(bareModelId(modelId))) return "glm";
return "";
}
/**
* Adaptive thinking `display` is supported starting with Claude Opus 4.7 and
* the Claude Fable/Mythos 5 generation. Older adaptive-thinking models
+7 -4
View File
@@ -33,6 +33,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
staticModels?: readonly ModelSpec<TApi>[];
/** Optional override for the cache database path. Default: <agent-dir>/models.db. */
cacheDbPath?: string;
/** Optional provider id override for cache namespacing. Defaults to providerId. */
cacheProviderId?: string;
/** Maximum cache age in milliseconds before considered stale. Default: 24h. */
cacheTtlMs?: number;
/** When true, a successful dynamic fetch is the complete provider catalog and prunes static-only models. */
@@ -107,13 +109,14 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
options: ModelManagerOptions<TApi, TModelsDevPayload>,
strategy: ModelRefreshStrategy = "online-if-uncached",
): Promise<ModelResolutionResult<TApi>> {
const cacheProviderId = options.cacheProviderId ?? options.providerId;
const now = options.now ?? Date.now;
const ttlMs = options.cacheTtlMs ?? DEFAULT_CACHE_TTL_MS;
const dbPath = options.cacheDbPath;
const staticModels = options.staticModels
? passModelList<TApi>(options.staticModels)
: (getBundledModels(options.providerId as GeneratedProvider) as Model<TApi>[]);
const cache = readModelCache<TApi>(options.providerId, ttlMs, now, dbPath);
const cache = readModelCache<TApi>(cacheProviderId, ttlMs, now, dbPath);
const dynamicModelsAuthoritative = options.dynamicModelsAuthoritative ?? false;
const staticFingerprint = fingerprintStatic(staticModels, dynamicModelsAuthoritative);
const cacheFingerprintMatches = cache?.staticFingerprint === staticFingerprint && staticFingerprint.length > 0;
@@ -160,7 +163,7 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
? retainModelIds(mergedSnapshot, dynamicModels)
: mergedSnapshot;
writeModelCache(
options.providerId,
cacheProviderId,
now(),
collapseBuiltModelVariants(snapshotModels),
true,
@@ -170,9 +173,9 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
} else {
// Dynamic fetch failed — update cache with a non-authoritative snapshot so
// stale state remains visible while retry backoff still applies.
const latestCache = readModelCache<TApi>(options.providerId, ttlMs, now, dbPath);
const latestCache = readModelCache<TApi>(cacheProviderId, ttlMs, now, dbPath);
writeModelCache(
options.providerId,
cacheProviderId,
now(),
collapseBuiltModelVariants(
mergeDynamicModels(
+59 -26
View File
@@ -4259,7 +4259,8 @@
"cacheWrite": 0
},
"contextWindow": null,
"maxTokens": null
"maxTokens": null,
"contextPromotionTarget": "aimlapi/gpt-5.4-2026-03-05"
},
"gpt-5.5-pro-2026-04-23": {
"id": "gpt-5.5-pro-2026-04-23",
@@ -4278,7 +4279,8 @@
"cacheWrite": 0
},
"contextWindow": null,
"maxTokens": null
"maxTokens": null,
"contextPromotionTarget": "aimlapi/gpt-5.4-2026-03-05"
},
"gpt-oss-120b": {
"id": "gpt-oss-120b",
@@ -9577,7 +9579,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "amazon-bedrock/openai.gpt-5.4"
},
"openai.gpt-oss-120b": {
"id": "openai.gpt-oss-120b",
@@ -12202,7 +12205,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "cloudflare-ai-gateway/openai/gpt-5.4"
},
"openai/o1": {
"id": "openai/o1",
@@ -22904,6 +22908,7 @@
"disableReasoningOnForcedToolChoice": true,
"disableReasoningOnToolChoice": false,
"supportsToolChoice": true,
"supportsForcedToolChoice": true,
"maxTokensField": "max_completion_tokens",
"requiresToolResultName": false,
"requiresAssistantAfterToolResult": false,
@@ -24595,7 +24600,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "kilo/openai/gpt-5.4"
},
"openai/gpt-5.5-pro": {
"id": "openai/gpt-5.5-pro",
@@ -24624,7 +24630,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "kilo/openai/gpt-5.4"
},
"openai/gpt-audio": {
"id": "openai/gpt-audio",
@@ -25327,6 +25334,7 @@
"disableReasoningOnForcedToolChoice": false,
"disableReasoningOnToolChoice": false,
"supportsToolChoice": true,
"supportsForcedToolChoice": true,
"maxTokensField": "max_completion_tokens",
"requiresToolResultName": false,
"requiresAssistantAfterToolResult": false,
@@ -25778,6 +25786,7 @@
"disableReasoningOnForcedToolChoice": false,
"disableReasoningOnToolChoice": false,
"supportsToolChoice": true,
"supportsForcedToolChoice": true,
"maxTokensField": "max_completion_tokens",
"requiresToolResultName": false,
"requiresAssistantAfterToolResult": false,
@@ -26058,6 +26067,7 @@
"disableReasoningOnForcedToolChoice": false,
"disableReasoningOnToolChoice": false,
"supportsToolChoice": true,
"supportsForcedToolChoice": true,
"maxTokensField": "max_completion_tokens",
"requiresToolResultName": false,
"requiresAssistantAfterToolResult": false,
@@ -28539,7 +28549,7 @@
"cacheRead": 0.12,
"cacheWrite": 0
},
"contextWindow": 512000,
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "budget",
@@ -28781,7 +28791,7 @@
"cacheRead": 0.12,
"cacheWrite": 0
},
"contextWindow": 512000,
"contextWindow": 1000000,
"maxTokens": 128000,
"thinking": {
"mode": "budget",
@@ -39194,8 +39204,8 @@
"cacheRead": 0,
"cacheWrite": 0
},
"contextWindow": null,
"maxTokens": null
"contextWindow": 128000,
"maxTokens": 16384
},
"openai/gpt-5-codex": {
"id": "openai/gpt-5-codex",
@@ -39763,7 +39773,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "nanogpt/openai/gpt-5.4"
},
"openai/gpt-chat-latest": {
"id": "openai/gpt-chat-latest",
@@ -51042,6 +51053,9 @@
},
"contextWindow": 262144,
"maxTokens": 262144,
"compat": {
"supportsForcedToolChoice": false
},
"thinking": {
"mode": "effort",
"efforts": [
@@ -51081,6 +51095,9 @@
"high",
"xhigh"
]
},
"compat": {
"supportsToolChoice": false
}
},
"mimo-v2-pro": {
@@ -51110,6 +51127,9 @@
"high",
"xhigh"
]
},
"compat": {
"supportsToolChoice": false
}
},
"mimo-v2.5": {
@@ -51131,6 +51151,9 @@
},
"contextWindow": 1000000,
"maxTokens": 128000,
"compat": {
"supportsToolChoice": false
},
"thinking": {
"mode": "effort",
"efforts": [
@@ -51160,6 +51183,9 @@
},
"contextWindow": 1048576,
"maxTokens": 128000,
"compat": {
"supportsToolChoice": false
},
"thinking": {
"mode": "effort",
"efforts": [
@@ -55012,13 +55038,13 @@
"text"
],
"cost": {
"input": 0.098,
"output": 0.196,
"input": 0.09,
"output": 0.18,
"cacheRead": 0.02,
"cacheWrite": 0
},
"contextWindow": 1048576,
"maxTokens": 384000,
"maxTokens": 65536,
"thinking": {
"mode": "effort",
"efforts": [
@@ -57075,9 +57101,9 @@
"image"
],
"cost": {
"input": 0.95,
"output": 4,
"cacheRead": 0.19,
"input": 0.75,
"output": 3.5,
"cacheRead": 0.16,
"cacheWrite": 0
},
"contextWindow": 262144,
@@ -58513,7 +58539,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "openrouter/openai/gpt-5.4"
},
"openai/gpt-5.5-pro": {
"id": "openai/gpt-5.5-pro",
@@ -58542,7 +58569,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "openrouter/openai/gpt-5.4"
},
"openai/gpt-audio": {
"id": "openai/gpt-audio",
@@ -59989,7 +60017,7 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": null
"maxTokens": 16384
},
"qwen/qwen3-next-80b-a3b-thinking": {
"id": "qwen/qwen3-next-80b-a3b-thinking",
@@ -64583,7 +64611,7 @@
"cacheWrite": 0
},
"contextWindow": 128000,
"maxTokens": null,
"maxTokens": 16384,
"compat": {
"supportsUsageInStreaming": false
}
@@ -69051,7 +69079,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "vercel-ai-gateway/openai/gpt-5.4"
},
"openai/gpt-5.5-pro": {
"id": "openai/gpt-5.5-pro",
@@ -69080,7 +69109,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "vercel-ai-gateway/openai/gpt-5.4"
},
"openai/gpt-oss-120b": {
"id": "openai/gpt-oss-120b",
@@ -75141,7 +75171,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "zenmux/openai/gpt-5.4"
},
"openai/gpt-5.5-instant": {
"id": "openai/gpt-5.5-instant",
@@ -75170,7 +75201,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "zenmux/openai/gpt-5.4"
},
"openai/gpt-5.5-pro": {
"id": "openai/gpt-5.5-pro",
@@ -75199,7 +75231,8 @@
"high",
"xhigh"
]
}
},
"contextPromotionTarget": "zenmux/openai/gpt-5.4"
},
"openai/gpt-image-1.5": {
"id": "openai/gpt-image-1.5",
@@ -68,11 +68,11 @@ export const CATALOG_PROVIDERS = [
},
{
id: "amazon-bedrock",
defaultModel: "us.anthropic.claude-opus-4-6-v1",
defaultModel: "us.anthropic.claude-opus-4-8",
},
{
id: "anthropic",
defaultModel: "claude-opus-4-6",
defaultModel: "claude-opus-4-8",
createModelManagerOptions: (config: ModelManagerConfig) => anthropicModelManagerOptions(config),
},
{
@@ -177,7 +177,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "litellm",
defaultModel: "claude-opus-4-6",
defaultModel: "claude-opus-4-8",
envVars: ["LITELLM_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => litellmModelManagerOptions(config),
catalogDiscovery: { label: "LiteLLM", allowUnauthenticated: true },
@@ -219,7 +219,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "nanogpt",
defaultModel: "openai/gpt-5.4",
defaultModel: "openai/gpt-5.5",
envVars: ["NANO_GPT_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => nanoGptModelManagerOptions(config),
catalogDiscovery: { label: "NanoGPT" },
@@ -247,13 +247,13 @@ export const CATALOG_PROVIDERS = [
},
{
id: "openai",
defaultModel: "gpt-5.4",
defaultModel: "gpt-5.5",
envVars: ["OPENAI_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => openaiModelManagerOptions(config),
},
{
id: "openai-codex",
defaultModel: "gpt-5.4",
defaultModel: "gpt-5.5",
envVars: ["OPENAI_CODEX_OAUTH_TOKEN"],
specialModelManager: true,
},
@@ -271,7 +271,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "openrouter",
defaultModel: "openai/gpt-5.4",
defaultModel: "openai/gpt-5.5",
envVars: ["OPENROUTER_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => openrouterModelManagerOptions(config),
catalogDiscovery: { label: "OpenRouter", allowUnauthenticated: true },
@@ -403,7 +403,7 @@ export const CATALOG_PROVIDERS = [
},
{
id: "zenmux",
defaultModel: "anthropic/claude-opus-4.6",
defaultModel: "anthropic/claude-opus-4.8",
envVars: ["ZENMUX_API_KEY"],
createModelManagerOptions: (config: ModelManagerConfig) => zenmuxModelManagerOptions(config),
catalogDiscovery: { label: "ZenMux" },
@@ -5,6 +5,7 @@ import {
} from "../discovery/openai-compatible";
import { Effort } from "../effort";
import { toFireworksPublicModelId } from "../fireworks-model-id";
import { isGlmVisionModelId, isReasoningGlmModelId } from "../identity/family";
import type { ModelManagerOptions } from "../model-manager";
import { getBundledModels } from "../models";
import type { Api, FetchImpl, Model, ModelSpec, Provider, ThinkingConfig } from "../types";
@@ -1030,8 +1031,8 @@ export function zhipuCodingPlanModelManagerOptions(
const id = defaults.id;
return {
...defaults,
reasoning: ZHIPU_REASONING_MODELS[id] === true || id.includes("thinking"),
input: ZHIPU_VISION_PATTERN.test(id) ? (["text", "image"] as const) : ["text"],
reasoning: isReasoningGlmModelId(id) || id.includes("thinking"),
input: isGlmVisionModelId(id) ? (["text", "image"] as const) : ["text"],
compat: {
thinkingFormat: "zai",
reasoningContentField: "reasoning_content",
@@ -1045,26 +1046,6 @@ export function zhipuCodingPlanModelManagerOptions(
};
}
// Reasoning-capable GLM models on the BigModel coding-plan SKU. Keep this
// explicit rather than regex-matching `glm-[45]\.\d` so newly-added integers
// like `glm-5` / `glm-5-turbo` are covered and unrelated future SKUs (e.g.
// `glm-5-preview`) do not silently flip into thinking mode.
const ZHIPU_REASONING_MODELS: Readonly<Record<string, true>> = {
"glm-4.5": true,
"glm-4.5-air": true,
"glm-4.6": true,
"glm-4.7": true,
"glm-5": true,
"glm-5-turbo": true,
"glm-5.1": true,
"glm-5.2": true,
};
// Vision-capable GLM models follow the `glm-<N>[.<N>]v[-<variant>]` shape
// (e.g. `glm-4v`, `glm-4.5v`, `glm-4v-plus`). The previous `id.includes("v")`
// check matched anything with a `v` — including the non-vision `glm-5-preview`.
const ZHIPU_VISION_PATTERN = /^glm-[45](?:\.\d+)?v(?:-|$)/;
// ---------------------------------------------------------------------------
// 7.5 Fireworks
// ---------------------------------------------------------------------------
@@ -2394,6 +2375,8 @@ export function litellmModelManagerOptions(
// 22. vLLM
// ---------------------------------------------------------------------------
const VLLM_DISCOVERY_TIMEOUT_MS = 10_000;
export interface VllmModelManagerConfig {
apiKey?: string;
baseUrl?: string;
@@ -2406,6 +2389,7 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
const references = createBundledReferenceMap<"openai-completions">("vllm" as Parameters<typeof getBundledModels>[0]);
return {
providerId: "vllm",
cacheProviderId: `vllm:${Bun.hash(baseUrl).toString(36)}`,
fetchDynamicModels: () =>
fetchOpenAICompatibleModels({
api: "openai-completions",
@@ -2420,6 +2404,7 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
};
},
fetch: config?.fetch,
signal: AbortSignal.timeout(VLLM_DISCOVERY_TIMEOUT_MS),
}),
};
}
+6
View File
@@ -186,6 +186,12 @@ export interface OpenAICompat {
requiresAssistantContentForToolCalls?: boolean;
/** Whether the provider supports the `tool_choice` parameter. Default: true. */
supportsToolChoice?: boolean;
/**
* Whether forced `tool_choice` values (`"required"` or named tools) are accepted.
* When false, request builders keep tools available but downgrade forced choices
* to provider-default auto selection. Default: true.
*/
supportsForcedToolChoice?: boolean;
/**
* Drop reasoning fields (`reasoning_effort`, OpenRouter `reasoning`) for
* the request when `tool_choice` forces a tool call. Mirrors the Anthropic
@@ -20,6 +20,8 @@ export const ANTIGRAVITY_SYSTEM_INSTRUCTION =
"You are pair programming with a USER to solve their coding task. The task may require creating a new codebase, modifying or debugging an existing codebase, or simply answering a question." +
"**Absolute paths only**" +
"**Proactiveness**";
export const ANTIGRAVITY_NO_PREAMBLE_INSTRUCTION =
'CRITICAL: NEVER output rule checks, formatting guidelines, constraint checklists (e.g. "No emdashes"), or your thinking/personality preambles in the final response. Output only the final response.';
/**
* Antigravity / Cloud Code Assist user agent. Lives in its own file so discovery
* and usage code can read it without pulling the heavy google-gemini-cli provider
+2 -2
View File
@@ -5,14 +5,14 @@ describe("catalog provider descriptors", () => {
test("descriptors cover standard model providers, excluding special-managed ones", () => {
const zenmux = PROVIDER_DESCRIPTORS.find(descriptor => descriptor.providerId === "zenmux");
expect(zenmux).toBeDefined();
expect(zenmux?.defaultModel).toBe("anthropic/claude-opus-4.6");
expect(zenmux?.defaultModel).toBe("anthropic/claude-opus-4.8");
// The descriptor factory carries the provider identity through.
expect(zenmux?.createModelManagerOptions({ apiKey: "k" }).providerId).toBe("zenmux");
// openai-codex is special-managed (bespoke runtime factory) → excluded from descriptors,
// but still a known model provider with a default.
expect(PROVIDER_DESCRIPTORS.some(descriptor => descriptor.providerId === "openai-codex")).toBe(false);
expect(DEFAULT_MODEL_PER_PROVIDER["openai-codex"]).toBe("gpt-5.4");
expect(DEFAULT_MODEL_PER_PROVIDER["openai-codex"]).toBe("gpt-5.5");
expect(DEFAULT_MODEL_PER_PROVIDER.minimax).toBe("MiniMax-M3");
expect(DEFAULT_MODEL_PER_PROVIDER["minimax-code"]).toBe("MiniMax-M3");
expect(DEFAULT_MODEL_PER_PROVIDER["minimax-code-cn"]).toBe("MiniMax-M3");
@@ -126,6 +126,41 @@ describe("generated model policies", () => {
expect(models[0]?.maxTokens).toBe(131_072);
});
it("pins MiniMax-M3 long-context providers to 1M context", () => {
const models = [
createSpec({
id: "MiniMax-M3",
api: "anthropic-messages",
provider: "minimax",
contextWindow: 512_000,
maxTokens: 128_000,
}),
createSpec({
id: "MiniMax-M3",
api: "anthropic-messages",
provider: "minimax-cn",
contextWindow: 512_000,
maxTokens: 128_000,
}),
createSpec({
id: "MiniMax-M3",
api: "openai-completions",
provider: "minimax-code",
contextWindow: 512_000,
maxTokens: 128_000,
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.contextWindow).toBe(1_000_000);
expect(models[0]?.maxTokens).toBe(128_000);
expect(models[1]?.contextWindow).toBe(1_000_000);
expect(models[1]?.maxTokens).toBe(128_000);
expect(models[2]?.contextWindow).toBe(512_000);
expect(models[2]?.maxTokens).toBe(128_000);
});
it("normalizes Copilot generated fallback limits", () => {
const models: ModelSpec<Api>[] = [
createSpec({
@@ -161,6 +196,34 @@ describe("generated model policies", () => {
expect(models[2]?.maxTokens).toBe(64000);
});
it("marks OpenCode Go MiMo models as not supporting tool_choice", () => {
const models: ModelSpec<"openai-completions">[] = [
createSpec({
id: "mimo-v2.5-pro",
api: "openai-completions",
provider: "opencode-go",
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.compat?.supportsToolChoice).toBe(false);
});
it("marks OpenCode Go Kimi K2.7 Code as not supporting forced tool_choice", () => {
const models: ModelSpec<"openai-completions">[] = [
createSpec({
id: "kimi-k2.7-code",
api: "openai-completions",
provider: "opencode-go",
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.compat?.supportsForcedToolChoice).toBe(false);
});
it("links spark variants and gpt-5.5 to their context promotion targets", () => {
const models = [
createSpec({ id: "gpt-5.3-codex-spark", api: "openai-codex-responses", provider: "openai-codex" }),
@@ -174,6 +237,38 @@ describe("generated model policies", () => {
expect(models[1]?.contextPromotionTarget).toBe("openai-codex/gpt-5.4");
});
it("links every gpt-5.5 flavor to its gpt-5.4 sibling across namespaced and dated provider ids", () => {
const models = [
// Namespaced provider ids (id carries an `openai/` prefix).
createSpec({ id: "openai/gpt-5.5", api: "openai-responses", provider: "openrouter" }),
createSpec({ id: "openai/gpt-5.5-pro", api: "openai-responses", provider: "openrouter" }),
createSpec({ id: "openai/gpt-5.4", api: "openai-responses", provider: "openrouter" }),
createSpec({ id: "openai/gpt-5.4-pro", api: "openai-responses", provider: "openrouter" }),
createSpec({ id: "openai/gpt-5.4-mini", api: "openai-responses", provider: "openrouter" }),
// Dated snapshot ids on a provider with no plain `gpt-5.4`.
createSpec({ id: "gpt-5.5-2026-04-23", api: "openai-responses", provider: "aimlapi" }),
createSpec({ id: "gpt-5.4-2026-03-05", api: "openai-responses", provider: "aimlapi" }),
// Dotted namespace (amazon-bedrock `openai.gpt-5.x`).
createSpec({ id: "openai.gpt-5.5", api: "openai-responses", provider: "amazon-bedrock" }),
createSpec({ id: "openai.gpt-5.4", api: "openai-responses", provider: "amazon-bedrock" }),
];
linkOpenAIPromotionTargets(models);
// Base and pro both promote to the plainest same-provider gpt-5.4 (base wins
// over `-pro`/`-mini`), and the namespaced target round-trips through
// parseModelString (first-slash split → provider `openrouter`, id `openai/gpt-5.4`).
expect(models[0]?.contextPromotionTarget).toBe("openrouter/openai/gpt-5.4");
expect(models[1]?.contextPromotionTarget).toBe("openrouter/openai/gpt-5.4");
// A gpt-5.4 model itself is never given a promotion target.
expect(models[2]?.contextPromotionTarget).toBeUndefined();
expect(models[3]?.contextPromotionTarget).toBeUndefined();
expect(models[4]?.contextPromotionTarget).toBeUndefined();
// Dated and dotted siblings resolve by parsed version, not literal id.
expect(models[5]?.contextPromotionTarget).toBe("aimlapi/gpt-5.4-2026-03-05");
expect(models[7]?.contextPromotionTarget).toBe("amazon-bedrock/openai.gpt-5.4");
});
it("sets freeform apply_patch metadata for first-party GPT-5 Responses models", () => {
const models: ModelSpec<Api>[] = [
createSpec({ id: "gpt-5.4", api: "openai-responses", provider: "openai" }),
@@ -1,10 +1,13 @@
import { describe, expect, test } from "bun:test";
import {
isClaudeModelId,
isGlmVisionModelId,
isKimiK26ModelId,
isKimiModelId,
isMinimaxM2FamilyModelId,
isOpenAIGptOssModelId,
isReasoningGlmModelId,
modelFamilyToken,
supportsAdaptiveThinkingDisplay,
} from "@oh-my-pi/pi-catalog/identity";
@@ -106,3 +109,75 @@ describe("isOpenAIGptOssModelId", () => {
expect(isOpenAIGptOssModelId("MiniMax-M2.7")).toBe(false);
});
});
describe("isReasoningGlmModelId", () => {
test("matches the glm-4.5+ base / air / turbo reasoning lines", () => {
expect(isReasoningGlmModelId("glm-4.5")).toBe(true);
expect(isReasoningGlmModelId("glm-4.5-air")).toBe(true);
expect(isReasoningGlmModelId("glm-4.6")).toBe(true);
expect(isReasoningGlmModelId("glm-4.7")).toBe(true);
expect(isReasoningGlmModelId("glm-5")).toBe(true);
expect(isReasoningGlmModelId("glm-5-turbo")).toBe(true);
expect(isReasoningGlmModelId("glm-5.1")).toBe(true);
expect(isReasoningGlmModelId("glm-5.2")).toBe(true);
// Family match is future-proof: new integers need no allowlist entry.
expect(isReasoningGlmModelId("glm-5.3")).toBe(true);
expect(isReasoningGlmModelId("glm-6")).toBe(true);
// Namespaced ids are stripped before classification.
expect(isReasoningGlmModelId("z-ai/glm-5-turbo")).toBe(true);
});
test("excludes pre-4.5, vision, flash, and preview SKUs", () => {
expect(isReasoningGlmModelId("glm-4")).toBe(false);
expect(isReasoningGlmModelId("glm-4.4")).toBe(false);
expect(isReasoningGlmModelId("glm-5-preview")).toBe(false);
expect(isReasoningGlmModelId("glm-4.5-flash")).toBe(false);
expect(isReasoningGlmModelId("glm-4.7-flashx")).toBe(false);
expect(isReasoningGlmModelId("glm-4.5v")).toBe(false);
expect(isReasoningGlmModelId("qwen3.5")).toBe(false);
});
});
describe("isGlmVisionModelId", () => {
test("matches the `v` vision shape across versions and variants", () => {
expect(isGlmVisionModelId("glm-4v")).toBe(true);
expect(isGlmVisionModelId("glm-4.5v")).toBe(true);
expect(isGlmVisionModelId("glm-4v-plus")).toBe(true);
});
test("excludes non-vision GLM ids (the old `includes('v')` false positives)", () => {
expect(isGlmVisionModelId("glm-5-preview")).toBe(false);
expect(isGlmVisionModelId("glm-4.5")).toBe(false);
expect(isGlmVisionModelId("glm-5-turbo")).toBe(false);
});
});
describe("modelFamilyToken", () => {
test("groups point releases within a vendor and separates across vendors", () => {
expect(modelFamilyToken("claude-opus-4-7")).toBe("anthropic");
expect(modelFamilyToken("claude-opus-4-8")).toBe("anthropic");
expect(modelFamilyToken("claude-opus-4-7")).toBe(modelFamilyToken("claude-opus-4-8"));
expect(modelFamilyToken("gpt-5.4")).toBe("openai");
expect(modelFamilyToken("gemini-3-pro")).toBe("gemini");
expect(modelFamilyToken("claude-opus-4-8")).not.toBe(modelFamilyToken("gpt-5.4"));
});
test("folds aggregator mirrors and namespace prefixes onto the lineage", () => {
expect(modelFamilyToken("anthropic/claude-opus-4.8")).toBe("anthropic");
expect(modelFamilyToken("openrouter/anthropic/claude-opus-4-8")).toBe("anthropic");
});
test("classifies non-first-party families", () => {
expect(modelFamilyToken("moonshotai/kimi-k2")).toBe("kimi");
expect(modelFamilyToken("qwen/qwen3-coder")).toBe("qwen");
});
test("classifies GLM across provider mirrors so same-lineage SKUs fold together", () => {
expect(modelFamilyToken("glm-5.2")).toBe("glm");
expect(modelFamilyToken("zai/glm-5.2")).toBe(modelFamilyToken("zhipu-coding-plan/glm-5.2"));
expect(modelFamilyToken("zai/glm-5.2")).toBe("glm");
});
test("returns an empty token for unclassifiable ids so callers fall back to provider", () => {
expect(modelFamilyToken("some-unknown-model")).toBe("");
});
});
@@ -0,0 +1,96 @@
/**
* Issue #2558 — `400 Error when using Claude Haiku 4.6 via Github Copilot`
*
* Reporter: sending any tool-bearing turn to a GitHub Copilot Claude model
* (e.g. `github-copilot/claude-haiku-4.5`) returns
* `400 tools.0.custom.eager_input_streaming: Extra inputs are not permitted`.
*
* Root cause: `buildAnthropicCompat` defaulted `supportsEagerToolInputStreaming`
* to `true` regardless of host. That made `convertTools` emit
* `eager_input_streaming: true` on every tool sent to
* `api.githubcopilot.com/v1/messages`, which the Copilot proxy rejects.
*
* Fix: turn the flag off for the `github-copilot` host in the Anthropic
* compat builder, AND stop pushing the legacy
* `fine-grained-tool-streaming-2025-05-14` beta header on the Copilot
* transport (the proxy doesn't whitelist Anthropic beta features either).
*/
import { describe, expect, it } from "bun:test";
import { buildAnthropicClientOptions, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
import type { Context, TJsonSchema, Tool } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import type { Model, ModelSpec } from "@oh-my-pi/pi-catalog/types";
const COPILOT_BEARER = JSON.stringify({ token: "ghc_test" });
const TOOLS: Tool[] = [
{
name: "ping",
description: "ping",
parameters: {
type: "object",
properties: { msg: { type: "string" } },
required: ["msg"],
} as TJsonSchema,
},
];
const COPILOT_MODEL_SPEC: ModelSpec<"anthropic-messages"> = {
id: "claude-haiku-4.5",
name: "Claude Haiku 4.5",
api: "anthropic-messages",
provider: "github-copilot",
baseUrl: "https://api.githubcopilot.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 200_000,
maxTokens: 8_192,
};
const CONTEXT: Context = {
systemPrompt: ["Stay concise."],
messages: [{ role: "user", content: "Hi", timestamp: Date.now() }],
tools: TOOLS,
};
function aborted(): AbortSignal {
const controller = new AbortController();
controller.abort();
return controller.signal;
}
describe("issue #2558 — GitHub Copilot Anthropic transport rejects eager_input_streaming", () => {
const model: Model<"anthropic-messages"> = buildModel(COPILOT_MODEL_SPEC);
it("disables eager tool-input streaming on the github-copilot host", () => {
expect(model.provider).toBe("github-copilot");
expect(model.compat.supportsEagerToolInputStreaming).toBe(false);
});
it("omits the per-tool eager_input_streaming flag on the wire payload", async () => {
const { promise, resolve } = Promise.withResolvers<unknown>();
streamAnthropic(model, CONTEXT, {
apiKey: COPILOT_BEARER,
signal: aborted(),
onPayload: payload => resolve(payload),
});
const payload = (await promise) as { tools?: Array<Record<string, unknown>> };
expect(payload.tools).toHaveLength(1);
expect(payload.tools?.[0]).not.toHaveProperty("eager_input_streaming");
});
it("omits the fine-grained-tool-streaming beta header on the github-copilot transport", () => {
const options = buildAnthropicClientOptions({
model,
apiKey: COPILOT_BEARER,
extraBetas: [],
stream: true,
interleavedThinking: false,
hasTools: true,
});
// Either the header is absent or, if other betas pile in later, it must
// not list `fine-grained-tool-streaming-2025-05-14`.
expect(options.defaultHeaders["anthropic-beta"] ?? "").not.toContain("fine-grained-tool-streaming-2025-05-14");
});
});
@@ -0,0 +1,20 @@
import { describe, expect, it } from "bun:test";
import modelsJson from "../src/models.json";
describe("minimax bundled catalog", () => {
it("pins MiniMax-M3 long-context entries to 1M context", () => {
const providers = [
{ id: "minimax", models: modelsJson.minimax },
{ id: "minimax-cn", models: modelsJson["minimax-cn"] },
];
for (const provider of providers) {
const model = provider.models["MiniMax-M3"];
expect(model).toBeDefined();
expect(model.provider).toBe(provider.id);
expect(model.contextWindow).toBe(1_000_000);
expect(model.maxTokens).toBe(128_000);
}
});
});
@@ -25,9 +25,9 @@ describe("zenmux provider support", () => {
test("registers built-in descriptor and default model", () => {
const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "zenmux");
expect(descriptor).toBeDefined();
expect(descriptor?.defaultModel).toBe("anthropic/claude-opus-4.6");
expect(descriptor?.defaultModel).toBe("anthropic/claude-opus-4.8");
expect(descriptor?.catalogDiscovery?.envVars).toContain("ZENMUX_API_KEY");
expect(DEFAULT_MODEL_PER_PROVIDER.zenmux).toBe("anthropic/claude-opus-4.6");
expect(DEFAULT_MODEL_PER_PROVIDER.zenmux).toBe("anthropic/claude-opus-4.8");
});
test("registers ZenMux in OAuth provider selector", () => {
+291 -7
View File
@@ -5,8 +5,144 @@
### Added
- Added isolated profile support via `--profile <name>` / `OMP_PROFILE` and shell alias bootstrap via `--alias <command>`, including launch/ACP bootstrap handling, extension-flag-safe parsing, profile-scoped user config discovery, and symlinked extension-directory discovery.
- Fixed paste and image placeholders crashing when the editor renders before theme initialization.
- Added `ModelRegistry.create(authStorage, modelsPath?)` async factory that runs the JSON → YAML migration step on `models.{yml,yaml}` asynchronously ahead of the sync constructor's bundled-model load. The sync `new ModelRegistry(...)` constructor still works (tests rely on it); production boot paths now use the factory so the migration's I/O lands off the event-loop hot path.
- Added `ConfigFile.tryLoadAsync()`, `ConfigFile.loadAsync()`, `ConfigFile.loadOrDefaultAsync()`, `ConfigFile.getMtimeMsAsync()`, and `ConfigFile.warmup(file)` so the rest of the codebase can migrate config reads off the sync path.
### Fixed
- Fixed `Test & smoke (TS)` CI timeouts caused by parallel test files racing on the process-global Settings singleton. `CustomEditor` now accepts a `magicKeywordsEnabledOverride` injection point so the shimmer-gate test can assert behaviour without calling `resetSettingsForTest()` / `Settings.init()`; the "streaming tool call preview height" describe drops its gratuitous Settings reset+init. Production wiring is unchanged ([#2582](https://github.com/can1357/oh-my-pi/issues/2582))
- Fixed MCP OAuth fallback rendering to show a short terminal hyperlink and keep the raw authorization URL on one unwrapped copy line ([#2121](https://github.com/can1357/oh-my-pi/issues/2121)).
- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts
- Fixed selector-style UI components to honor `tui.select.up` and `tui.select.down` keybindings instead of hard-coding raw Up/Down arrow bytes ([#1535](https://github.com/can1357/oh-my-pi/issues/1535)).
- Fixed a collapsed, still-streaming tool preview (an `eval`/`bash`/`ssh` box with output streaming in) reading as "weirdly truncated" — top border and head rows missing — once its box outgrew the viewport, snapping back to whole only while expanded with `ctrl+o` and breaking again when collapsed. A streaming preview was classified commit-unstable whenever collapsed, so the transcript offered none of its rows to native scrollback; once the box outgrew the window its head fell into the gap between the commit boundary and the window top, committed nowhere and repainted nowhere. The `provisionalPendingPreview` flag now applies only to the pending call preview (before any result) — once a streaming result exists the result renderer is the live, top-anchored shape and the block is commit-stable in both collapsed and expanded states, so its durable head always reaches scrollback.
- Fixed a crash in subagent task execution and extensions when a string (instead of a string array) was returned or set for the system prompt. Gracefully wrap string values in arrays.
## [15.13.0] - 2026-06-14
### Breaking Changes
- Replaced the `omp setup stt` command with `omp setup speech`. The old `stt` setup component is gone (no alias); `omp setup speech` now provisions the full speech stack — audio recorder, speech-to-text model, and text-to-speech model.
- Renamed the `tts.enabled` setting to `speechgen.enabled` (same boolean, default off; no alias). It still gates the on-demand `tts` speech-generation tool, now labelled "Speech Generation" in the settings panel.
### Added
- Added `paste.largeMenuThreshold` setting (0/100/250/500/1000, default 100) to control when large pasted content triggers the large-paste menu or stays as a normal `[Paste]` marker
- Added a large-paste editor menu for pasted text over the threshold that lets users choose to wrap the paste in a fenced code block, wrap it in `<pasted_text>` XML tags, or save it as `local://attachment-N` for on-demand reading
- Added `snapcompact-savings.jsonl` journaling for snapcompact tool-result compaction, recording session, provider, model, tool call, and estimated token savings whenever tool output is rendered as image frames
- Added `subagent:<id>` loop-phase breadcrumbs around in-process subagent event dispatch and finalization so the TUI event-loop watchdog can attribute a main-thread stall to subagent execution ([#2485](https://github.com/can1357/oh-my-pi/issues/2485))
- `highlightMagicKeywords(text, resetTo?, phase?)` now accepts an optional `phase` ∈ [0, 1) that rotates the gradient cyclically; sent bubbles omit it (static palette unchanged). `hasMagicKeyword(text)` exported from `modes/magic-keywords` is the cheap shimmer-gate the editor uses on every render.
- Added a `fastModeScope` setting (`both` | `openai` | `claude`, default `both`) controlling which providers `/fast on` (and the fast-mode toggle) target. `both` keeps the prior unscoped priority behavior; `openai`/`claude` scope fast mode to one family. `/fast status` now reports the active scope.
- Added the `mnemopi.embeddingVariant` setting (`en` | `multilingual`) selecting a stronger SOTA local embedding model — `en` → `BAAI/bge-base-en-v1.5` (768d), `multilingual` → `intfloat/multilingual-e5-large` (1024d). Resolution precedence is `mnemopi.embeddingModel` setting > `MNEMOPI_EMBEDDING_MODEL` env > variant default, so the documented env override is still honored. Changing the active model wipes and rebuilds stored embeddings on the next writable start ([#2476](https://github.com/can1357/oh-my-pi/issues/2476))
- Added a `/guided-goal` slash command that interviews you to refine an objective before enabling goal mode, then seeds goal mode with the agreed objective. The bounded interview (up to six turns) runs on the plan or slow model and falls back with a hint when the goal is still too vague ([#2502](https://github.com/can1357/oh-my-pi/issues/2502)).
- Added a large-paste menu: when a paste reaches `paste.largeMenuThreshold` lines (default 100; `0` disables), the editor offers to wrap it in a code block, wrap it in `<pasted_text>` XML tags (both collapse to a `[Paste]` marker that expands on submit), or save it to the session's `local://` store and insert a clean `local://attachment-N` reference the agent can `read` on demand. Esc keeps the previous inline-paste behavior, so the content is never lost.
- Added `8on22-bw` (leading) and `11on16-bw` (tracking) options to the `snapcompact.shape` setting, the spacing-tuned cells that are now the per-provider defaults (Anthropic → tracking, OpenAI/Google → leading)
- Added a local on-device neural TTS backend for the `tts` tool and a `providers.tts` switch (`auto` | `local` | `xai`, default `auto`). `local` synthesizes speech with Kokoro-82M — SoTA on-device TTS quality — via `kokoro-js` on the shared ONNX runtime (`@huggingface/transformers` + `onnxruntime-node`) in a subprocess worker (mirroring the tiny-model worker), keeping the model warm across calls and emitting 24 kHz WAV/PCM16 with no network call; `xai` keeps the existing Grok Voice cloud path; `auto` prefers local but routes `.mp3` requests to xAI when credentials exist (no local MP3 encoder is bundled, so a local `.mp3` request is written as a sibling `.wav`). `kokoro-js` is never a hard dependency: it is lazily `bun install`ed into a version-keyed runtime dir on first use (with `onnxruntime-node` force-pinned to a Bun-safe version), so its transformers@3.x graph never pollutes the main tree. New `tts.localModel` (default `kokoro`) and `tts.localVoice` (default `af_heart`; American/British, female/male voices) settings select the on-device voice.
- Added a unified, interactive `omp setup speech` command that walks one reusable flow across all three speech dependencies: it lets you pick and persist the speech-to-text (`stt.modelName`) and text-to-speech (`tts.localModel`) models from a TUI list, then downloads both models plus an audio recorder with live progress. The recorder is now auto-provisioned cross-platform (a static `ffmpeg` binary is fetched via the shared tools-manager when no SoX/FFmpeg/arecord is present, with the PowerShell fallback on Windows) instead of dead-ending with "install sox manually". `--check` and `--json` report recorder + STT-model + TTS-model readiness without installing.
- Added `omp say <text>`, which synthesizes text with the local on-device TTS engine and plays it through the speakers (cross-platform: `afplay` on macOS, `paplay`/`aplay`/bundled `ffmpeg` on Linux, PowerShell `Media.SoundPlayer` on Windows). `--out <file>` writes a WAV instead of playing, `--voice`/`--model` override the `tts.localVoice`/`tts.localModel` settings, and an uninstalled model prints an actionable `omp setup speech` hint.
- Added streaming speech vocalization: with `speech.enabled` on, the assistant speaks its reply through the speakers as it streams. Assistant text deltas are fed *directly into the engine's incremental text input* (Kokoro's `TextSplitterStream` via the worker) as they arrive, rather than pre-chunked in JS and synthesized one batch call per sentence — the engine owns sentence segmentation and emits one audio chunk per sentence. A single persistent player (`StreamingAudioPlayer`) drains those chunks **gaplessly** (raw 32-bit-float PCM piped to one `ffmpeg`→PulseAudio/ALSA process on Linux; interruptible per-file `afplay`/PowerShell `SoundPlayer` on macOS/Windows), replacing the spawn-a-player-per-sentence path that added latency and audible gaps. Overspeech is handled end to end: a new turn, a sent message, or an Esc/Ctrl+C interrupt stops playback **instantly** (the player process is killed rather than letting the current sentence finish); holding the push-to-talk key **ducks** the volume while you speak and restores it when you stop; and sequential utterances queue and drain in order instead of overlapping. `speech.mode` (`all` | `assistant` | `yield`, default `assistant`) picks what is spoken — `all` adds thinking, `yield` speaks only the final message at turn end — and `speech.voice` selects the Kokoro voice. `ask`-tool questions are spoken in every mode. Synthesis reuses the local Kokoro engine (`tts.localModel`) through a new streaming synthesis path (`TtsClient.synthesizeStream`) that pushes text in and streams audio chunks back over the worker protocol.
- Added live (streaming) speech-to-text: with `stt.enabled` on, transcription now appears in the composer *as you speak* instead of all at once after you stop. The recorder streams raw 16 kHz mono PCM from sox/ffmpeg/arecord stdout to the warm STT worker, where an energy-based endpointer (no extra model) splits speech into segments at natural pauses; each finalized segment is committed into the editor while the in-progress segment shows a live volatile preview that refreshes in place and is kept out of the undo history. Works with both the default Parakeet (sherpa-onnx) and the Whisper (transformers.js) tiers. Recorders that cannot stream to a pipe (the Windows PowerShell mci fallback) transparently fall back to single-shot transcription.
- Added `skills.enableAgentsUser` and `skills.enableAgentsProject` settings (default on) so the canonical OMP-native `~/.agent[s]/skills` and project-walkup `.agent[s]/skills` are configurable independently from the third-party Claude/Codex/Pi toggles.
- Added a read-only `ctx.models` facade for extensions: `list()` (authenticated models), `current()` (live session model), `resolve(spec)` (a model string or role alias → `Model`, using the same settings-backed aliases and match preferences as core selection), and `family(model)` (opaque canonical-identity lineage token for cross-family comparisons). Lets extension tools select models the way core does without reaching into the mutable registry ([#2406](https://github.com/can1357/oh-my-pi/issues/2406))
- Added RPC prompt lifecycle hints so hosts can distinguish scheduled agent turns from local-only slash commands via `data.agentInvoked` and `prompt_result`.
- Added extension lifecycle events for tool approval prompts: `tool_approval_requested` before the approval wait and `tool_approval_resolved` after approve, deny, or approval prompt failure.
### Changed
- Changed `handoff` custom messages (`customType: "handoff"`) to render in the transcript as a compaction-style expandable divider in both the main session and Agent Hub views, and expanded handoff details now show the handoff context body without `<handoff-context>` tags
- Changed the double-tap-← gesture (empty editor, main session) to stay inert when there are no subagents to show, instead of opening an empty Agent Hub roster. The explicit Agent Hub / observe keybindings still open the empty roster. The gating reuses the hub's own row count (after its persisted-subagent scan), so it matches exactly what the hub would display.
- Changed the `job` tool's `async.pollWaitDuration` setting (relabeled **Max Poll Time**) to add a `smart` value, now the default. A fixed value (`5s`–`5m`) still blocks for exactly that long; `smart` adapts: a blocking poll starts at a 5s floor and climbs a ladder (5s → 10s → 30s → 1m → 5m) with each back-to-back poll, so a tight poll loop backs off and stops spending turns on "still running" frames, then resets to the 5s floor after ~1 minute without polling (i.e. when the agent steps away to do real work). Escalation is tracked per agent (owner-scoped on `AsyncJobManager`).
- Added the `compat.supportsForcedToolChoice` custom-model flag for OpenAI-compatible models whose endpoints accept tools but reject forced `tool_choice` values ([#2546](https://github.com/can1357/oh-my-pi/issues/2546)).
- Changed speech-to-text to run fully local on-device with a tiered, multi-engine model picker. Transcription runs in a subprocess worker (mirroring the tiny-model worker; the native ONNX addons are hard-killed on shutdown to dodge the Bun NAPI-finalizer segfault) instead of shelling out to Python `openai-whisper`, keeps the model warm across recordings, and decodes WAV to 16 kHz mono float32 in-process. `stt.modelName` now selects on-device tiers across two engines: `parakeet` (default) — NVIDIA Parakeet TDT 0.6B v3 (25 languages) via the native `sherpa-onnx-node`, the Open ASR Leaderboard accuracy + throughput leader (lower WER than, and ~20× faster decoding than, Whisper large-v3) — plus `fast`/`balanced`/`turbo` mapping to Whisper base/small/large-v3-turbo (multilingual, up to 99 languages) via `@huggingface/transformers`. `omp setup speech` no longer mentions pip/python-whisper and reports recorder + model-cache readiness.
- Changed the speech-to-text trigger from the `Alt+H` keybinding to a hold-`Space` push-to-talk gesture. Holding the space bar emits an OS auto-repeat burst; once more than 5 spaces land in the editor it recognizes the hold, deletes (tracks back) those inserted spaces, and starts recording, then stops and transcribes when the repeats stop (the space bar is released). `app.stt.toggle` is now unbound by default but can be rebound to a chord for press-to-toggle; the gesture is gated on `stt.enabled`, and `Shift+Space` still inserts a literal space.
- `task.eager` ("Prefer Task Delegation") and `todo.eager` ("Create Todos Automatically") are now three-level enums (`default` / `preferred` / `always`) instead of booleans. For `todo.eager`, `preferred` renders a soft first-message reminder while `always` forces the `todo` tool (the previous "on" behavior); for `task.eager`, `preferred` adds a soft (SHOULD) delegation nudge to the system prompt while `always` uses hard (MUST/ONLY) wording plus a first-turn delegation reminder. Existing boolean configs migrate automatically (`true → always`, `false → default`). On models that cannot be forced to call `todo`, `todo.eager: "always"` now emits the first-turn reminder without forcing the call (previously such models received nothing) ([#2539](https://github.com/can1357/oh-my-pi/issues/2539), [#2540](https://github.com/can1357/oh-my-pi/pull/2540) by [@metaphorics](https://github.com/metaphorics)).
- Fixed `/model`-switching to a non-default OpenRouter model returning `404 No route: POST /chat/completions` when the provider is routed through the auth-gateway broker. The background catalog refresh re-ran `mergeDiscoveredModel` on every openrouter entry; for models whose bundled record already existed, the merge re-applied `baseUrl`, `headers`, and `compat` but dropped `transport: pi-native` because the raw `/v1/models` payload carries no transport hint. The next `/model` switch then picked the now-transport-less entry and routed through the default openai-completions client to `${baseUrl}/chat/completions` — a path the auth-gateway never serves. `mergeDiscoveredModel` now propagates the override/existing transport on the rediscovery branch ([#2555](https://github.com/can1357/oh-my-pi/issues/2555)).
### Fixed
- Fixed npm plugin installs to reject packages whose declared extension entry points cannot load because imports or nested dependencies are unresolved ([#2312](https://github.com/can1357/oh-my-pi/issues/2312)).
- Fixed the deferred MCP discovery banner (`Connecting to MCP servers: …`) overdrawing the chat input bar. `onMCPConnecting` wrote the banner straight to `process.stderr` while the TUI owned the terminal; it now emits on an `mcp:connecting` event channel that `InteractiveMode` renders through `showStatus` (the status container), mirroring the existing LSP-startup pattern so the banner can never paint over the input box border ([#2483](https://github.com/can1357/oh-my-pi/issues/2483))
- Fixed Win+Shift+S screenshot paste on Windows dead-ending on `Image not found`: Windows Terminal forwards a bracketed paste of the Snipping Tool's transient `…\MicrosoftWindows.Client.Core_*\TempState\…` file path, which is already gone (or never materialized) by the time omp reads it, so `handleImagePathPaste` failed with ENOENT even though the screenshot bitmap was still on the clipboard. The handler now falls back to the clipboard image across every local read failure (missing file, undecodable file, or generic error) before degrading, and the fallback is skipped over SSH where the clipboard lives on the remote host rather than the terminal holding the screenshot.
- Fixed npm prebuilt extension compatibility shims deriving their own package root through bare `@oh-my-pi/pi-coding-agent` resolution, which could select an older Bun cache copy in global installs and reintroduce mixed-runtime plugin loading stack overflows.
- Honor the `context_length` reported by OpenAI-compatible `/v1/models` discovery (`discovery: { type: "proxy" }` and `discovery: { type: "openai-models-list" }`) when present, so aggregator-reported windows override the stale bundled reference; the value is validated through the positive-number guard so a `0`/negative/stringly-typed upstream value cleanly falls back to the bundled reference (then `128000`) instead of pinning a broken window ([#2466](https://github.com/can1357/oh-my-pi/pull/2466) by [@androw](https://github.com/androw)).
- Fixed `web_search` SearXNG fallback when HTTP 200 responses contain no usable results plus `unresponsive_engines`; SearXNG now raises a transient provider error, and the fallback loop rejects any provider response with no renderable content before formatting an invisible success ([#2571](https://github.com/can1357/oh-my-pi/issues/2571)).
- Fixed `read` on a GitHub commit URL (`github.com/<owner>/<repo>/commit/<sha>`) returning the raw commit HTML page instead of structured content. `parseGitHubUrl` had no `commit` case, so commit URLs fell through to generic HTML rendering; they now resolve via the commits API and render as markdown (subject, author, stats, parents, full commit message, and a per-file unified diff), matching the existing blob/tree/issue/PR handling.
- Fixed release runs being silently cancelled by a later `main` push, which left tagged versions (`v15.12.6` in the wild) without a GitHub Release or npm publish. The CI workflow's `concurrency` group was `${{ github.workflow }}-${{ github.ref }}`, and since the release-script commit + `v*` tag are pushed atomically to `refs/heads/main`, the release run shared the `CI-refs/heads/main` group with every subsequent push; `cancel-in-progress: true` then killed it before `release_binary` / `release_github` / `release_npm` could run, and no future run carried the release tag at HEAD. The group now resolves to a per-sha `release-<sha>` slot with `cancel-in-progress: false` whenever the push subject matches `chore: bump version to ` (the release-script convention) or `github.ref` is a `v*` tag (`workflow_dispatch` recovery), so release runs are isolated from PR/main churn ([#2564](https://github.com/can1357/oh-my-pi/issues/2564)).
- Allowed `compat.streamIdleTimeoutMs: 0` in `models.yml`. The schema was `.positive()`, so the documented "set to 0 to disable" escape hatch was only reachable via the global env var ([#2422](https://github.com/can1357/oh-my-pi/issues/2422))
- Fixed LaTeX math delimiters (`$`/`$$`) and commands (such as `\text`, `\times`) rendering raw in the terminal by instructing the model to write equations as plain text / Unicode in its replies. The instruction is scoped to conversational output, so it does not constrain LaTeX or Markdown/KaTeX content the agent is asked to write to files ([#2550](https://github.com/can1357/oh-my-pi/pull/2550) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)).
- Fixed the MCP stdio transport `close()` potentially hanging while awaiting a read loop that can block indefinitely; teardown now detaches the read loop instead of awaiting it ([#2550](https://github.com/can1357/oh-my-pi/pull/2550) by [@usr-bin-roygbiv](https://github.com/usr-bin-roygbiv)).
- Fixed per-project memory isolation pulling other projects' memories into recall. Legacy (pre-#2412) mnemopi banks were rescued into recall when *any* working-memory row tagged the active cwd, but recall reads a bank wholesale and cannot filter rows by cwd, so a mixed-cwd bank leaked unrelated projects' memories. A legacy bank is now rescued only when *every* row tags the active cwd ([#2412](https://github.com/can1357/oh-my-pi/issues/2412)).
- Fixed a prompt template whose name collides with a builtin slash-command *alias* (e.g. a `models` template beside the builtin `/model`, which owns the `models` alias) appearing as a duplicate entry in the slash-command autocomplete picker. The picker's reserved-name filter now also excludes command aliases, matching runtime resolution (slash commands expand before prompt templates, so the template was already unreachable).
- Fixed the tool-result renderer re-shaping on every `invalidate()` (spinner tick, stream chunk, resize, keystroke), which made large grep/find/read results block the main thread for seconds and made typing sluggish. `ToolExecutionComponent.#updateDisplay()` now memoizes on a dirty key (result version, expand state, partial flag, spinner frame, image visibility, theme epoch, background-task freeze state, the resolved terminal image protocol, and a display-input version that covers streamed call args, the async edit-diff preview, and Kitty image conversions) and a `#displayBuilt` guard that also fast-paths the `#contentText` fallback, so the O(result-size) shaping runs once per change instead of every frame without freezing streamed args, previews, converted images, a backgrounded task settling to its static form, or images that arrive before the async image-protocol probe resolves. Image-bearing results also re-shape on terminal resize (keyed on the resolved image dimensions only when images are present) so inline images rescale, while image-free results never re-shape on resize ([#2484](https://github.com/can1357/oh-my-pi/issues/2484))
- Fixed `setTheme()` not bumping the theme epoch on its invalid-theme fallback path: a failed theme load swaps the active theme to the dark fallback, so memoized renderers (the tool-result renderer above) must re-shape — previously they kept the failed theme's stale colors until some other state changed ([#2484](https://github.com/can1357/oh-my-pi/issues/2484))
- Fixed tool-call spinners animating out of phase across parallel tool calls — each live tool block advanced its glyph from its own per-instance start time, so concurrent spinners showed different frames. Glyphs now derive from a single shared monotonic clock (`sharedSpinnerFrame`), keeping every live block in lockstep.
- Fixed the per-turn token-usage row (`display.showTokenUsage`) churning and duplicating in scrollback, most visibly with parallel tool calls. The row was rendered inside the assistant block above the turn's tool blocks, so finalizing the block was deferred and late appends recommitted the already-committed tool rows. The assistant block now always finalizes as soon as a tool-call appears, and the usage row is emitted as a standalone finalized block below the turn's tool blocks across all three render paths (live, transcript rebuild, agent-hub).
- Fixed the editor input box claiming a disproportionate share of small terminals (<=18 rows): the editor max-height floor (6 rows) ignored the available space. The height now yields to terminal size (`EDITOR_MIN_CHROME_ROWS`), reserving rows for the transcript and status line whenever the terminal can host both; on terminals too small for both, the editor collapses to its real bordered minimum instead of overshooting a fictitious cap.
- Fixed `ultrathink` / `orchestrate` / `workflowz` keywords not glowing while typing — the editor appended a CURSOR_MARKER (ESC-prefixed) after each render, so the magic-keyword regex's right-boundary `(?!\S)` tripped on ESC and dropped the gradient until a trailing character was typed. The marker-aware editor decorate (`packages/tui`) plus a phase-aware `highlightMagicKeywords` overload restore the live glow and add a Claude-Code-style shimmer while the prompt is focused, gated on `magicKeywords.enabled` ([#2475](https://github.com/can1357/oh-my-pi/issues/2475)).
- Fixed the compaction flow (`/compact` and plan-mode "Approve and compact context") leaving UI artifacts. `executeCompaction` added a `Spacer(1)` to the transcript that the sibling handoff path never adds and that leaked as an orphan blank line whenever compaction was cancelled or failed; that spacer is removed. On success the compaction loader is now stopped and the status container cleared *before* the transcript is rebuilt, so the live loader row no longer flickers over the reconciled transcript near the status seam (the idempotent `finally` still covers the cancel/fail paths) ([#2486](https://github.com/can1357/oh-my-pi/issues/2486))
- Fixed plan approval's "Approve and compact context" running the compaction summarizer on the pre-plan model instead of the plan model, cold-missing the plan model's prompt cache. Compaction now runs on the plan model (warm cache); the switch to the execution/pre-plan model happens only after a successful compaction and before any input queued during compaction is dispatched, so the queued turn runs on the post-compaction model. A cancelled compaction now also restores the pre-plan model (it previously stranded the session on the plan model), while a failed compaction stays on the plan model with its context intact.
- Fixed `Alt+Up` (dequeue) reporting "No queued messages to restore" for messages — including skills — typed while the session was compacting. `restoreQueuedMessagesToEditor` now drains `compactionQueuedMessages` alongside the agent queue, so the `Alt+Up to edit` hint restores every pending message it advertises.
- Fixed `restoreQueuedMessagesToEditor` (Alt+Up dequeue and Esc-abort) producing colliding `[Image #N]` markers when the editor draft already held pending image(s): queued text was prepended but queued images were appended, so positional marker → image lookup at submit time resolved to the wrong image. Each queued message's image markers are now renumbered by the running pending-image count before merge so the combined text stays aligned with the merged `pendingImages` order ([#2531](https://github.com/can1357/oh-my-pi/issues/2531)).
- Fixed `AgentBusyError` ("Agent is already processing. Use steer() or followUp()...") surfacing on mode transitions — as `Failed to finalize approved plan: ...` when a plan was approved while the agent was still streaming the post-`resolve` continuation (or a turn started by the approve-time compaction/clear), and as an error toast when a loop auto-submit or goal continuation fired during a streaming/compaction race. Plan approval now aborts any in-flight turn before dispatching the executor's first prompt, and `submitInteractiveInput` routes streaming-time loop, goal-continuation, and manual submissions through the follow-up queue (`streamingBehavior: "followUp"`) instead of throwing (synthetic continue-shortcuts stay developer-attributed and keep their prior behavior). Extends the manual-`/goal` fix in [#2454](https://github.com/can1357/oh-my-pi/issues/2454) to the continuation and plan-approval paths.
- Fixed `/plan` cycling between `plan` and `plan_paused` with no path back to mode `none`. `handlePlanModeCommand` had branches for entering and pausing but fell through to `#enterPlanMode()` when invoked from the paused state, so once a session entered plan mode the only operator-visible toggle re-entered it. The handler now matches `planModePaused` and fully exits — clearing `planModeHasEntered` and appending a `mode_change` to `"none"` — so `/goal` (and any other mode gated on `planModeEnabled || planModePaused`) can run again after a third `/plan` ([#2510](https://github.com/can1357/oh-my-pi/issues/2510)).
- Fixed HTML session export rendering empty text tokens (`text`, `userMessageText`, `customMessageText`, `toolTitle`) as the dark-theme grey `#e5e5e7` on every theme not literally named `light`, making transcripts illegible on custom light themes like `sandstone`, `limestone`, and `porcelain`. `getResolvedThemeColors` and the standalone `isLightTheme` helper now classify against the resolved `statusLineBg` luminance (the same surface `Theme.isLight` uses), so the HTML `defaultText` falls back to `#000000` on light themes and the standalone helper stays in lockstep with `Theme.isLight` ([#2516](https://github.com/can1357/oh-my-pi/issues/2516)).
- Fixed the Agent Hub opening on its own from a stray mouse click. The double-tap-← gesture (empty editor) fired on any two `left` keys within 500ms, but terminals with "click to move cursor" / pointer features (iTerm2 option-click, WezTerm, kitty, tmux) synthesize a burst of arrow keys on click — delivered sub-millisecond apart in one stdin read — so a single click could pop the hub with no key ever pressed. The gesture now requires the second tap to land a human-plausible interval after the first (≥40ms, <500ms) and ignores any third-or-later rapid tap, so synthesized bursts are rejected while a deliberate double-tap still works. The same hardening applies to the focused-subagent ←← "return to main" gesture.
- Fixed the `ctrl+p` model-role cycle indicator (the `default / gpt / fable / …` chip track) stacking duplicate copies in the scrollback when other chat activity landed between two cycles. The track was emitted through `showStatus`, whose back-to-back coalescing only merges when the previous status is still the last transcript child; any interleaved append broke that identity check and appended a second track. It now renders into a dedicated anchored container above the editor (cleared and rebuilt in place each cycle, like the Todos HUD) and auto-clears after a short linger, so rapid presses or concurrent activity can never duplicate it.
- Fixed the `Working…` loader vanishing for the rest of a turn after an auto-compaction (context-overflow recovery) or auto-retry. Those overlays took over the shared status container with a bare `statusContainer.clear()`, which detached the working loader but left `loadingAnimation` set; the resumed turn's `agent_start` → `ensureLoadingAnimation()` is guarded by `if (!this.loadingAnimation)`, so it skipped re-attaching the loader and the spinner stayed gone while the agent kept streaming. The overlay handlers now fully tear the working loader down (stop + dereference) via `#stopWorkingLoader()`, so the next `agent_start` recreates and re-attaches it.
- Fixed JS eval helper optional arguments rejecting Python-style positional calls. `read(path, offset, limit)` now works alongside `read(path, { offset, limit })`, `null`/`undefined` skip optional positional slots, and non-local URI reads such as `artifact://...` delegate through the read tool so line slicing works on spilled artifacts.
- Fixed legacy extension compatibility remapping for `@*-pi-ai/utils/oauth` imports so background workers load the relocated `@oh-my-pi/pi-ai/oauth` exports instead of resolving missing `src/utils/oauth/*` files ([#2566](https://github.com/can1357/oh-my-pi/issues/2566)).
- Fixed eager todo initialization prompting GPT-5.5 to emit unsupported task metadata fields, which could leave fresh sessions stuck on the forced first `todo` call ([#2561](https://github.com/can1357/oh-my-pi/issues/2561)).
- Fixed Windows plan-mode task fan-out crashing the TUI when nested async task progress formed a cycle; task rendering now cuts recursive snapshots and long Windows `local://` roots are shortened under temp storage ([#2551](https://github.com/can1357/oh-my-pi/issues/2551)).
- Fixed a band of streaming assistant output being lost from native scrollback — committed nowhere, repainted nowhere — once a reply grew taller than the viewport. Markdown whose layout keeps changing above the streaming tail (most visibly a table whose columns re-align as rows arrive) never earns a byte-stable commit-safe end, so as its head scrolled above the window the rows fell into the gap between the commit boundary and the window top and vanished. `TranscriptContainer` now reports a `getNativeScrollbackSnapshotSafeEnd()` for commit-stable live blocks (their whole body is durable content), and the renderer commits those scrolled-off rows audit-exempt — a later layout change of an already-committed row freezes a slightly-stale row in scrollback (duplication never loss) instead of dropping it. Provisional blocks (collapsing tool/edit previews) are unaffected.
- Fixed unknown `--`-prefixed flags being silently consumed as prompt text, which let a stale or typoed flag start a real agent session (connecting to MCP servers, waiting on the model) instead of failing fast. `parseArgs` now tracks unrecognized flag-shaped tokens and `runRootCommand` calls `reportUnrecognizedFlags` immediately after the post-extension reparse, exiting `2` with `Error: unknown flag: --…` before any session, MCP, or initial-message work runs. Extension-registered flags still pass cleanly since the validation runs after the extension-aware reparse, and `--` is honored as a POSIX positional separator so flag-shaped prompts (`omp -p -- --explain-this`) survive the new guard ([#2459](https://github.com/can1357/oh-my-pi/issues/2459)).
- Fixed `/goal <objective>` and `/goal set <objective>` during streaming so goal context is steered immediately but objective submission waits for the active turn to finish instead of spamming `AgentBusyError`. The interactive goal-continuation timer is now streaming-aware too: if a turn starts inside the 800 ms idle window the timer was scheduled in, it drops the tick instead of submitting a stale `goal-continuation` that would resurface the same `AgentBusyError`; the next `agent_end` reschedules ([#2454](https://github.com/can1357/oh-my-pi/issues/2454)).
- Fixed `~/.agent[s]/skills` not appearing as `/skill:<name>` commands when every named source toggle (`skills.enableCodexUser`, `skills.enableClaudeUser`, `skills.enableClaudeProject`, `skills.enablePiUser`, `skills.enablePiProject`) was off: `loadSkills` gated the `agents` provider on `anyBuiltInSkillSourceEnabled`, so a user who turned off the Claude/Codex/Pi sources to clean noise also lost their own canonical OMP-native skills. The `agents` provider now reads the dedicated `enableAgentsUser`/`enableAgentsProject` toggles, and the unknown-third-party fall-through gate is restricted to the named third-party toggles so keeping the default agents toggles on no longer silently re-enables `opencode`/`github`/`claude-plugins`/`gemini` skill sources ([#2401](https://github.com/can1357/oh-my-pi/issues/2401)).
- Fixed Claude Code marketplace plugin skills installed under `skills/<name>/SKILL.md` to also appear as bare slash commands such as `/understand`, matching Claude-native plugin docs. The slash command name is taken from the skill directory basename so display-style frontmatter names like `name: Understand Anything` still resolve to `/understand` ([#2415](https://github.com/can1357/oh-my-pi/issues/2415)).
- Fixed ACP `/move` builtin test expectations to compare the resolved destination path so the test is portable on Windows and Unix ([#2381](https://github.com/can1357/oh-my-pi/pull/2381) by [@oldschoola](https://github.com/oldschoola)).
- Fixed vLLM discovery so `providers.vllm.baseUrl` drives the built-in endpoint, additional OpenAI-compatible vLLM provider IDs work through `openai-models-list`, and discovered `max_model_len` or fallback `context_length` values set context windows instead of falling back to 128k.
- Fixed extension discovery ignoring package directories symlinked into an `extensions/` directory.
### Removed
- Removed the Python `openai-whisper` dependency and `pip` install path from speech-to-text — the bundled `transcribe.py` and all Python/whisper probes in `omp setup speech` are gone; the recorder (SoX/FFmpeg/arecord) remains the only external tool.
- Changed the speech-to-text trigger from the `Alt+H` keybinding to a hold-`Space` push-to-talk gesture. Holding the space bar emits an OS auto-repeat burst; once more than 10 spaces land in the editor it recognizes the hold, deletes (tracks back) those inserted spaces, and starts recording, then stops and transcribes when the repeats stop (the space bar is released). `app.stt.toggle` is now unbound by default but can be rebound to a chord for press-to-toggle; the gesture is gated on `stt.enabled`, and `Shift+Space` still inserts a literal space.
- Added an experimental, opt-in **auto-learn** loop (default off, `autolearn.enabled`). When enabled, after the agent stops it is nudged to capture reusable lessons: durable facts go to long-term memory and repeatable procedures become **managed skills** — `SKILL.md` files written to an isolated `~/.omp/agent/managed-skills` directory that is discovered and surfaced like authored skills but never overwrites user-authored skills (authored names always win). Two tools back this: `manage_skill` (create/update/delete managed skills) and `learn` (record a lesson, optionally minting a managed skill in the same call). `learn` works with the `hindsight`, `mnemopi`, or file-based `local` memory backend; under `local`, lessons append to a `learned.md` in the project's memory root (kept separate from the consolidation artifacts so they survive a consolidation pass) and are injected into future sessions. The nudge is passive by default (a hidden reminder rides the next turn); `autolearn.autoContinue` instead auto-runs one capture turn at stop, and `autolearn.minToolCalls` (default 5) gates trivial turns. Plan/goal-mode turns and subagents are never nudged.
- Fixed `/plan` cycling between `plan` and `plan_paused` with no path back to mode `none`, while preserving prompted paused-mode requests. The no-arg third toggle now fully exits — clearing `planModeHasEntered` and appending a `mode_change` to `"none"` — and `/plan <prompt>` from `plan_paused` re-enters plan mode and submits the prompt as the first turn ([#2510](https://github.com/can1357/oh-my-pi/issues/2510)).
- Fixed HTML session export rendering empty text tokens (`text`, `userMessageText`, `customMessageText`, `toolTitle`) as the dark-theme grey `#e5e5e7` on every theme not literally named `light`, making transcripts illegible on custom light themes like `sandstone`, `limestone`, and `porcelain`. `isLightTheme` now classifies against the resolved `statusLineBg` luminance (the same surface `Theme.isLight` uses), while HTML `defaultText` contrasts the actual export surface (`export.cardBg` / `export.pageBg` / derived `userMessageBg`) so light-status themes with dark export cards keep light transcript text ([#2516](https://github.com/can1357/oh-my-pi/issues/2516)).
## [15.12.6] - 2026-06-14
### Breaking Changes
- Removed the `writeLine` and `writeLineSync` methods from the public `SessionStorageWriter` contract, requiring custom `SessionStorage` backends to switch to the `append` API
### Added
- Added package-level exports for `SessionContext`, session entry types, session listing/loader helpers, and migration APIs via `session/session-context`, `session/session-entries`, `session/session-listing`, `session/session-loader`, and `session/session-migrations`
- Added asynchronous session-write `append(...)`-based persistence API in session storage implementations so callers can stream writes without sync line-appending methods
### Changed
- Changed session persistence internals to expose `writeTextAtomic(...)` on session storage writers for atomic whole-file replacements
- Changed online session-title generation to support tool-choice-less title models. Providers/models that cannot be forced to call a tool (chat-completions hosts without `tool_choice` support such as DeepSeek V4, and Claude Fable/Mythos) are now prompted to wrap the title in `<title>...</title>` markers instead of the `set_title` tool call; extraction is lenient, accepting a plain sentence or a truncated/unclosed tag. A `TITLE_SYSTEM.md` override is reused in this mode with the marker instruction appended.
### Fixed
- Fixed session JSONL persistence so the first assistant turn materializes the file synchronously, leaves the append writer open, and writes later entries with a sync append writer even during writer-close races instead of waiting on a queued rewrite.
- Fixed submitted user messages emitting OSC 133 command-start markers without a matching command-finished marker, which made some terminals group later transcript output under the first prompt instead of appending it normally.
- Fixed queued forced tool choices being rejected and requeued or dropped when their named tool is no longer active for the upcoming turn, preventing eager todo and pending-action reminders from forcing unavailable tools. ([#1701](https://github.com/can1357/oh-my-pi/issues/1701))
### Removed
- Removed the `re-roots past a cwd-less legacy session in a shared explicit sessionDir` relocation test case and the `stores symlink-equivalent home cwd sessions under home-relative directories` file-operations test case.
## [15.12.5] - 2026-06-13
### Changed
- Terminal resize now repaints only the viewport while a drag is in flight and defers the full transcript replay until the drag settles. Outside a multiplexer, every SIGWINCH used to erase and replay the entire transcript at the new width — re-laying-out (and, for markdown, re-lexing) all of history on each event, work thrown away the instant the next event arrived and re-done dozens of times a second during a drag. The TUI now composes and paints only the visible tail mid-drag — a new `ViewportTailProvider` fast path that the transcript implements by rendering blocks bottom-up and skipping everything above the fold, touching no commit/scrollback state — then runs the single authoritative rewrap + native-scrollback rebuild ~120 ms after the last resize event.
@@ -204,6 +340,159 @@
- `/context` (TUI panel and ACP report) now shows estimated snapcompact wire savings when `snapcompact.systemPrompt` or `snapcompact.toolResults` is enabled — per-feature text → frames token deltas, the reason a swap does not apply (savings margin, image budget, or text-only model), and the estimated size of the next request. The estimate and the live provider-request transform share one planner (`planInlineSwaps`) so displayed numbers cannot drift from wire behavior.
- Added `/debug dump-request` and `/debug next-request` as aliases for `/debug dump-next-request` when arming a one-shot AI provider request dump
- Added `/debug dump-next-request <path>` to dump the next AI provider HTTP request JSON to a chosen file.
- Added mouse-driven interaction to `/settings`, including tab and setting row hover highlighting, wheel scrolling, and left-click activation for entries and submenus
- Added fullscreen `/settings` mouse-event handling so scrolling and clicks work in an alternate-screen overlay
- `ModelRegistry.resolver` now accepts a model directly — `resolver(model, sessionId)` — deriving `provider`, `baseUrl`, and `modelId` from it; all model-scoped call sites migrated from the verbose `resolver(model.provider, { sessionId, baseUrl, modelId })` form.
- Added experimental `snapcompact.systemPrompt` and `snapcompact.toolResults` settings (off by default, `/settings` → Context → Experimental) that render the system prompt and large historical tool results as dense snapcompact PNG frames on vision-capable models to cut token cost. Frames are built per-request in the provider-context transform, cached across turns, capped by a per-provider image budget, and gated on a token-savings estimate — they never reach `session.jsonl`.
- Added a Personality selector to `/settings` (Model → Prompt): `default` (the previous built-in reply style), `friendly`, `pragmatic`, or `none`. The selected spec renders into a dedicated `<personality>` system-prompt block (extracted from the former `<reply-guidelines>` section) and applies to the live session immediately; subagents always omit the block.
- Added `mnemopi.polyphonicRecall` and `mnemopi.enhancedRecall` config.yml settings (off by default, `/settings` → Memory → Mnemopi) that enable the mnemopi 4-voice polyphonic recall engine and the tiered query result cache without environment variables; `MNEMOPI_POLYPHONIC_RECALL` / `MNEMOPI_ENHANCED_RECALL` still override the configured values when set ([#2323](https://github.com/can1357/oh-my-pi/issues/2323)).
- Added the Expert Elixir language server (`expert`, invoked as `expert --stdio`) to the built-in LSP server list, auto-detected for Mix projects (`mix.exs`/`mix.lock`). When both are installed, `elixir-ls` remains the primary navigation server (Expert is ordered after it).
- Added `magicKeywords.enabled` and per-keyword `magicKeywords.ultrathink`, `magicKeywords.orchestrate`, and `magicKeywords.workflow` settings to disable hidden magic-keyword notices and ultrathink auto-thinking escalation ([#1796](https://github.com/can1357/oh-my-pi/issues/1796)).
- Added external-editor support for Plan Review section annotations, preserving multiline feedback for Refine plan ([#2305](https://github.com/can1357/oh-my-pi/issues/2305)).
- Added plain-RPC slash command discovery with command source metadata and startup/update notifications ([#2261](https://github.com/can1357/oh-my-pi/issues/2261)).
- Added the `statusLine.transparent` appearance setting (default off): when enabled, the status line skips the theme's `statusLineBg` fill and powerline end caps so the bar inherits the terminal's default background — useful in Ghostty and other terminals whose theme background does not match the theme's hardcoded status-line color ([#2306](https://github.com/can1357/oh-my-pi/issues/2306))
- Snapcompact compaction now passes the session model so frames render in the provider-optimal shape (unscii `8x8r-bw` for Anthropic-family/unknown APIs, `8x8r-sent` for Google, Lanczos-stretched `6x6u-sent` with `detail: "original"` for OpenAI), per the snapcompact 200k-token evals
- Added per-turn supersede pruning of stale `read` results: when a file is re-read, older copies of the same path/selector are pruned from context at cache-favorable moments (small suffix, idle gap, or alongside overflow pruning). Gated by the new `compaction.supersedeReads` setting (default on)
- Added soft request budgets for task subagents (explore/quick_task 40, others 90, configurable via `task.softRequestBudget`, 0 disables): crossing the budget injects a one-time wrap-up steer into the child; crossing 1.5× aborts the run gracefully
- Added cancelled/aborted subagent salvage: instead of `(no output)`, merged task results now carry the child's last activity snippet plus request/token stats, and per-child stats lines include request counts
- Added a repeat-read notice to the `read` tool: the third and later reads of the same file in a session append a one-line note suggesting range re-reads or the context echoed in edit results
- Added a hard inline byte cap (~50KB) at the bash and browser tool-result boundaries with head/tail elision and an `artifact://` footer for the full output, closing paths that previously let 100KB+ results land inline
- Added the Agent Hub overlay (`ctrl+s`, `alt+a`, or double-tap left arrow on an empty editor): a live table of registered subagents (status, unread IRC count, current task, last activity) with per-agent chat — Enter opens a transcript + input line that steers a running agent, prompts an idle one, and revives a parked one; `r` revives and `x` aborts/releases the selected agent
- Added the `snapcompact` compaction strategy (`compaction.strategy: "snapcompact"`): history is archived onto dense bitmap "snapcompact" frames a vision model reads back directly, instead of an LLM-generated summary — instant, free, and verbatim. Auto compaction (including overflow recovery) and manual `/compact` both honor it; falls back to context-full with a visible warning notice when the current model is text-only (e.g. Codex API surfaces) or when `/compact` is given custom instructions. Frames survive context rebuilds and later compactions (budget eviction is middle-out: the session-head frame is pinned); the expanded compaction message notes the attached frame count
- Added a persistent subagent lifecycle: finished subagents stay live as `idle`, are parked to disk after `task.agentIdleTtlMs` (default 7 minutes; `0` keeps them live until exit), and are revived automatically when messaged or prompted from the Agent Hub
- Added the `history://` protocol: `history://` lists every registered agent and `history://<agentId>` renders a concise markdown transcript (tool calls collapsed to one line each, thinking elided) for live and parked agents alike
- Added an IRC mailbox bus with bounded per-agent inboxes: `irc` `wait` blocks until a matching message arrives, `inbox` drains or peeks pending messages, and sending to an idle or parked agent wakes or revives it for a real turn
- Added a dedicated TUI renderer for the `irc` tool: directional send/receive headers with delivery-outcome coloring, quoted message bodies with expand-aware truncation, per-recipient receipt trees for broadcasts and failures, and status-badged peer listings with unread counts
- Added the `task.batch` setting (default on): the task tool's batch shape `{ agent, context, tasks[] }` spawns one subagent per item — each its own independent background job with the normal idle/parked lifecycle and optional per-item isolation — and prepends the required shared `context` to every spawned subagent's system prompt; disabling it restores the flat single-spawn schema
- Added RPC subagent subscription frames, snapshots, and transcript catch-up APIs for desktop clients embedding `omp --mode rpc`.
- Added opt-in `shellMinimizer.sourceOutlineLevel` and `shellMinimizer.legacyFilters` settings so shell minimization can tune source outlining and selectively fall back to conservative legacy routing.
- Added repeatable `--config <path>` CLI overlays for temporary `config.yml`-style settings without editing the persistent global config ([#1733](https://github.com/can1357/oh-my-pi/issues/1733)).
- Added `python.interpreter` to pin eval's Python backend to an explicit interpreter and skip automatic runtime discovery ([#1802](https://github.com/can1357/oh-my-pi/issues/1802)).
- Added `!command` resolution for `models.yml` provider `apiKey` values and provider/model headers ([#1888](https://github.com/can1357/oh-my-pi/issues/1888)).
- Documented the oMLX setup path through existing OpenAI-compatible local discovery ([#1957](https://github.com/can1357/oh-my-pi/issues/1957)).
- Added `TITLE_SYSTEM.md` discovery so users can override the automatic session-title generation prompt for online and local tiny title models without patching installed prompt files. The override is re-discovered when the session working directory changes via `/cwd`.
- Added a structured memory runtime surface for extensions and UI integrations to query backend status, search memories, and save explicit memories across the configured memory backend.
- Added support for Git repositories using the `reftable` storage format by detecting `extensions.refStorage = reftable` in the repository configuration and falling back to shelling out to Git commands (`git symbolic-ref`, `git rev-parse`) for reference and HEAD resolution.
- Added `/setup providers` (also available as `/setup` or `/providers`) to reopen the interactive provider setup scene from an active TUI session, letting users sign in and choose a web search provider without rerunning the full onboarding flow.
- Added `supportsReasoningParams`, `alwaysSendMaxTokens`, `strictResponsesPairing`, and a recursive `whenThinking` overlay (alongside `streamIdleTimeoutMs`/`supportsLongPromptCacheRetention`/`requiresToolResultId`/`replayUnsignedThinking`) to the OpenAI/Anthropic `compat` schema so custom model entries can configure those provider-specific capabilities
- Custom model `thinking` config now uses the catalog's explicit vocabulary: `efforts` (ordered list) plus optional `defaultLevel`, `effortMap`, and `supportsDisplay` overrides; the legacy `minLevel`/`maxLevel`/`levels` range shape is still accepted and normalized at parse time. Wire facts (`effortMap`/`supportsDisplay`) are backfilled from model identity when not set, so existing claude-proxy configs keep the 5-tier adaptive scale and summarized display without changes.
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("capacity: 5h → 2.40/5 accounts used (2.60× quota left)"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
- npm installs now execute a prebundled single-file entry: the published `bin.omp` points at `dist/cli.js` (built by `scripts/bundle-dist.ts` during `prepack`, ~18MB minified, natives/transformers/mupdf external), cutting npm-install cold start by roughly 3x versus transpiling the raw TypeScript graph per launch; `src/**` stays published for SDK consumers and worker fallbacks. The on-repo manifest keeps `bin.omp` at `src/cli.ts` — release rewrites it via the `publishBin` override in `scripts/ci-release-publish.ts` — so source installs (`bun link`, `install.sh --source`) keep working without a build step
- Plain interactive TTY launches print a dim two-line startup splash (`omp <version>` / `Initializing session…`) before session construction so first pixels appear immediately; suppressed for resume/fork/continue flows, quiet mode, `PI_TIMING`, and non-TTY stdio
- Added `/stats` to launch the local stats dashboard from an active session, syncing session files first and opening the same browser dashboard as `omp stats`.
- `/settings` now supports type-to-search filtering on setting labels, paths, descriptions, and values; Escape clears an active search before closing the panel.
- Added a read-only `view` op to the `todo` tool that echoes the current list without mutating state, so the agent can recover exact task text instead of guessing it from memory.
- Added an optional `fetch` option to `CustomToolContext` so custom tools can use a caller-provided HTTP implementation
- Added optional `fetch` overrides to `ModelRegistry` construction and MCP/web search/tool network calls, enabling callers to inject custom HTTP clients instead of relying on global `fetch`
- Added a `bash.enabled` setting to disable the model-facing bash tool while leaving user-initiated bang/RPC bash commands available.
- Added an `@<upstream>` model-selector suffix to pin an aggregator model to a single upstream provider per invocation, e.g. `--model openrouter/z-ai/glm-4.7@cerebras` (sets OpenRouter `provider.only`; Vercel AI Gateway models map to `vercelGatewayRouting.only`). Resolved through `parseModelPattern`, so it works for `--model`/`--smol`, model roles, and the SDK, and composes with a trailing thinking level (`...@cerebras:high`). The base must resolve to an aggregator (`openrouter.ai` / `ai-gateway.vercel.sh`); otherwise the `@` stays part of the id, so ids that legitimately contain `@` (`claude-opus-4-8@default`, `workers-ai/@cf/...`) are unaffected.
- Added a `/plan-review` command that manually (re-)opens the plan-review overlay while plan mode is active. Since there is no fixed plan filename, it reviews the newest `local://<slug>-plan.md` the agent wrote — useful for pulling the review back up after dismissing it, or reviewing a plan the agent wrote without calling `resolve`.
- Added Homebrew and mise package-manager update paths to the self-update command so installations launched from those tools are updated through their native workflows
- Added detection of Homebrew and mise install locations so self-update chooses the manager-specific updater when the active `omp` binary comes from a package-manager-managed path
- Added `astCondition` to TTSR rule frontmatter as a syntax-aware alternative to regex `condition`, enabling AST-based matching for edit/write tool snapshots
- Added a built-in `ts-redundant-clear-guard` rule that flags redundant guards around `clearTimeout`, `clearInterval`, and `clearImmediate` calls
- Added a built-in `ts-no-test-timers` rule that flags real timers (`Bun.sleep`, `setTimeout`, `setInterval`) in `*.test.ts` files, steering toward fake timers (`vi.useFakeTimers()` / `vi.advanceTimersByTime()`)
- Added support for paste marker highlighting with accent styling (`[Paste #N, +X lines]`/`[Paste #N, Y chars]`) in the prompt editor, matching the visual treatment of image references
- Added pixel dimensions to pasted/loaded image placeholders in the prompt — the marker now reads `[Image #N, WxH]` (falling back to `[Image #N]` when the header can't be decoded).
- The bundled shell now treats `nohup` as a builtin: `nohup … &` runs the command without masking `SIGHUP` or detaching it, so agent-started daemons stay tied to this agent's lifetime instead of leaking as orphans when the agent exits. Updated the bash tool prompt's daemon guidance to match (dropped the `nohup … & / setsid … & / disown` detach recommendation in favor of a large `timeout` plus the persistent session).
- Added per-tool `tool.*` theme symbol keys (nerd/unicode/ascii presets) plus a quiet `status.done` glyph, so each tool's result header can carry a signature icon instead of a generic status mark
- macOS release binaries are now signed with a Developer ID Application identity (hardened runtime + secure timestamp + JIT/library-validation entitlements) and notarized in CI when the `APPLE_*` signing secrets are configured; releases auto-fall back to ad-hoc signing until then. This makes the shipped binaries Gatekeeper-acceptable, unblocking an official Homebrew submission ([#776](https://github.com/can1357/oh-my-pi/issues/776)). See `docs/macos-signing-notarization.md`.
- Added a Homebrew install path: `brew install can1357/tap/omp`. The [can1357/homebrew-tap](https://github.com/can1357/homebrew-tap) formula installs the prebuilt release binary, and a `release_brew` CI job regenerates it (version + per-asset sha256) from each published release via `scripts/ci-update-brew-formula.ts` ([#776](https://github.com/can1357/oh-my-pi/issues/776)).
- Added clickable file path hyperlinks to read tool outputs (read-call rows, grouped summaries, and inline previews) using resolved or absolute file targets with selector-based line anchors for quick navigation
- Added a resolved-span echo to `replace block`/`delete block` edits: a successful block op now prints `replace block N → resolved lines A-B (K lines)` between the section header and the diff preview, so the model can confirm tree-sitter matched the construct it intended (e.g. catch a decorator left outside the block) instead of inferring the span from the diff after the fact.
- Added `raw-sse.txt` to debug report bundles, exporting recent raw provider SSE diagnostics when captured
- Added `/model` visibility for auto-selected role defaults: inferred `pi/smol`/`pi/slow`/designer choices now show as compact `[ROLE auto]` badges, while explicitly configured roles keep the existing solid badges and thinking labels.
- Added credential provenance to the `/login` and `/logout` provider picker: each authenticated provider now shows where its credential comes from — `(login)`, `(api key)`, `(env: VAR_NAME)`, `(config)`, `(--api-key)`, or `(custom provider)` — so a real OAuth login is distinguishable from an env var that merely aliases the provider (e.g. `COPILOT_GITHUB_TOKEN`). The origin is also matched by the picker's type-to-search filter.
- Added `display.smoothStreaming` setting (default `true`) to let users enable or disable smooth assistant-stream text reveal
- Added `/tan <work>` slash command to fork the current conversation into a background agent so tangential work can continue asynchronously while your main session stays active
- Added a background `/tan` dispatch message that records the handoff in the transcript and marks the delegated work as non-blocking
- Added `providerPromptCacheKey` support to `CreateAgentSessionOptions` so `/tan` background sessions can reuse the parent session’s prompt-cache lineage
- Added session cloning for `/tan` runs with copied artifacts and shared MCP proxy tools
- Added `SessionManager.forkFrom`’s optional `suppressBreadcrumb` mode to avoid breadcrumb updates when forking background `/tan` sessions
- Added OSC 5522 enhanced paste handling in `InputController`, so terminal clipboard events are decoded as image or text payloads and inserted without passing raw paste sequences to the editor
- Added bracketed image-path paste support in `CustomEditor` so a single pasted image file path (PNG/JPEG/GIF/WEBP) is loaded from disk and inserted as an image candidate
- Added direct support for `Image #N` insertion from pasted local image paths by routing successful image-path pastes through the same image normalization and resize flow as clipboard image pastes
- Added `/fresh` to rotate the provider-facing session id and clear in-memory provider stream/cache state without changing the local session file.
- Added a `ChatBlock` transcript primitive (`modes/components/chat-block.ts`) and a single `ctx.present(...)` sink (with `ctx.resetTranscript()`) so chat output is mounted in one place instead of the repeated `chatContainer.addChild(...)` + `ui.requestRender()` pattern scattered across controllers. `ChatBlock` carries a React/Svelte-style lifecycle — `onMount` starts effects, `onCleanup` registers teardown, `finish()` self-completes (stops timers and freezes the block at its final content), and `dispose()`/`resetTranscript()` tears everything down — so animated blocks own their own resources instead of leaking `setInterval`/`requestRender` bookkeeping into callers. The MCP "Connecting…" spinner is now such a block.
- Added a `framedBlock` output-block helper (`tui/output-block.ts`) plus a `borderColor` override and `applyBg: false` (no background fill) on output blocks, a `renderStatusLine` `iconOverride`, and an `icon.search` (magnifier) theme symbol — so tool renderers can draw self-contained muted-outline frames and search-family tools can show a magnifier instead of a checkmark.
- Added a GitHub Actions read handler to the `read`/web-fetch GitHub scraper. Fetching `github.com/{owner}/{repo}/actions/runs/{id}` renders the run metadata plus a per-job breakdown (steps listed for any job that did not succeed), and `…/actions/runs/{id}/job/{id}` (also the API-style `…/jobs/{id}`) renders a single job's metadata, step table, and full plain-text logs. Logs are fetched via the `actions/jobs/{id}/logs` redirect using `GITHUB_TOKEN`/`GH_TOKEN` when present, with the per-line ISO timestamp prefix and leading BOM stripped; the section degrades to an explicit notice when logs are unavailable (no token, private repo, or expired/unfinalized run).
- Added anonymous fallback for Perplexity web search, allowing `web_search` and explicit Perplexity provider usage when no Perplexity credentials are configured
- Added `gallery` CLI command to render built-in tool renderer output across streaming, in-progress, success, and failure states
- Added `omp gallery` filtering and rendering options (`--tool`, `--state`, `--width`, `--expanded`, and `--plain`) for focused renderer previews and plain-text output
- Added `omp gallery` fidelity for tools whose renderers are attached on the tool instance (`lsp`, `task`): the gallery now drives them through the same custom-tool render branch production uses, so regressions in that path surface in the gallery rather than only in a live session.
- Added `omp gallery --screenshot`, which renders the gallery through a real virtual terminal (VHS) and writes PNG screenshot(s) instead of ANSI, so agents (and anything that can only read raw bytes) can actually see the rendered output. The capture forces truecolor and matches the active theme/symbol preset; tall galleries split across multiple images (whole renderers are never cut). Tune with `--out`, `--font`, and `--font-size`; requires `vhs` on `PATH` and fails with install guidance when absent.
- Added `app.display.reset`, bound to `Ctrl+L` by default, to force an immediate terminal display reset/redraw without resizing the window.
- Added `timeout-pause` and `timeout-resume` eval bridge status events emitted around `agent()`/`llm()` operations
- Added a `/copy` picker: `/copy` now opens a fullscreen, outlined tree of recent assistant messages with their code blocks nested beneath (like `/tree`). Navigate with ↑↓, and Enter copies the highlighted node — a whole message, an individual code block, "All N blocks", or a bash/eval command interleaved with the assistant turn that issued it. A live preview pane shows the selected target, wrapping prose and syntax-highlighting code/commands.
- Added a persistent error banner pinned above the editor when an assistant turn ends on a provider error (e.g. Anthropic's "Output blocked by content filtering policy"). The transcript `Error: …` line scrolls away as the conversation grows, so terminal turns that ended on a stream error could pass unnoticed; the banner stays in the fixed region above the input and is cleared when the next turn starts.
- Added bold, underlined, clickable `[Image #N]` placeholders in the draft editor and sent user-message bubbles, backed by extension-bearing blob-store sidecar files so terminal `file://` links open in image viewers.
- Added the active model identifier (`provider/id`) to the system prompt's `<workstation>` block so the agent knows which model it is running as. Gated by the new `includeModelInPrompt` setting (default on); the base prompt is rebuilt on a mid-session model switch so the surfaced identifier stays current.
- Added `OLLAMA_HOST` support for implicit local Ollama discovery when `OLLAMA_BASE_URL` is unset, so OMP picks up the same host setting used by Ollama.
- Added `OLLAMA_CONTEXT_LENGTH` as a positive-integer context-window override for implicit local Ollama discovery, so users can correct OMP context budgeting without writing per-model overrides.
- Added an encrypted local auth-broker snapshot cache for `discoverAuthStorage`, with `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` and `OMP_AUTH_BROKER_SNAPSHOT_CACHE`, so fresh cached broker credentials can boot without a blocking `/v1/snapshot` fetch and survive broker-down startup windows.
- Added `dry-balance` CLI command to perform a dry-run OAuth account balancing check across configurable random session IDs, with sample and concurrency options, JSON output, and success/failure summary reporting
- Added `--json` output mode and machine-readable result format to `omp dry-balance` for automated use
- Added `omitMaxOutputTokens` to `models.yml` model definitions and `modelOverrides`, so users can opt a model out of the on-the-wire `max_output_tokens` / `max_tokens` cap while keeping the catalog `maxTokens` for local budgeting. Intended for Ollama-style proxies whose upstream output limit OMP cannot discover. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881))
- Added deferred session-title generation so greetings no longer become the session title. A first user message that is only a greeting / acknowledgement / filler ("hi", "thanks", "ok", a bare number, emoji-only, etc.) is now detected deterministically and skips titling entirely — no title model is invoked. Title generation then retries on each subsequent user message while the session stays unnamed, so the title is deduced from the first message that actually describes work. A capable online title model may additionally answer `none` to decline a non-greeting taskless message (normalized to "no title").
- Added env-driven OpenTelemetry trace export. When `OTEL_EXPORTER_OTLP_ENDPOINT` (or `OTEL_EXPORTER_OTLP_TRACES_ENDPOINT`) is set, `omp` registers a global OTLP/proto trace exporter and switches on the agent loop's telemetry, so the `invoke_agent` / `chat` / `execute_tool` spans actually reach a collector instead of a no-op tracer. Honors the standard `OTEL_*` env contract (endpoint, headers, `OTEL_SERVICE_NAME`, `OTEL_SDK_DISABLED` and `OTEL_TRACES_EXPORTER=none` parsed case-insensitively) and the `OTEL_INSTRUMENTATION_GENAI_CAPTURE_MESSAGE_CONTENT` capture toggle; it is a no-op when no endpoint is configured. Only the `http/protobuf` transport is supported — a `grpc` or `http/json` `OTEL_EXPORTER_OTLP*_PROTOCOL` declines rather than misrouting spans. This makes the existing telemetry usable from headless hosts that run `omp` as a spawned child process, where an in-process `TracerProvider` registered by the parent can't reach the child. Uses the `@opentelemetry/exporter-trace-otlp-proto` 2.x line, which exports cleanly under Bun.
## Fixed
- Fixed the status line session name (and the editor border / status-line gap fill) being nearly illegible on light themes.
- Added `IndexedSessionStorage` and `SessionStorageBackend` exports to support shared metadata-indexed session backends
- Added the `tui.maxInlineImages` setting (default `8`) capping how many inline images render as live terminal graphics. Once a new image pushes the count past the cap, the oldest images are hidden via a full redraw — replaced by their `[Image: …]` text placeholder and purged from the terminal's graphics store — so long sessions with many screenshots/diagrams stop piling up images (and, on Kitty, stop leaving scrollback ghosts). Set to `0` to keep every image inline.
- Added a "View: terminal state" item to the `/debug` menu that prints the detected terminal, live geometry and cell size, multiplexer, and the negotiated subprotocols actually in use — graphics (Kitty/iTerm2/Sixel), desktop notifications (BEL/OSC 9/OSC 99, plus whether OSC 99 was confirmed via a device-attributes probe), OSC 8 hyperlinks, 24-bit color, DECCARA rectangular-SGR background fills, and DEC 2026 synchronized output — alongside the scrollback-clear strategy (`CSI 22 J` vs `CSI 2 J` redraw / ED3 eager-erase risk) and the raw `TERM`/`TERM_PROGRAM`/`COLORTERM` detection signals.
- Added a "Test: terminal protocols" item to the `/debug` menu that renders one live sample of every special escape protocol the renderer can emit — SGR text attributes (bold/italic/underline/strikethrough/inverse/dim), themed and 24-bit truecolor, OSC 8 hyperlinks, OSC 66 text sizing (large text), and an inline graphics swatch via the active image protocol (Kitty/iTerm2/Sixel, with a text fallback) — and fires a desktop notification, so you can eyeball which protocols the current terminal actually honors. The sample image is a gradient PNG generated in-process, so the graphics test needs no asset on disk.
- Added the `tui.textSizing` setting (default off) that renders Markdown H1 headings at 2x scale via Kitty's OSC 66 text-sizing protocol. It replaces the undocumented `PI_TUI_TEXT_SIZING` env var with a real setting, and only takes effect on Kitty terminals (where OSC 66 is implemented) — it is ignored everywhere else so headings never emit raw escape bytes.
- Added a lifecycle status to the `/resume` session picker. Each session's tail (last 32 KiB) is now read alongside the existing header window in a single pass, and its final message classified as `done` (the agent ended its turn and yielded control back), `interrupted` (a trailing tool call or tool result the loop never continued from), `aborted`, `error`, or `pending` (a trailing user message with no reply). The status renders as a colored segment on each session's metadata line. When the final message is larger than the tail window the status is omitted rather than guessed.
- Added support for `disable-model-invocation: true` frontmatter field from the [Agent Skills standard](https://agentskills.io/specification). Skills using this field are now hidden from the system prompt listing, matching the behavior of `hide: true`.
- Added a bundled TypeScript rule that warns against leaving `@deprecated` compatibility shims behind instead of finishing a refactor.
- Added an all-projects scope to the session picker (`pi --resume` / `/resume`). Press `Tab` to toggle between the current folder's sessions and every session across all projects; the all-projects list is loaded lazily and shows each session's directory. When the current folder has no sessions the picker now opens straight into all-projects scope instead of printing "No sessions found".
- Migrated the Kagi web search provider to Kagi's V1 Search API (`POST /api/v1/search`), replacing the sunset V0 endpoint while keeping the `kagi` provider id, `KAGI_API_KEY` credential, and `/login kagi` flow unchanged ([#1272](https://github.com/can1357/oh-my-pi/pull/1272) by [@thismat](https://github.com/thismat))
- Added Anthropic `anthropic-ratelimit-unified-*` response-header warming for `/usage` and the status-line usage segment, throttled to reduce direct OAuth `/usage` probes during active use.
- Added `ask` option descriptions so agents can keep short labels and render explanatory text as separate muted rows in the selector.
- Added an extension API for rendering supplemental UI below visible assistant thinking blocks.
- Added default-on `lsp.diagnosticsDeduplicate` support so post-edit LSP diagnostics already shown for a file are suppressed within the session and only new or changed diagnostics are surfaced.
- Added support for decimal and `k`/`m` suffix turn-budget directives, enabling budgets like `+1.5k` and `+2m` in eval message parsing
- Changed eval budget resolution to honor a user `+Nk` directive over an active Goal Mode limit while falling back to Goal Mode when no per-turn ceiling is set
- Added `agent()` eval options `agent_type`/`agentType`, `model`, `context`, and `label`, and returned structured JSON when `schema` is provided in JS and Python eval cells
- Added a live, Task-tool-style progress tree for eval `agent()` calls, drawn below the notebook (code cell) box. Each subagent surfaces as a status line (icon · id · tool count · context · cost, plus duration on completion) with its current tool/intent while running, and updates mid-execution rather than only at the cell's final result. Progress events coalesce per subagent id so the persisted event list stays bounded across many throttled ticks.
- Added `agent()` to the `eval` runtime so JS and Python cells can spawn one subagent through the existing task executor; JS eval also gained bounded `parallel()` and `pipeline()` helpers for orchestrating subagent calls.
- Added a `workflow` magic keyword (mirrors `orchestrate`/`ultrathink`): the standalone word glows amber→green in the editor and appends a hidden notice steering the model to author deterministic multi-subagent fan-outs in `eval` (agent/parallel/pipeline). Matching is whitespace-delimited and case-sensitive (lowercase only); the singular and plural both trigger, but capitalized forms, inflections like `workflowed`, and path-embedded occurrences like `workflow.ts` do not.
- Added `parallel()` and `pipeline()` to the Python `eval` runtime (thread-pool over the synchronous `agent()` bridge), mirroring the JS helpers: bounded pool (default 4, max 16), input-order preservation, a barrier between every `pipeline` stage, and contextvar propagation so `agent()` works inside worker threads.
- Added `log()`, `phase()`, and a `budget` object to both `eval` runtimes (Python and JS). `log`/`phase` emit progress/phase status lines; `budget.total`/`budget.spent()`/`budget.remaining()`/`budget.hard` expose a real per-turn output-token budget. A `+Nk` directive in the user's message sets an advisory budget (the model self-limits via `budget.remaining()`); `+Nk!` (or an active Goal Mode budget) makes it a hard ceiling that blocks further eval `agent()` spawns once reached. `budget.spent()` counts output tokens spent this turn across the main loop and all eval-spawned subagents.
- Added search support for virtual internal URLs (including `omp://` roots) by resolving and scanning in-memory internal resources as search targets alongside filesystem paths
- Added expansion of virtual internal URL search targets so `search` can match multiple internal documents when given `omp://`
- Added `/omfg <complaint>` slash command that drafts a TTSR rule from a complaint, validates it against the current conversation, saves it to project or `~/.omp/agent/rules`, and registers it live.
- Added `/shake` slash command and the `shake` / `shake-summary` compaction strategies that reduce context by mechanically dropping heavy content instead of LLM summarization. `/shake` (alias `/shake elide`) strips heavy tool-call results and large fenced/XML blocks, offloads the originals to one session artifact, and leaves a recoverable `artifact://<id>` placeholder; `/shake summary` compresses the same regions with a local on-device model (`providers.shakeSummaryModel`, default `qwen3-1.7b`) and falls back to elide per region when the model is unavailable; `/shake images` strips image blocks. Auto-maintenance honors the `shake` / `shake-summary` strategies (16k protect window); on context overflow a shake that reclaims nothing falls back to context-full summarization.
- Added `providers.shakeSummaryModel` setting selecting the local on-device model used by `/shake summary` and the `shake-summary` compaction strategy. Runs entirely on-device (downloads on first use) and never calls a remote/cloud LLM.
- Added `providers.autoThinkingModel` setting so users can choose the `auto` thinking classifier backend (online smol or local tiny-memory model)
- Added an `auto` thinking level that classifies each real user turn and resolves to a concrete low-through-xhigh effort, with online smol classification by default and an opt-in local on-device classifier.
- Added a `Web search` setup tab that lets users choose the preferred `providers.webSearch` provider during onboarding
- Added manual authorization-code/redirect URL prompts for OAuth providers that require non-callback login in the setup wizard
- Added an `omp completions <bash|zsh|fish>` command that prints a shell completion script generated from the live command/flag metadata, so completions never drift from the actual CLI. Subcommands, flags, and enum values complete statically; `--model`/`--smol`/`--slow`/`--plan` resolve against the bundled model catalog and `--resume` against on-disk sessions via a hidden `__complete` helper.
- Added a `/switch` slash command that opens the temporary model selector for the current session, mirroring the `alt+p` keybinding.
- Added `replace block N:` and `delete block N` operators to the `edit` tool: they resolve the syntactic block beginning on line N via tree-sitter (native `blockRangeAt`) and replace or delete its full line span, so a construct can be rewritten or removed without counting its closing line. Unresolvable blocks (unsupported language, blank/closing-delimiter line, or a parse error) are rejected with guidance to use an explicit `replace N..M:` / `delete N..M` range.
- Added an animated pending border for `bash` and `eval` execution blocks: while a command/cell is running, a single dark segment glides clockwise around the block's outer edge (top → right → bottom → left), replacing the previous static accent border. Motion is eased per edge (decelerating into each corner) and timed against a fixed lap duration mapped onto the live perimeter, so streaming a new output line or resizing the terminal nudges the segment proportionally instead of resetting its position. Driven by the existing spinner cadence and gated on the `display.shimmer` setting (no motion when `disabled`).
- Added `providers.tinyModelDevice` and `providers.tinyModelDtype` settings (Providers tab) controlling local tiny-model acceleration for session titles and Mnemopi memory tasks. `providers.tinyModelDevice` selects the ONNX execution provider (`default` keeps the platform pick — DirectML on Windows, CUDA on Linux x64, CPU elsewhere); `providers.tinyModelDtype` selects quantization/precision (`default` keeps each model's shipped `q4`, e.g. `fp16` trades speed for fidelity). The `PI_TINY_DEVICE` / `PI_TINY_DTYPE` env vars override the matching setting. Also added `PI_TINY_DTYPE` as the env counterpart to `PI_TINY_DEVICE`; an unrecognized device/precision fails loudly at worker startup instead of silently loading a different one.
- Added a bundled set of default rules shipped with the agent (TypeScript/Rust convention rules registered as TTSR conditions). They load via the new lowest-priority `builtin-defaults` discovery provider, so any user/project/tool rule of the same name overrides the bundled copy. Disable the whole set with `ttsr.builtinRules: false`, or drop individual rules (bundled or your own) by name via `ttsr.disabledRules`.
- Added a `symbols.spinnerFrames` field to custom theme JSON so themes can override the loader/tool-execution spinner. Accepts either a flat `string[]` (used for both spinner types) or `{ "status"?: string[], "activity"?: string[] }` to override each independently; anything not specified falls back to the symbol preset. Documented in `docs/theme.md` and validated by `theme-schema.json`. ([#1553](https://github.com/can1357/oh-my-pi/issues/1553))
- Added prompt-mode autocomplete for supported internal URL schemes (`skill://`, `rule://`, `agent://`, `artifact://`, `local://`, `memory://`, and `omp://`) so typing those tokens now suggests existing resources as completion candidates
- Added fuzzy matching and ranked suggestion ordering for internal URL completion, including rule and skill descriptions, with accepted completion replacing just the typed token and inserting the chosen URL followed by a space
- Changed internal URL completions now include nested `local://` path suggestions from the configured local workspace
- Added Mnemopi memory inference model selection with an online mode or local transformers.js options (`qwen3-1.7b`, `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`) so memory extraction and consolidation can run via the shared tiny-model worker
- Changed memory tiny-model handling to route local memory prompts through the same queueed tiny-model worker pipeline with bounded completion output
- Added a Providers → Tiny Model setting for session titles, defaulting to the online `pi/smol` path with five optional local CPU transformers.js models. A local model — and the one-time `@huggingface/transformers` runtime install in compiled binaries — is downloaded and loaded only when explicitly selected (or via `omp tiny-models download`); the default online path never spawns the title worker for inference. Selecting a local model adds a delayed `pi/smol` fallback so titles never block, plus in-chat download progress.
- Added a persistent live agent roster pinned below the editor (focus it with `Ctrl+S` or `Alt+Down`), including view-as switching into delegated agent sessions with human-readable delegate names and UI pinning to suppress idle reaping while viewed. The roster stays hidden until at least one delegated agent exists and releases focus back to the editor once the last one is gone.
- Recorded the originating session ID alongside each prompt in `history.db` (new `session_id` column, surfaced as `HistoryEntry.sessionId`), so recalled prompts can be traced back to the session they came from. Existing history databases gain the column automatically on next launch.
- Added compact inline TUI renderers for the `retain`, `recall`, and `reflect` memory tools. `retain` now shows one themed bullet line per stored item (truncated to width) under a status header with the stored/queued count, and `recall`/`reflect` collapse to a single query header (recall reports the match count and hides recalled memories until expanded) instead of dumping the raw JSON argument tree.
- Added a randomly picked tip beneath the welcome screen, sourced from an embedded `tips.txt` (one tip per line). The line is italicized with a purple `Tip:` label and a dimmed light-blue body, and the tip is chosen once per welcome instance so intro-animation and LSP re-renders don't shuffle it.
- Added a Mnemopi-only `memory_edit` agent tool for updating, forgetting, or invalidating recalled memories by id, and added `/memory stats` plus `/memory diagnose` slash commands for backend maintenance visibility.
- Added an `orchestrate` magic keyword that mirrors `ultrathink`: dropping the standalone word in a message paints it with a cool teal→violet gradient in the editor and appends a hidden system notice that switches the model into the multi-phase, parallel-subagent orchestration contract. Matching is word-bounded and case-insensitive, so `orchestrated`/`orchestrating` never trigger it.
- Added a model-tier slider to the plan-approval prompt ("Plan mode - next step"). Left/right arrows move it from any list position to pick which configured role model (`cycleOrder`, e.g. `smol › default › slow`) executes the approved plan, with each tier colored by its role and the resolved model name shown beneath the track. The chosen tier is applied before dispatch and carries through the fresh/compacted execution session; the slider is hidden when fewer than two role models resolve.
- `omp plugin install` now accepts GitHub/GitLab/Bitbucket shorthand (`github:user/repo`, `gitlab:user/repo`, …) and full git URLs (`https://github.com/user/repo`, `git@github.com:user/repo`, …) in addition to npm specs and marketplace refs.
- Added `ModelRegistry.create(authStorage, modelsPath?)` async factory that runs the JSON → YAML migration step on `models.{yml,yaml}` asynchronously ahead of the sync constructor's bundled-model load. The sync `new ModelRegistry(...)` constructor still works (tests rely on it); production boot paths now use the factory so the migration's I/O lands off the event-loop hot path.
- Added `ConfigFile.tryLoadAsync()`, `ConfigFile.loadAsync()`, `ConfigFile.loadOrDefaultAsync()`, `ConfigFile.getMtimeMsAsync()`, and `ConfigFile.warmup(file)` so the rest of the codebase can migrate config reads off the sync path.
### Changed
@@ -612,7 +901,6 @@
- Fixed `/login` API-key prompts (OpenCode Zen, Perplexity OTP, GitHub Enterprise URL, manual OAuth redirect URL, …) silently dropping pasted content on kitty/Linux/Wayland — and any other terminal supporting OSC 5522 enhanced paste. `InputController` enables kitty's enhanced clipboard protocol on TUI start and consumes the resulting OSC 5522 packets in an `addInputListener` that runs before focus dispatch, so the paste never reached the modal `Input`'s bracketed-paste handler; the routing then stuffed the text into the main `CustomEditor` unconditionally, even when `selector-controller` had detached the editor and focused a temporary OAuth input. The pasted API key accumulated in the hidden editor and only resurfaced in the main prompt when the user dismissed the modal with Enter or Esc. The enhanced-paste callback now consults `ui.getFocused()` and routes the text to the focused component when it exposes a `pasteText` hook, falling back to the editor only when no modal target is in focus; image pastes refuse with a status message instead of stuffing a binary blob into the hidden editor. ([#2127](https://github.com/can1357/oh-my-pi/issues/2127))
- Fixed an auto-compaction dead loop when `compaction.strategy` was `shake` and the configured threshold was low enough that a single shake pass could not bring the context below it (e.g. a 50K-token threshold on a session well above it). Each pass auto-continued, the next agent turn re-triggered the threshold check, and the second shake had nothing new to drop, so the session spun forever. The shake recovery path now estimates post-shake context and, when it is still above the threshold (or shake reclaimed nothing on overflow recovery), surfaces a one-shot warning and falls back to the summarization-driven `context-full` compaction so progress actually resumes ([#2119](https://github.com/can1357/oh-my-pi/issues/2119)).
- Fixed `/skill:` prompts so magic keywords and turn-budget directives in skill args inject the same hidden notices as normal user prompts, matching the editor highlight behavior ([#2128](https://github.com/can1357/oh-my-pi/issues/2128)).
- Fixed MCP OAuth fallback rendering to show a short terminal hyperlink and keep the raw authorization URL on one unwrapped copy line ([#2121](https://github.com/can1357/oh-my-pi/issues/2121)).
- Fixed the `task` tool rendering a success bullet and a `success` frame state for detail-less error results (e.g. an argument-validation failure that never executes): the header now shows the error glyph with an error border and `error` state, and surfaces the dispatched agent name.
- Fixed Agent Control Center new-agent creation so Windows Ctrl+Enter sequences submitted as a single LF generate the agent instead of inserting a newline ([#2118](https://github.com/can1357/oh-my-pi/issues/2118)).
- Fixed plan-mode subagents preserving read-only specialty tools such as `report_finding` while still stripping mutating tools ([#1998](https://github.com/can1357/oh-my-pi/issues/1998)).
@@ -778,7 +1066,7 @@
- Fixed the Python `read(path, offset, limit)` prelude helper rejecting documented positional arguments with `TypeError: read() takes 1 positional argument but 3 were given`. The signature was keyword-only (`def read(path, *, offset=1, limit=None)`) while the eval helper table advertises positional optional args; agents that called `read("file.py", 10, 20)` literally crashed. The `*` is removed so both `read("f", 10, 20)` and `read("f", offset=10, limit=20)` work.
- Fixed `eval` reset cells failing with `"Python kernel reset already in progress"` / `"JS context reset already in progress"` when two cells happened to overlap on the same session (e.g. a rapid resubmit, or a parallel-cell race). The executor now coalesces concurrent resets — additional callers wait for the in-flight reset to finish and then run on the freshly restarted kernel — instead of throwing a user-visible error for what is purely an internal coordination state.
- Fixed the `eval` tool description advertising the `agent()` helper unconditionally even in subagent sessions whose parent forbids spawning. When `getSessionSpawns()` returns `""`, the prelude doc now omits `agent()` so the model is not promised a helper that can only ever throw "Cannot spawn 'task'. Allowed: none (spawns disabled for this agent)".
- Fixed plan mode rejecting `local://` plan-artifact edits when addressed via the absolute path the `read` tool echoes back in the `[path#tag]` header. `enforcePlanModeWrite` previously only matched the literal `local://` scheme; it now also accepts any absolute path whose realpath resolves inside the session's local sandbox root, so the absolute spelling and the `local://` spelling are interchangeable in plan mode.
- Fixed plan mode rejecting `local://` plan-artifact edits when addressed via the absolute path the `read` tool echoes back in the `[path#tag]` header. `enforcePlanModeWrite` now accepts bracketed hashline headers as well as clean absolute paths whose realpath resolves inside the session's local sandbox root, so the absolute spelling, the `[absolute#tag]` edit header, and the `local://` spelling are interchangeable in plan mode. ([#2472](https://github.com/can1357/oh-my-pi/issues/2472))
- Fixed snapshot tags freshly minted by `read` being rejected as stale by a subsequent `edit` against the same file when the two sides reached the file via symlink-equivalent spellings (e.g. macOS `/tmp/…` vs `/private/tmp/…`, or `read local://foo.md` recording under the file's `fs.realpath` while `edit local://foo.md` looked up under the raw `path.resolve(localRoot, …)` form). The file snapshot store now keys every record/lookup through a `realpath`-canonicalized key (`canonicalSnapshotKey`), fusing all spellings of the same on-disk file onto one snapshot entry.
- Fixed `read` of a `github.com/<owner>/<repo>` URL with `:raw` returning the full JS-rendered HTML shell. Repo roots now resolve to the decoded README via the GitHub API (`/repos/<owner>/<repo>/readme`), falling back to the raw HTML only when the API returns no usable payload.
- Fixed `issue://` and `pr://` reads returning stale OPEN/CLOSED state after a successful `gh issue close` / `gh pr merge` (or any other state-changing `gh` invocation) in the same session. The `bash` tool now invalidates the matching `github-cache` rows before executing any `gh (issue|pr) <close|reopen|merge|delete|edit|comment|lock|unlock|pin|unpin|transfer|develop|ready|review>` command.
@@ -866,7 +1154,6 @@
- Fixed tool-output file paths not being clickable OSC 8 `file://` hyperlinks in several renderers. `read` titles for plain text and image files (the common case) emitted no link at all because the renderer only linked when a `resolvedPath` was recorded — which the ordinary file/image read paths never set, keeping the absolute path only in `meta.source`; the renderer now falls back to that source path. `write` headers were never wrapped in a hyperlink and now link to the absolute path written (file, archive entry, SQLite, and conflict resolutions). `edit`/`apply_patch` headers wrapped the model-supplied (often cwd-relative) argument path, producing a root-anchored `file:///rel/path` URI; they now link the absolute `details.path` instead. Finally, `search`, `ast_grep`, and `ast_edit` produced doubled link targets (`/proj/src/src/file.ts`) for searches scoped to a subdirectory, because the renderer resolved the cwd-relative display paths against the scope directory rather than cwd — the scoped-search base is now the session cwd (with the scoped file's absolute path still seeding single-file body lines).
- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts
- Fixed the bash tool corrupting commands that embed multi-byte UTF-8 (e.g. `✓`/`×` inside a `grep -E` pattern) ahead of a trailing `| head`/`| tail`. The `bash.stripTrailingHeadTail` rewrite cut at char-offset positions reported by `brush-parser` while slicing the command by byte offset, so the trailing-pipe strip landed mid-pattern and dropped the closing quote — turning `… |✓|×|XCTAssert" | tail -80` into `… |✓|×-80` and making execution fail with `pi-natives:command: unterminated double quote`. Fixed in `pi_shell::fixup` (`@oh-my-pi/pi-natives`).
- Fixed `omp dry-balance --bench` to recover from 401 token failures by re-minting the failing OAuth credential in place before switching accounts
- Fixed duplicate file entries in grouped outputs for `find`, `search`, `ast_grep`, `ast_edit`, and `lsp` diagnostics when the same path appeared multiple times
- Fixed search, grep, and edit output rendering so repeated directory group blank-line boundaries no longer break nested path/link reconstruction
- Fixed `omp dry-balance --bench` flooding the terminal with staircased, duplicated spinner/status lines (and an indented summary) when the tty has ONLCR/OPOST disabled (raw mode). The interactive progress region separated rows with a bare LF and repositioned with a column-preserving `\x1b[<n>A` cursor-up, both of which only land at column 0 when the terminal translates LF→CRLF; with that translation off, every 80 ms redraw cascaded down and to the right into scrollback. The live region now carriage-returns before every cleared row, terminates each row with CRLF, and caps each row to the terminal width so a wrapped line cannot desync the cursor-up from the logical line count.
@@ -1379,7 +1666,6 @@
- Fixed the agent roster staying pinned under the editor when all delegated agents are idle or dormant; it now reappears when explicitly focused with `Alt+Down` / session observe.
- Fixed selector-style UI components to honor `tui.select.up` and `tui.select.down` keybindings instead of hard-coding raw Up/Down arrow bytes ([#1535](https://github.com/can1357/oh-my-pi/issues/1535)).
- Fixed the bash (and `recipe`) tool result footer not rendering for failed commands. A non-zero exit threw a `ToolError`, which dropped the result details, so the styled `⟨Wall … | Timeout …⟩` footer was replaced by the raw `Wall time: … seconds` / `Command exited with code N` lines. Non-zero exits now resolve as a non-throwing error result that keeps `wallTimeMs`/`timeoutSeconds`/`exitCode`, and the footer shows `⟨Wall … | Timeout … | Exit: N⟩` with the textual notices folded out of the output pane. Aborts, timeouts, and missing-exit-status still throw as before.
- Fixed selector-style UI components to honor `tui.select.up` and `tui.select.down` keybindings instead of hard-coding raw Up/Down arrow bytes ([#1535](https://github.com/can1357/oh-my-pi/issues/1535)).
### Removed
@@ -1581,8 +1867,6 @@
- Added `read.summarize.minTotalLines` setting (default 100) to set the minimum file length that triggers read summarization
- Added `<file>:<lines>` support to `search` `paths`, allowing file-scoped constraints such as `:N-M`, `:N+K`, and comma-separated ranges
- Added `ModelRegistry.create(authStorage, modelsPath?)` async factory that runs the JSON → YAML migration step on `models.{yml,yaml}` asynchronously ahead of the sync constructor's bundled-model load. The sync `new ModelRegistry(...)` constructor still works (tests rely on it); production boot paths now use the factory so the migration's I/O lands off the event-loop hot path.
- Added `ConfigFile.tryLoadAsync()`, `ConfigFile.loadAsync()`, `ConfigFile.loadOrDefaultAsync()`, `ConfigFile.getMtimeMsAsync()`, and `ConfigFile.warmup(file)` so the rest of the codebase can migrate config reads off the sync path.
### Changed
@@ -10371,4 +10655,4 @@ Initial public release.
- Git branch display in footer
- Message queueing during streaming responses
- OAuth integration for Gmail and Google Calendar access
- HTML export with syntax highlighting and collapsible sections
- HTML export with syntax highlighting and collapsible sections
+4 -3
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-coding-agent",
"version": "15.12.5",
"version": "15.13.0",
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
"homepage": "https://omp.sh",
"author": "Can Boluk",
@@ -35,7 +35,7 @@
"check": "biome check . && bun run check:types",
"check:types": "tsgo -p tsconfig.json --noEmit",
"lint": "biome lint .",
"test": "bun test --parallel",
"test": "bun test --parallel=2",
"fix": "biome check --write --unsafe . && bun run format-prompts && bun run generate-docs-index",
"fmt": "biome format --write . && bun run format-prompts",
"format-prompts": "bun scripts/format-prompts.ts",
@@ -80,7 +80,8 @@
"zod": "catalog:"
},
"optionalDependencies": {
"@huggingface/transformers": "catalog:"
"@huggingface/transformers": "catalog:",
"sherpa-onnx-node": "1.13.2"
},
"devDependencies": {
"@types/bun": "catalog:"
@@ -6,6 +6,27 @@ const DELIVERY_RETRY_JITTER_MS = 200;
const DEFAULT_RETENTION_MS = 5 * 60 * 1000;
const DEFAULT_MAX_RUNNING_JOBS = 15;
/**
* Adaptive ("smart") `job` poll-wait ladder (ms). A tight poll loop climbs
* these rungs so each immediate re-poll backs off and stops spending turns on
* "still running" frames; the floor (first rung) is the shortest wait and the
* top rung is the longest a smart poll will ever block. Only used when
* `async.pollWaitDuration` is set to `smart`; fixed durations wait verbatim.
*/
const POLL_WAIT_LADDER_MS = [5_000, 10_000, 30_000, 60_000, 300_000] as const;
/**
* Going at least this long between poll calls means the agent stepped out of
* the poll loop to do real work — the next poll drops back to the ladder floor.
*/
const POLL_ESCALATION_RESET_MS = 60_000;
interface PollEscalationState {
/** Index into POLL_WAIT_LADDER_MS used for the most recent poll wait. */
level: number;
/** Timestamp (ms) when the most recent poll wait returned. */
lastPollEndAt: number;
}
export interface AsyncJob {
id: string;
type: "bash" | "task";
@@ -96,6 +117,7 @@ export class AsyncJobManager {
readonly #suppressedDeliveries = new Set<string>();
readonly #watchedJobs = new Set<string>();
readonly #evictionTimers = new Map<string, NodeJS.Timeout>();
readonly #pollEscalation = new Map<string | undefined, PollEscalationState>();
readonly #onJobComplete: AsyncJobManagerOptions["onJobComplete"];
readonly #maxRunningJobs: number;
readonly #retentionMs: number;
@@ -295,6 +317,32 @@ export class AsyncJobManager {
return removed;
}
/**
* Compute the next adaptive ("smart") wait (ms) for a blocking `job` poll by
* the given owner. Consecutive polls — those starting within
* POLL_ESCALATION_RESET_MS of the previous poll returning — climb
* POLL_WAIT_LADDER_MS so a tight wait loop backs off; a longer gap means the
* agent left to do real work, so the wait resets to the floor. Pair each call
* with `recordPollWaitEnd()` once the wait returns.
*/
nextPollWaitMs(ownerId: string | undefined, now: number = Date.now()): number {
const prev = this.#pollEscalation.get(ownerId);
const reset = !prev || now - prev.lastPollEndAt >= POLL_ESCALATION_RESET_MS;
const level = reset ? 0 : Math.min(prev.level + 1, POLL_WAIT_LADDER_MS.length - 1);
this.#pollEscalation.set(ownerId, { level, lastPollEndAt: prev?.lastPollEndAt ?? now });
return POLL_WAIT_LADDER_MS[level];
}
/**
* Mark a blocking poll wait as finished so the idle-reset window is measured
* from now. Polling again before POLL_ESCALATION_RESET_MS elapses keeps
* climbing the ladder; waiting longer resets it to the floor.
*/
recordPollWaitEnd(ownerId: string | undefined, now: number = Date.now()): void {
const prev = this.#pollEscalation.get(ownerId);
this.#pollEscalation.set(ownerId, { level: prev?.level ?? 0, lastPollEndAt: now });
}
acknowledgeDeliveries(jobIds: string[]): number {
const uniqueJobIds = Array.from(new Set(jobIds.map(id => id.trim()).filter(id => id.length > 0)));
if (uniqueJobIds.length === 0) return 0;
@@ -405,6 +453,7 @@ export class AsyncJobManager {
this.#inFlightDeliveries.length = 0;
this.#suppressedDeliveries.clear();
this.#watchedJobs.clear();
this.#pollEscalation.clear();
return drained;
}
@@ -0,0 +1,139 @@
/**
* Auto-learn session controller (experimental).
*
* Subscribes to the session event stream and, after a substantive turn,
* nudges the agent to capture reusable lessons. Default posture is passive
* (a hidden reminder rides the next real turn); with `autolearn.autoContinue`
* it auto-runs exactly one synthetic capture turn at stop.
*
* Installed once per top-level session (taskDepth 0). The subscription lives
* for the session's lifetime — `newSession` resets the session in place
* without re-running startup — so the controller needs no disposal.
*/
import { logger } from "@oh-my-pi/pi-utils";
import type { Settings } from "../config/settings";
import autolearnGuidance from "../prompts/system/autolearn-guidance.md" with { type: "text" };
import autolearnGuidanceLearn from "../prompts/system/autolearn-guidance-learn.md" with { type: "text" };
import autolearnNudge from "../prompts/system/autolearn-nudge.md" with { type: "text" };
import type { AgentSession, AgentSessionEvent } from "../session/agent-session";
const AUTOLEARN_NUDGE = autolearnNudge.trim();
const DEFAULT_MIN_TOOL_CALLS = 5;
/**
* Build the standing auto-learn guidance for the system prompt from the tools
* actually present in the active set, or null when `manage_skill` is absent.
*
* Driven by tool presence rather than live settings: the `learn`/`manage_skill`
* registry is built ONCE at session start (and only for top-level sessions), so
* keying the guidance on `autolearn.enabled` would let a mid-session enable — or
* a subagent that filtered the tools out — inject guidance pointing at tools the
* session never built. The `learn` addendum is included only when the `learn`
* tool is present (it requires a memory backend).
*/
export function buildAutoLearnInstructions(available: { manageSkill: boolean; learn: boolean }): string | null {
if (!available.manageSkill) return null;
const parts = [autolearnGuidance.trim()];
if (available.learn) parts.push(autolearnGuidanceLearn.trim());
return parts.join("\n\n");
}
export interface AutoLearnControllerOptions {
session: AgentSession;
settings: Settings;
}
export class AutoLearnController {
readonly #session: AgentSession;
readonly #settings: Settings;
#toolCalls = 0;
/**
* Whether the in-flight turn BEGAN while goal mode was active. Captured at
* agent_start because a `goal` tool can complete or drop the goal mid-turn,
* clearing the live flag before agent_end — so the end-of-turn state alone
* would let a goal-continuation turn slip through and get nudged.
*/
#turnStartedInGoalMode = false;
/** Swallow the agent_end produced by an auto-run capture turn so it cannot re-trigger. */
#suppressNext = false;
constructor(options: AutoLearnControllerOptions) {
this.#session = options.session;
this.#settings = options.settings;
// The listener closure captures `this`, so the session's listener array
// keeps the controller alive — no stored unsubscribe needed.
this.#session.subscribe(event => this.#onEvent(event));
}
#onEvent(event: AgentSessionEvent): void {
if (event.type === "agent_start") {
// Capture goal-mode state at the turn boundary, before any tool runs.
this.#turnStartedInGoalMode = this.#session.getGoalModeState()?.enabled === true;
return;
}
if (event.type === "tool_execution_end") {
this.#toolCalls++;
return;
}
if (event.type === "agent_end") {
this.#onAgentEnd();
}
}
#onAgentEnd(): void {
// Snapshot and reset every turn: the counter describes only the
// just-finished turn, so below-threshold, disabled, and plan-mode stops
// must not let tool calls accumulate into a later turn.
const toolCalls = this.#toolCalls;
this.#toolCalls = 0;
// Snapshot the turn-start goal flag alongside the counter so a turn that
// observed no agent_start can never inherit a stale value.
const startedInGoalMode = this.#turnStartedInGoalMode;
this.#turnStartedInGoalMode = false;
if (this.#suppressNext) {
this.#suppressNext = false;
return;
}
// Honor a live opt-out: the subscription outlives the setting, so re-check
// the current flag rather than trusting install-time state.
if (!this.#settings.get("autolearn.enabled")) return;
const minToolCalls = this.#settings.get("autolearn.minToolCalls") ?? DEFAULT_MIN_TOOL_CALLS;
if (toolCalls < minToolCalls) return;
// Never interrupt plan-mode review.
if (this.#session.getPlanModeState()?.enabled) return;
// Never divert a goal loop. Skip when the turn STARTED in goal mode — a
// `goal` tool may have completed/dropped the goal before this stop — or is
// still in it: a passive nudge would ride the goal continuation, and
// auto-continue would compete with it.
if (startedInGoalMode || this.#session.getGoalModeState()?.enabled) return;
// Auto-run a capture turn only when explicitly enabled; otherwise the
// hidden reminder rides the next real turn passively.
const autoContinue = this.#settings.get("autolearn.autoContinue") === true;
// Arm suppression synchronously: the synthetic capture turn's agent_end
// fires inside sendCustomMessage (before it resolves), so the flag must be
// set before then. Disarm when no turn actually started — a deferred/queued
// dispatch or a failed send produces no agent_end, and a latched flag would
// otherwise swallow the next real stop.
if (autoContinue) this.#suppressNext = true;
this.#session
.sendCustomMessage(
{
customType: "autolearn-nudge",
content: AUTOLEARN_NUDGE,
display: false,
attribution: "user",
},
{ deliverAs: "nextTurn", triggerTurn: autoContinue },
)
.then(started => {
if (!started) this.#suppressNext = false;
})
.catch(err => {
this.#suppressNext = false;
logger.warn("auto-learn nudge delivery failed", { err });
});
}
}
@@ -0,0 +1,257 @@
/**
* Managed-skills primitives for the experimental auto-learn feature.
*
* Managed skills are auto-generated/enhanced `SKILL.md` files kept in an
* isolated directory (`~/.omp/agent/managed-skills`) separate from
* user-authored skills (`~/.omp/agent/skills`). They are discovered and
* surfaced like normal skills, but every write here is confined to
* `getManagedSkillsDir()` — auto-management can never touch authored skills.
*/
import { constants as fsConstants, type Stats } from "node:fs";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { isEnoent } from "@oh-my-pi/pi-utils";
import { YAML } from "bun";
import { SOURCE_PATHS } from "../discovery/helpers";
/** Provider id stamped on discovered managed skills (distinguishes them from authored). */
export const MANAGED_SKILLS_PROVIDER_ID = "omp-managed";
/** Hard cap on a managed SKILL.md body to keep generated skills bounded. */
export const MAX_MANAGED_SKILL_BYTES = 64_000;
const SKILL_NAME_PATTERN = /^[a-z0-9][a-z0-9-]{0,63}$/;
/** Resolve the isolated managed-skills directory (`~/.omp/agent/managed-skills`). */
export function getManagedSkillsDir(home: string = os.homedir()): string {
return path.join(home, SOURCE_PATHS.native.userAgent, "managed-skills");
}
/**
* Validate + normalize a managed-skill name. Throws on anything outside the
* strict allowlist so a bad name can never escape `getManagedSkillsDir()`
* (blocks `..`, slashes, empty, and uppercase).
*/
export function sanitizeSkillName(raw: string): string {
const name = raw.trim().toLowerCase();
if (!SKILL_NAME_PATTERN.test(name)) {
throw new Error(
`Invalid skill name "${raw}". Use lowercase letters, digits, and hyphens (1-64 chars, starting with a letter or digit).`,
);
}
return name;
}
/**
* Whether `name` is a safe managed-skill name (the exact post-sanitize shape).
* Used to validate names read from disk at discovery time — a managed
* `SKILL.md` whose `frontmatter.name` was not produced by `sanitizeSkillName`
* (e.g. hand-placed) must not render unescaped into the system prompt.
*/
export function isValidManagedSkillName(name: string): boolean {
return SKILL_NAME_PATTERN.test(name);
}
/**
* Neutralize a machine-generated managed-skill description so it cannot break
* out of the system prompt's `<skills>` listing. Managed descriptions are
* generated from prior task content and persist across sessions, so this is a
* trust boundary: strip control/format chars, angle brackets (`<system-directive>`
* / `</skills>`), and Markdown fence delimiters (backticks, `~~~`), then collapse
* to a single line. Applied on BOTH write and read so existing files are safe too.
*/
export function sanitizeManagedDescription(raw: string): string {
return raw
.replace(/[\p{Cc}\p{Cf}]/gu, " ")
.replace(/[<>`]/g, "")
.replace(/~{2,}/g, "~")
.replace(/\s+/g, " ")
.trim();
}
/**
* Serialize the minimal `name`/`description` frontmatter block via the repo's
* YAML helper (round-trips through `parseFrontmatter`).
*/
export function toSkillFrontmatter(name: string, description: string): string {
const frontmatter = YAML.stringify(
{ name, description: sanitizeManagedDescription(description) },
null,
2,
).trimEnd();
return `---\n${frontmatter}\n---\n`;
}
export interface WriteManagedSkillInput {
action: "create" | "update";
name: string;
description: string;
body: string;
}
/**
* Serialize create/update/delete on the same skill name. Both tools are
* non-exclusive, so a parallel tool batch in one turn can run two mutations on
* the same skill at once (e.g. an update observing the file mid-delete). This
* per-name promise chain runs same-skill mutations in submission order while
* different names still proceed in parallel. In-process only; cross-process
* races are out of scope.
*/
const skillMutationChains = new Map<string, Promise<unknown>>();
function serializeSkillMutation<T>(name: string, op: () => Promise<T>): Promise<T> {
const prev = skillMutationChains.get(name) ?? Promise.resolve();
const run = prev.then(op, op);
const guarded = run.catch(() => {});
skillMutationChains.set(name, guarded);
void guarded.finally(() => {
if (skillMutationChains.get(name) === guarded) skillMutationChains.delete(name);
});
return run;
}
/**
* Reject when the managed-skills root itself is a symlink. lstat on a child
* follows intermediate components, so a symlinked root would let an otherwise
* valid name write/delete outside the isolated directory (e.g. onto authored
* skills). Checked before composing any child path.
*/
async function assertManagedRootSafe(): Promise<void> {
const rootStat = await fs.lstat(getManagedSkillsDir()).catch(err => {
if (isEnoent(err)) return null;
throw err;
});
if (rootStat?.isSymbolicLink()) {
throw new Error("The managed-skills root is a symlink; refusing to operate outside the managed directory.");
}
}
const UPDATE_FILE_OPEN_FLAGS = fsConstants.O_WRONLY | fsConstants.O_NOFOLLOW;
function assertManagedSkillFileSafeForUpdate(name: string, fileStat: Stats): void {
if (!fileStat.isFile()) {
throw new Error(`Managed skill "${name}" SKILL.md is not a regular file; refusing to overwrite it.`);
}
if (fileStat.nlink > 1) {
throw new Error(
`Managed skill "${name}" SKILL.md has ${fileStat.nlink} hard links; refusing to overwrite a file that may be user-authored elsewhere.`,
);
}
}
async function openManagedSkillFileForUpdate(name: string, file: string) {
try {
return await fs.open(file, UPDATE_FILE_OPEN_FLAGS);
} catch (err) {
if ((err as { code?: string }).code === "ELOOP") {
throw new Error(`Managed skill "${name}" SKILL.md is a symlink; refusing to overwrite it.`);
}
throw err;
}
}
/** Create or update a managed `SKILL.md`. Returns the resolved file path. */
export async function writeManagedSkill(input: WriteManagedSkillInput): Promise<{ path: string }> {
const name = sanitizeSkillName(input.name);
const description = sanitizeManagedDescription(input.description);
const body = input.body.trim();
// Reject empty content: an all-whitespace/control description sanitizes to ""
// and the `requireDescription` discovery scan then silently drops the skill,
// so the tool would report success for a skill that never appears.
if (!description) {
throw new Error(`Managed skill "${name}" needs a non-empty description.`);
}
if (!body) {
throw new Error(`Managed skill "${name}" needs a non-empty body.`);
}
const content = `${toSkillFrontmatter(name, description)}\n${body}\n`;
// Cap the UTF-8 byte size of the FINAL file (body + description + frontmatter),
// not the UTF-16 code-unit length of the body alone.
const bytes = Buffer.byteLength(content, "utf8");
if (bytes > MAX_MANAGED_SKILL_BYTES) {
throw new Error(
`Managed skill is ${bytes} bytes; the limit is ${MAX_MANAGED_SKILL_BYTES}. Trim the body or description.`,
);
}
return serializeSkillMutation(name, async () => {
await assertManagedRootSafe();
const dir = path.join(getManagedSkillsDir(), name);
const file = path.join(dir, "SKILL.md");
// Reject a symlinked skill directory: an intermediate symlink would let the
// write escape the isolated managed root. lstat does not follow the final
// component, so a symlinked `dir` is caught here.
const dirStat = await fs.lstat(dir).catch(err => {
if (isEnoent(err)) return null;
throw err;
});
if (dirStat?.isSymbolicLink()) {
throw new Error(
`Managed skill "${name}" resolves through a symlink; refusing to write outside the managed directory.`,
);
}
if (input.action === "create") {
await fs.mkdir(dir, { recursive: true });
// O_CREAT|O_EXCL ("wx"): atomic create that fails if the file already
// exists (closing the check-then-write race) and refuses a symlinked SKILL.md.
try {
await fs.writeFile(file, content, { flag: "wx" });
} catch (err) {
if ((err as { code?: string }).code === "EEXIST") {
throw new Error(`Managed skill "${name}" already exists. Use action "update" to change it.`);
}
throw err;
}
return { path: file };
}
// update: the file must already exist, be a plain managed file, and must
// not share an inode with a user-authored file via hard link. Open the
// checked file handle before truncating so a path swap after lstat cannot
// redirect the write into a symlink or newly hard-linked target.
const fileStat = await fs.lstat(file).catch(err => {
if (isEnoent(err)) return null;
throw err;
});
if (fileStat === null) {
throw new Error(`Managed skill "${name}" does not exist. Use action "create" to add it.`);
}
if (fileStat.isSymbolicLink()) {
throw new Error(`Managed skill "${name}" SKILL.md is a symlink; refusing to overwrite it.`);
}
assertManagedSkillFileSafeForUpdate(name, fileStat);
const handle = await openManagedSkillFileForUpdate(name, file);
try {
const openStat = await handle.stat();
assertManagedSkillFileSafeForUpdate(name, openStat);
await handle.truncate(0);
await handle.writeFile(content);
} finally {
await handle.close();
}
return { path: file };
});
}
/** Delete a managed skill directory. Throws when it does not exist. */
export async function deleteManagedSkill(name: string): Promise<void> {
const safe = sanitizeSkillName(name);
await serializeSkillMutation(safe, async () => {
await assertManagedRootSafe();
const dir = path.join(getManagedSkillsDir(), safe);
// Refuse to follow a symlinked skill directory (rm would delete the target).
const dirStat = await fs.lstat(dir).catch(err => {
if (isEnoent(err)) return null;
throw err;
});
if (dirStat?.isSymbolicLink()) {
throw new Error(`Managed skill "${safe}" is a symlink; refusing to delete outside the managed directory.`);
}
try {
await fs.rm(dir, { recursive: true });
} catch (err) {
if (isEnoent(err)) {
throw new Error(`Managed skill "${safe}" does not exist.`);
}
throw err;
}
});
}
@@ -1,4 +1,4 @@
import type { SessionEntry } from "../session/session-manager";
import type { SessionEntry } from "../session/session-entries";
import { inferMetricUnitFromName, isBetter } from "./helpers";
import type { RunRow, SessionRow } from "./storage";
import type {
@@ -1,6 +1,6 @@
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import type { ExtensionAPI, ExtensionContext } from "../extensibility/extensions";
import type { SessionEntry } from "../session/session-manager";
import type { SessionEntry } from "../session/session-entries";
import type { TruncationResult } from "../session/streaming-output";
export type MetricDirection = "lower" | "higher";
@@ -29,6 +29,7 @@ export const commands: CommandEntry[] = [
{ name: "join", load: () => import("./commands/join").then(m => m.default) },
{ name: "models", load: () => import("./commands/models").then(m => m.default) },
{ name: "plugin", load: () => import("./commands/plugin").then(m => m.default) },
{ name: "say", load: () => import("./commands/say").then(m => m.default) },
{ name: "setup", load: () => import("./commands/setup").then(m => m.default) },
{ name: "shell", load: () => import("./commands/shell").then(m => m.default) },
{ name: "read", load: () => import("./commands/read").then(m => m.default) },
+45 -13
View File
@@ -64,6 +64,8 @@ async function showHelp(config: CliConfig): Promise<void> {
async function runSmokeTest(): Promise<void> {
const { smokeTestSyncWorker, startServer } = await import("@oh-my-pi/omp-stats");
const { smokeTestTinyTitleWorker } = await import("./tiny/title-client");
const { smokeTestSttWorker } = await import("./stt/asr-client");
const { smokeTestTtsWorker } = await import("./tts/tts-client");
await smokeTestSyncWorker();
const statsServer = await startServer(0);
@@ -79,6 +81,8 @@ async function runSmokeTest(): Promise<void> {
}
await smokeTestTinyTitleWorker();
await smokeTestSttWorker();
await smokeTestTtsWorker();
process.stdout.write("smoke-test: ok\n");
}
@@ -86,6 +90,8 @@ const TINY_WORKER_ARGS = new Set(["--tiny-worker", "__tiny_worker"]);
const STATS_SYNC_WORKER_ARG = "__omp_stats_sync_worker";
const TAB_WORKER_ARG = "__omp_tab_worker";
const JS_EVAL_WORKER_ARG = "__omp_js_eval_worker";
const STT_WORKER_ARG = "__omp_stt_worker";
const TTS_WORKER_ARG = "__omp_tts_worker";
async function runWorkerEntrypoint(arg: string | undefined): Promise<boolean> {
if (arg === STATS_SYNC_WORKER_ARG) {
@@ -118,21 +124,34 @@ async function runWorkerEntrypoint(arg: string | undefined): Promise<boolean> {
await import("./eval/js/worker-entry");
return true;
}
if (arg === STT_WORKER_ARG) {
const { startSttWorker } = await import("./stt/asr-worker");
await runIpcSubprocessWorker(startSttWorker);
return true;
}
if (arg === TTS_WORKER_ARG) {
const { startTtsWorker } = await import("./tts/tts-worker");
await runIpcSubprocessWorker(startTtsWorker);
return true;
}
return false;
}
/**
* Hidden subcommand that boots the tiny-model worker inside this process
* over the parent's IPC channel. The agent's main process spawns the same
* binary with this flag so `onnxruntime-node` (loaded transitively by
* `@huggingface/transformers`) lives in a child address space. The parent
* `SIGKILL`s the child on shutdown so the NAPI finalizer never runs in
* either process — that finalizer segfaults Bun on Windows (issue #1606).
* Boot a subprocess-isolated transformers.js worker over the parent's IPC
* channel and block until the parent disconnects. The tiny-model, STT, and TTS
* workers each run `onnxruntime-node` (loaded transitively by
* `@huggingface/transformers`) in a child address space because its NAPI
* finalizer segfaults Bun on shutdown (issue #1606); the parent `SIGKILL`s the
* child so that finalizer never runs in either process. This wires `process`
* IPC to the worker's typed transport, keeps the event loop alive while the
* worker is idle, and hard-kills the process on parent `disconnect`.
*/
async function runTinyWorker(): Promise<void> {
const { startTinyTitleWorker } = await import("./tiny/worker");
async function runIpcSubprocessWorker<In, Out>(
start: (transport: { send(message: Out): void; onMessage(handler: (message: In) => void): () => void }) => void,
): Promise<void> {
const { promise: shuttingDown, resolve: shutdown } = Promise.withResolvers<void>();
const send = (message: unknown): void => {
const send = (message: Out): void => {
// `process.send` only exists when spawned with an IPC channel; the
// parent always spawns us that way. If it's missing, the parent
// vanished and there's no one to talk to.
@@ -147,10 +166,10 @@ async function runTinyWorker(): Promise<void> {
shutdown();
}
};
startTinyTitleWorker({
start({
send,
onMessage(handler) {
const wrap = (data: unknown): void => handler(data as never);
const wrap = (data: unknown): void => handler(data as In);
process.on("message", wrap);
return () => {
process.off("message", wrap);
@@ -159,8 +178,8 @@ async function runTinyWorker(): Promise<void> {
});
const keepalive = setInterval(() => {}, 2 ** 30);
// Parent went away (crashed, SIGKILL, etc.) — commit suicide so we don't
// linger as an orphan. SIGKILL via `process.kill` keeps us symmetrical
// with the parent's hard-kill on shutdown: skip every JS/native finalizer.
// linger as an orphan. SIGKILL via `process.kill` keeps us symmetrical with
// the parent's hard-kill on shutdown: skip every JS/native finalizer.
process.on("disconnect", () => shutdown());
try {
await shuttingDown;
@@ -170,6 +189,19 @@ async function runTinyWorker(): Promise<void> {
process.kill(process.pid, "SIGKILL");
}
/**
* Hidden subcommand that boots the tiny-model worker inside this process over
* the parent's IPC channel. The agent's main process spawns the same binary
* with this flag so `onnxruntime-node` (loaded transitively by
* `@huggingface/transformers`) lives in a child address space. The parent
* `SIGKILL`s the child on shutdown so the NAPI finalizer never runs in either
* process — that finalizer segfaults Bun on Windows (issue #1606).
*/
async function runTinyWorker(): Promise<void> {
const { startTinyTitleWorker } = await import("./tiny/worker");
await runIpcSubprocessWorker(startTinyTitleWorker);
}
/** Run the CLI with the given argv (no `process.argv` prefix). */
export async function runCli(argv: string[]): Promise<void> {
let resolvedArgv = argv;
+12 -13
View File
@@ -105,20 +105,14 @@ export function parseArgs(inputArgs: string[], extensionFlags?: Map<string, { ty
};
// `--` ends option parsing (POSIX end-of-options). Everything after it is
// literal positional text, so `omp -- --profile work` sends the tokens
// `--profile` and `work` as the message instead of selecting a profile.
let passThrough = false;
// literal positional text, so flag-shaped messages are not parsed or rejected.
let sawSeparator = false;
for (let i = 0; i < args.length; i++) {
let arg = args[i];
if (passThrough) {
if (sawSeparator) {
result.messages.push(arg);
continue;
}
if (arg === "--") {
passThrough = true;
continue;
}
if (arg === PROFILE_BOOTSTRAP_BOUNDARY_ARG) {
continue;
}
@@ -208,10 +202,15 @@ export function parseArgs(inputArgs: string[], extensionFlags?: Map<string, { ty
result.autoApprove = true;
} else if (arg.startsWith("@")) {
result.fileArgs.push(arg.slice(1)); // Remove @ prefix
} else if (!arg.startsWith("-") || arg === "-" || arg === "--") {
// Plain positional, lone `-` (stdin marker), or POSIX positional
// separator `--` — pass through as a message rather than flagging it.
if (arg !== "--") result.messages.push(arg);
} else if (!arg.startsWith("-") || arg === "-") {
// Plain positional or lone `-` (stdin marker) — pass through as a
// message rather than flagging it.
result.messages.push(arg);
} else if (arg === "--") {
// POSIX positional separator: drop the token and switch the loop
// into "everything from here is a positional" mode. The guard at
// the top of the loop body handles the remaining tokens.
sawSeparator = true;
} else {
// Flag-shaped (`-x`, `--name`) but unrecognized at this parse. Record
// it so the post-extension reparse can decide whether to surface it
@@ -2,7 +2,8 @@ import { ProcessTerminal, TUI } from "@oh-my-pi/pi-tui";
import { logger } from "@oh-my-pi/pi-utils";
import { SessionSelectorComponent } from "../modes/components/session-selector";
import { HistoryStorage } from "../session/history-storage";
import { type SessionInfo, SessionManager } from "../session/session-manager";
import type { SessionInfo } from "../session/session-listing";
import { SessionManager } from "../session/session-manager";
import { FileSessionStorage } from "../session/session-storage";
/**
+149 -48
View File
@@ -4,12 +4,18 @@
* Handles `omp setup` for onboarding and `omp setup <component>` for optional dependencies.
*/
import * as path from "node:path";
import { $which, APP_NAME, getPythonEnvDir } from "@oh-my-pi/pi-utils";
import { $which, APP_NAME, getProjectDir, getPythonEnvDir } from "@oh-my-pi/pi-utils";
import { $ } from "bun";
import chalk from "chalk";
import { Settings, settings } from "../config/settings";
import { theme } from "../modes/theme/theme";
import { downloadSttModel, isSttModelCached } from "../stt/downloader";
import { isSttModelKey, STT_MODEL_OPTIONS } from "../stt/models";
import { detectRecorder, ensureRecorder } from "../stt/recorder";
import { downloadTtsModel, isTtsLocalModelKey, isTtsModelCached, TTS_LOCAL_MODEL_OPTIONS } from "../tts";
import { selectSetupModel } from "./setup-model-picker";
export type SetupComponent = "python" | "stt";
export type SetupComponent = "python" | "speech";
export interface SetupCommandArgs {
component: SetupComponent;
@@ -19,7 +25,7 @@ export interface SetupCommandArgs {
};
}
const VALID_COMPONENTS: SetupComponent[] = ["python", "stt"];
const VALID_COMPONENTS: SetupComponent[] = ["python", "speech"];
const MANAGED_PYTHON_ENV = getPythonEnvDir();
@@ -114,8 +120,8 @@ export async function runSetupCommand(cmd: SetupCommandArgs): Promise<void> {
case "python":
await handlePythonSetup(cmd.flags);
break;
case "stt":
await handleSttSetup(cmd.flags);
case "speech":
await handleSpeechSetup(cmd.flags);
break;
}
}
@@ -149,58 +155,153 @@ async function handlePythonSetup(flags: { json?: boolean; check?: boolean }): Pr
process.exit(1);
}
async function handleSttSetup(flags: { json?: boolean; check?: boolean }): Promise<void> {
const { checkDependencies, formatDependencyStatus } = await import("../stt/setup");
const status = await checkDependencies();
/**
* One installable speech dependency. `isReady`/`status` are read-only probes;
* `pick` (optional) lets an interactive user choose + persist a model; `ensure`
* performs the download, streaming a normalized progress event.
*/
interface SpeechComponent {
name: string;
isReady(): Promise<boolean>;
status(): Promise<string>;
pick?(): Promise<boolean>;
ensure(onProgress: (progress: { stage: string; percent?: number }) => void): Promise<void>;
}
function buildSpeechComponents(): SpeechComponent[] {
return [
{
name: "Recorder",
isReady: async () => detectRecorder() !== null,
status: async () => {
const recorder = detectRecorder();
return recorder ? `${recorder.tool} (${recorder.bin})` : "none — ffmpeg will be downloaded";
},
ensure: async onProgress => {
await ensureRecorder(onProgress);
},
},
{
name: "Speech-to-Text model",
isReady: () => isSttModelCached(settings.get("stt.modelName")),
status: async () => {
const key = settings.get("stt.modelName");
return (await isSttModelCached(key)) ? key : `${key} — not downloaded`;
},
pick: async () => {
const chosen = await selectSetupModel(
"Speech-to-Text model",
[...STT_MODEL_OPTIONS],
settings.get("stt.modelName"),
);
if (chosen === null) return false;
if (isSttModelKey(chosen)) {
settings.set("stt.modelName", chosen);
await settings.flush();
}
return true;
},
ensure: onProgress =>
downloadSttModel(settings.get("stt.modelName"), progress =>
onProgress({ stage: `Downloading ${progress.label} model`, percent: progress.percent }),
),
},
{
name: "Text-to-Speech model",
isReady: () => isTtsModelCached(settings.get("tts.localModel")),
status: async () => {
const key = settings.get("tts.localModel");
return (await isTtsModelCached(key)) ? key : `${key} — model/runtime not installed`;
},
pick: async () => {
const chosen = await selectSetupModel(
"Text-to-Speech model",
[...TTS_LOCAL_MODEL_OPTIONS],
settings.get("tts.localModel"),
);
if (chosen === null) return false;
if (isTtsLocalModelKey(chosen)) {
settings.set("tts.localModel", chosen);
await settings.flush();
}
return true;
},
ensure: async onProgress => {
const ok = await downloadTtsModel(settings.get("tts.localModel"), progress =>
onProgress({ stage: progress.stage, percent: progress.percent }),
);
if (!ok) throw new Error("Failed to download the local text-to-speech model.");
},
},
];
}
/**
* Unified `omp setup speech` flow. Drives every {@link SpeechComponent} through
* one path: report (`--json`/`--check`) or install (interactive pick + ensure
* with single-line progress; non-TTY skips pickers and installs configured
* values).
*/
async function handleSpeechSetup(flags: { json?: boolean; check?: boolean }): Promise<void> {
await Settings.init({ cwd: getProjectDir() });
const components = buildSpeechComponents();
if (flags.json) {
console.log(JSON.stringify(status, null, 2));
if (!status.recorder.available || !status.python.available || !status.whisper.available) process.exit(1);
return;
}
console.log(formatDependencyStatus(status));
if (status.recorder.available && status.python.available && status.whisper.available) {
console.log(chalk.green(`\n${theme.status.success} Speech-to-text is ready`));
const report: Record<string, { ready: boolean; status: string }> = {};
let allReady = true;
for (const component of components) {
const ready = await component.isReady();
if (!ready) allReady = false;
report[component.name] = { ready, status: await component.status() };
}
console.log(JSON.stringify(report, null, 2));
if (!allReady) process.exit(1);
return;
}
if (flags.check) {
process.exit(1);
console.log(chalk.bold("Speech dependencies:"));
let allReady = true;
for (const component of components) {
const ready = await component.isReady();
if (!ready) allReady = false;
const mark = ready ? chalk.green("[ok]") : chalk.yellow("[missing]");
console.log(` ${mark} ${component.name}: ${await component.status()}`);
}
if (!allReady) process.exit(1);
return;
}
if (!status.python.available) {
console.error(chalk.red(`\n${theme.status.error} Python not found`));
console.error(chalk.dim("Install Python 3.8+ and ensure it's in your PATH"));
process.exit(1);
}
if (!status.recorder.available) {
console.error(chalk.yellow(`\n${theme.status.warning} No recording tool found`));
console.error(chalk.dim(status.recorder.installHint));
}
if (!status.whisper.available) {
console.log(chalk.dim(`\nInstalling openai-whisper...`));
const { resolvePython } = await import("../stt/transcriber");
const pythonCmd = resolvePython()!;
const result = await $`${pythonCmd} -m pip install -q openai-whisper`.nothrow();
if (result.exitCode !== 0) {
console.error(chalk.red(`\n${theme.status.error} Failed to install openai-whisper`));
console.error(chalk.dim("Try manually: pip install openai-whisper"));
const interactive = Boolean(process.stdout.isTTY);
for (const component of components) {
if (interactive && component.pick) {
await component.pick();
}
if (await component.isReady()) {
console.log(chalk.green(`${theme.status.success} ${component.name} ready`));
continue;
}
console.log(chalk.dim(`Preparing ${component.name}...`));
try {
await component.ensure(progress => {
const percent = typeof progress.percent === "number" ? ` (${progress.percent}%)` : "";
process.stdout.write(`\r${chalk.dim(`${progress.stage}${percent}`)}\x1b[K`);
});
process.stdout.write("\n");
} catch (err) {
process.stdout.write("\n");
const msg = err instanceof Error ? err.message : `Failed to set up ${component.name}`;
console.error(chalk.red(`${theme.status.error} ${msg}`));
process.exit(1);
}
}
const recheck = await checkDependencies();
if (recheck.recorder.available && recheck.python.available && recheck.whisper.available) {
console.log(chalk.green(`\n${theme.status.success} Speech-to-text is ready`));
} else {
console.error(chalk.red(`\n${theme.status.error} Setup incomplete`));
console.log(formatDependencyStatus(recheck));
process.exit(1);
}
console.log(chalk.green(`\n${theme.status.success} Speech is ready`));
console.log(
chalk.dim(
"Enable speech-to-text via stt.enabled, then hold Space to talk (or bind app.stt.toggle); enable the speech-generation tool via speechgen.enabled; speak replies aloud via speech.enabled.",
),
);
}
/**
@@ -215,7 +316,7 @@ ${chalk.bold("Usage:")}
${chalk.bold("Components:")}
python Verify a Python 3 interpreter is reachable for code execution
stt Install speech-to-text dependencies (openai-whisper, recording tools)
speech Pick + download the speech-to-text and text-to-speech models and an audio recorder
${chalk.bold("Options:")}
-c, --check Check if dependencies are installed without installing
@@ -224,8 +325,8 @@ ${chalk.bold("Options:")}
${chalk.bold("Examples:")}
${APP_NAME} setup Run the onboarding wizard
${APP_NAME} setup python Check Python execution dependencies
${APP_NAME} setup stt Install speech-to-text dependencies
${APP_NAME} setup stt --check Check if STT dependencies are available
${APP_NAME} setup speech Set up speech (pick STT + TTS models, install a recorder)
${APP_NAME} setup speech --check Check if speech dependencies are available
${APP_NAME} setup python --check Check if Python execution is available
`);
}
@@ -0,0 +1,43 @@
/**
* Standalone TUI model picker used by `omp setup speech`.
*
* Mirrors {@link ./session-picker.ts} for the standalone-TUI lifecycle: spin up
* a one-shot {@link TUI} over a {@link SelectList}, resolve on select/cancel, and
* tear the UI down. The standalone TUI auto-renders on input, so no manual
* render wiring is needed beyond `addChild`/`setFocus`/`start`.
*/
import { ProcessTerminal, type SelectItem, SelectList, TUI } from "@oh-my-pi/pi-tui";
import { getSelectListTheme } from "../modes/theme/theme";
/**
* Show a single-column model picker and resolve with the chosen item's value,
* or `null` if the user cancelled. `currentValue` pre-selects the matching row.
*/
export async function selectSetupModel(
title: string,
items: SelectItem[],
currentValue: string,
): Promise<string | null> {
const { promise, resolve } = Promise.withResolvers<string | null>();
const ui = new TUI(new ProcessTerminal());
let resolved = false;
const finish = (value: string | null): void => {
if (resolved) return;
resolved = true;
ui.stop();
resolve(value);
};
const list = new SelectList(items, Math.min(items.length, 10), getSelectListTheme());
const currentIndex = items.findIndex(item => item.value === currentValue);
if (currentIndex >= 0) list.setSelectedIndex(currentIndex);
list.onSelect = item => finish(item.value);
list.onCancel = () => finish(null);
process.stdout.write(`${title}\n`);
ui.addChild(list);
ui.setFocus(list);
ui.start();
return promise;
}
+1 -1
View File
@@ -20,7 +20,7 @@ import { AgentLifecycleManager } from "../registry/agent-lifecycle";
import { AgentRegistry } from "../registry/agent-registry";
import type { AgentSessionEvent } from "../session/agent-session";
import { stripImagesFromMessage, USER_INTERRUPT_LABEL } from "../session/messages";
import type { SessionEntry as StoredSessionEntry } from "../session/session-manager";
import type { SessionEntry as StoredSessionEntry } from "../session/session-entries";
import { TASK_SUBAGENT_LIFECYCLE_CHANNEL, TASK_SUBAGENT_PROGRESS_CHANNEL } from "../task";
import { generateRoomKey, generateWriteToken, importRoomKey } from "./crypto";
import {
+1 -1
View File
@@ -25,7 +25,7 @@ import {
} from "@oh-my-pi/pi-wire";
import type { ContextUsage } from "../extensibility/extensions/types";
import type { AgentSessionEvent } from "../session/agent-session";
import type { SessionEntry, SessionHeader } from "../session/session-manager";
import type { SessionEntry, SessionHeader } from "../session/session-entries";
export type {
CollabPromptDetails,
+102
View File
@@ -0,0 +1,102 @@
/**
* Synthesize text with the local TTS engine and play it (or save it with --out).
*
* Demonstrates the on-device speech stack end to end: the first run downloads
* the configured local model, synthesis happens in the TTS worker subprocess,
* and the resulting WAV is either played through the speakers or written to disk.
*/
import * as os from "node:os";
import * as path from "node:path";
import { getProjectDir, Snowflake } from "@oh-my-pi/pi-utils";
import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
import chalk from "chalk";
import { Settings, settings } from "../config/settings";
import { playAudioFile, removeTempFile } from "../tts/player";
import { shutdownTtsClient, ttsClient } from "../tts/tts-client";
import { encodeWav } from "../tts/wav";
export default class Say extends Command {
static description = "Synthesize text with the local TTS engine and play it through the speakers";
static args = {
text: Args.string({ required: true, description: "Text to speak" }),
};
static flags = {
voice: Flags.string({ description: "Voice id" }),
model: Flags.string({ description: "Local TTS model key" }),
out: Flags.string({ char: "o", description: "Write WAV to this path instead of playing" }),
};
static examples = [
'omp say "hello world"',
'omp say "hello world" --out /tmp/hello.wav',
'omp say "bonjour" --voice af_heart --model kokoro',
];
async run(): Promise<void> {
const { args, flags } = await this.parse(Say);
const text = args.text ?? "";
await Settings.init({ cwd: getProjectDir() });
const model = flags.model ?? settings.get("tts.localModel");
const voice = flags.voice ?? settings.get("tts.localVoice");
let exitCode = 0;
const unsubscribe = ttsClient.onProgress(event => {
if (event.status === "progress" && typeof event.progress === "number") {
process.stderr.write(
`\r${chalk.dim(`downloading ${event.file ?? model}: ${Math.round(event.progress)}%`)}`,
);
} else if (event.status === "done" || event.status === "ready") {
// Clear the progress line once the download finishes.
process.stderr.write("\r\x1b[K");
}
});
try {
const audio = await ttsClient.synthesize(model, text, { voice });
if (!audio) {
process.stderr.write(
chalk.red(
`error: could not synthesize with local TTS model "${model}". ` +
"Run `omp setup speech` to install it.\n",
),
);
exitCode = 1;
return;
}
const wav = encodeWav(audio.pcm, audio.sampleRate);
const durationSec = audio.pcm.length / audio.sampleRate;
if (flags.out) {
await Bun.write(flags.out, wav);
process.stdout.write(
`${chalk.green("saved")} ${flags.out} ` +
`${chalk.dim(`(${voice}, ${model}, ${durationSec.toFixed(1)}s, ${wav.byteLength} bytes)`)}\n`,
);
return;
}
const tmp = path.join(os.tmpdir(), `omp-say-${Snowflake.next()}.wav`);
await Bun.write(tmp, wav);
try {
await playAudioFile(tmp);
process.stdout.write(
`${chalk.green("spoke")} ${chalk.dim(`(${voice}, ${model}, ${durationSec.toFixed(1)}s)`)}\n`,
);
} finally {
await removeTempFile(tmp);
}
} catch (err) {
process.stderr.write(chalk.red(`error: ${err instanceof Error ? err.message : String(err)}\n`));
exitCode = 1;
} finally {
unsubscribe();
await shutdownTtsClient();
}
if (exitCode !== 0) process.exit(exitCode);
}
}
+1 -1
View File
@@ -7,7 +7,7 @@ import { runSetupCommand, type SetupCommandArgs, type SetupComponent } from "../
import { runRootCommand } from "../main";
import { initTheme } from "../modes/theme/theme";
const COMPONENTS: SetupComponent[] = ["python", "stt"];
const COMPONENTS: SetupComponent[] = ["python", "speech"];
export interface OnboardingSetupDependencies {
runRoot?: typeof runRootCommand;
@@ -38,6 +38,9 @@ function buildToolSession(
return {
cwd: options.cwd,
hasUI: false,
// Programmatic fan-out: results feed the commit agent's evidence, not a
// model choosing further spawns, so the specialization nudge is noise here.
suppressSpawnAdvisory: true,
getSessionFile: () => ctx.sessionManager.getSessionFile() ?? null,
getSessionSpawns: () => options.spawns,
settings: options.settings,
@@ -212,8 +212,8 @@ export const KEYBINDINGS = {
description: "Search history",
},
"app.stt.toggle": {
defaultKeys: "alt+h",
description: "Toggle speech-to-text",
defaultKeys: [],
description: "Toggle speech-to-text (default gesture: hold Space)",
},
} as const satisfies KeybindingDefinitions;
@@ -393,12 +393,16 @@ export async function discoverOpenAIModelsList(
const response = apiKey
? await withAuth(apiKey, key => attempt({ ...baseHeaders, Authorization: `Bearer ${key}` }))
: await attempt(baseHeaders);
const payload = (await response.json()) as { data?: Array<{ id: string }> };
const payload = (await response.json()) as {
data?: Array<{ id?: string; max_model_len?: unknown; context_length?: unknown }>;
};
const models = payload.data ?? [];
const discovered: Model<Api>[] = [];
for (const item of models) {
const id = item.id;
if (!id) continue;
const contextWindow =
toPositiveNumberOrUndefined(item.max_model_len) ?? toPositiveNumberOrUndefined(item.context_length) ?? 128000;
discovered.push(
buildModel({
id,
@@ -409,8 +413,8 @@ export async function discoverOpenAIModelsList(
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 128000,
maxTokens: discoveryDefaultMaxTokens(providerConfig.api),
contextWindow,
maxTokens: Math.min(contextWindow, discoveryDefaultMaxTokens(providerConfig.api)),
headers,
compat: {
supportsStore: false,
@@ -463,7 +467,7 @@ export async function discoverProxyModels(
? await withAuth(apiKey, key => attempt({ ...baseHeaders, Authorization: `Bearer ${key}` }))
: await attempt(baseHeaders);
const payload = (await response.json()) as {
data?: Array<{ id?: string; name?: string; supported_endpoint_types?: string[] }>;
data?: Array<{ id?: string; name?: string; supported_endpoint_types?: string[]; context_length?: number }>;
};
const items = payload.data ?? [];
const discovered: Model<Api>[] = [];
@@ -499,7 +503,9 @@ export async function discoverProxyModels(
// upstream bundled catalogs, so keep costs local-unknown even when
// we successfully recover the upstream model identity.
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: reference?.contextWindow ?? 128000,
// Prefer the context_length the API reports for this model; fall
// back to the bundled reference, then a sane default.
contextWindow: toPositiveNumberOrUndefined(item.context_length) ?? reference?.contextWindow ?? 128000,
maxTokens: reference?.maxTokens ?? discoveryDefaultMaxTokens(api),
headers,
// OpenAI-compat fields are no-ops on anthropic models; the
@@ -24,9 +24,11 @@ import {
resolveVariantAlias,
} from "@oh-my-pi/pi-catalog/variant-collapse";
// Sentinel for local-only OAuth token (LM Studio, vLLM) — declared inline to avoid loading
// any provider module at startup. Must match `DEFAULT_LOCAL_TOKEN` in oauth/lm-studio.ts.
// Sentinels for local-only OAuth tokens — declared inline to avoid loading
// provider modules at startup. Must match packages/ai/src/registry/lm-studio.ts
// and packages/ai/src/registry/vllm.ts.
const DEFAULT_LOCAL_TOKEN = "lm-studio-local";
const DEFAULT_VLLM_LOCAL_TOKEN = "vllm-local";
const SPECIAL_MODEL_MANAGER_PROVIDER_IDS: readonly string[] = [
"google-antigravity",
@@ -82,6 +84,10 @@ export function isAuthenticated(apiKey: string | undefined | null): apiKey is st
return Boolean(apiKey) && apiKey !== kNoAuth;
}
function isDiscoveryBearerApiKey(apiKey: string | undefined | null): apiKey is string {
return isAuthenticated(apiKey) && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== DEFAULT_VLLM_LOCAL_TOKEN;
}
/** Provider override config (baseUrl, headers, apiKey, compat, transport) without custom models */
interface ProviderOverride {
baseUrl?: string;
@@ -102,9 +108,19 @@ interface ProviderOverride {
* `token-plan-sgp.xiaomimimo.com` at discovery time)
* 3. Existing bundled baseUrl (the host baked into `models.json`)
*
* `transport` resolution priority:
* 1. `providerOverride.transport` (e.g. `pi-native` for auth-gateway users)
* 2. `existing.transport` (carried over from boot-time override application)
* 3. `model.transport` (rarely set — discovery defaults omit it)
*
* Without (1), the user's override would lose to discovery; without (2)
* preferred over (3), the bundled `api.xiaomimimo.com` would shadow the
* tp- token-plan host and produce 401s on the first stream call.
* Without explicit transport propagation, an openrouter (or any) entry
* marked `transport: pi-native` in models.yml silently reverts to the
* default openai-completions transport after the background catalog
* refresh — so the first `/model` switch after boot hits the raw OpenAI
* chat-completions URL instead of the gateway's `/v1/pi/stream` (#2555).
* See `xiaomi-tp-discovery-merge.test.ts` and the `refresh()` baseUrl-override
* regression in `model-registry.test.ts`.
*/
@@ -118,6 +134,7 @@ export function mergeDiscoveredModel<TApi extends Api>(
...model,
baseUrl: providerOverride?.baseUrl ?? model.baseUrl ?? existing.baseUrl,
headers: existing.headers ? { ...existing.headers, ...model.headers } : model.headers,
transport: providerOverride?.transport ?? existing.transport ?? model.transport,
compat: model.compatConfig,
} as ModelSpec<TApi>);
}
@@ -889,6 +906,18 @@ export class ModelRegistry {
});
}
#resolveStartupModelCacheProviderId(providerId: string): string {
const descriptor = PROVIDER_DESCRIPTORS.find(candidate => candidate.providerId === providerId);
if (!descriptor) {
return providerId;
}
const baseUrl =
this.#runtimeProviderOverrides.get(providerId)?.baseUrl ??
this.#providerOverrides.get(providerId)?.baseUrl ??
this.getProviderBaseUrl(providerId);
return descriptor.createModelManagerOptions({ baseUrl, fetch: this.#fetch }).cacheProviderId ?? providerId;
}
#loadCachedStandardProviderModels(): { models: Model<Api>[]; authoritativeFreshProviders: Set<string> } {
const configuredDiscoveryProviders = new Set(this.#discoverableProviders.map(provider => provider.provider));
const cachedModels: Model<Api>[] = [];
@@ -897,7 +926,8 @@ export class ModelRegistry {
if (configuredDiscoveryProviders.has(providerId)) {
continue;
}
const cache = readModelCache<Api>(providerId, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const cacheProviderId = this.#resolveStartupModelCacheProviderId(providerId);
const cache = readModelCache<Api>(cacheProviderId, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
if (!cache) {
continue;
}
@@ -927,7 +957,12 @@ export class ModelRegistry {
#loadCachedDiscoverableModels(): Model<Api>[] {
const cachedModels: Model<Api>[] = [];
for (const providerConfig of this.#discoverableProviders) {
const cache = readModelCache<Api>(providerConfig.provider, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const cache = readModelCache<Api>(
this.#configuredDiscoveryCacheProviderId(providerConfig),
24 * 60 * 60 * 1000,
Date.now,
this.#cacheDbPath,
);
if (!cache) {
this.#providerDiscoveryStates.set(providerConfig.provider, {
provider: providerConfig.provider,
@@ -1189,11 +1224,19 @@ export class ModelRegistry {
this.#rebuildCanonicalIndex();
}
#configuredDiscoveryCacheProviderId(providerConfig: DiscoveryProviderConfig): string {
if (providerConfig.discovery.type === "openai-models-list") {
return `${providerConfig.provider}:openai-models-list-context-v2`;
}
return providerConfig.provider;
}
async #discoverProviderModels(
providerConfig: DiscoveryProviderConfig,
strategy: ModelRefreshStrategy,
): Promise<Model<Api>[]> {
const cached = readModelCache<Api>(providerConfig.provider, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const cacheProviderId = this.#configuredDiscoveryCacheProviderId(providerConfig);
const cached = readModelCache<Api>(cacheProviderId, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const requiresAuth = !this.#keylessProviders.has(providerConfig.provider);
if (requiresAuth) {
const apiKey = await this.#peekApiKeyForProvider(providerConfig.provider);
@@ -1231,6 +1274,7 @@ export class ModelRegistry {
providerId,
staticModels: [],
cacheDbPath: this.#cacheDbPath,
cacheProviderId,
cacheTtlMs: 24 * 60 * 60 * 1000,
fetchDynamicModels,
});
@@ -1272,7 +1316,9 @@ export class ModelRegistry {
fetch: this.#fetch,
getBearerApiKeyResolver: async provider => {
const apiKey = await this.getApiKeyForProvider(provider);
if (!apiKey || apiKey === DEFAULT_LOCAL_TOKEN || apiKey === kNoAuth) return undefined;
if (!isDiscoveryBearerApiKey(apiKey)) {
return undefined;
}
return this.resolver(provider);
},
};
@@ -1377,11 +1423,20 @@ export class ModelRegistry {
for (let i = 0; i < standardProviderDescriptors.length; i++) {
const descriptor = standardProviderDescriptors[i];
const apiKey = standardProviderKeys[i];
if (isAuthenticated(apiKey) || descriptor.allowUnauthenticated) {
const hasExplicitVllmConfig =
descriptor.providerId === "vllm" &&
(this.#runtimeProviderOverrides.has(descriptor.providerId) ||
this.#providerOverrides.has(descriptor.providerId) ||
this.#keylessProviders.has(descriptor.providerId));
if (isAuthenticated(apiKey) || descriptor.allowUnauthenticated || hasExplicitVllmConfig) {
const discoveryBaseUrl =
this.#runtimeProviderOverrides.get(descriptor.providerId)?.baseUrl ??
this.#providerOverrides.get(descriptor.providerId)?.baseUrl ??
this.getProviderBaseUrl(descriptor.providerId);
options.push(
descriptor.createModelManagerOptions({
apiKey: isAuthenticated(apiKey) ? apiKey : undefined,
baseUrl: this.getProviderBaseUrl(descriptor.providerId),
apiKey: isDiscoveryBearerApiKey(apiKey) ? apiKey : undefined,
baseUrl: discoveryBaseUrl,
fetch: this.#fetch,
}),
);

Some files were not shown because too many files have changed in this diff Show More