Merge remote-tracking branch 'origin/main' into feat/marketplace-omp-plugin-path

# Conflicts:
#	docs/marketplace.md
This commit is contained in:
David Marshall
2026-06-03 12:50:46 -05:00
974 changed files with 103880 additions and 12672 deletions
+44 -7
View File
@@ -12,7 +12,7 @@ inputs:
description: Target arch (x64, arm64)
required: true
variant:
description: Optional build variant (baseline, modern)
description: Optional build variant (baseline, modern); required for native x64 builds.
required: false
default: ""
target:
@@ -46,17 +46,54 @@ runs:
toolchain_bin="$(dirname "$(rustup which cargo)")"
echo "$toolchain_bin" >> "$GITHUB_PATH"
echo "Prepended $toolchain_bin to PATH"
# `Swatinem/rust-cache` keys target/ off its restore-time environment, so
# set RUSTFLAGS before restoring it. If x64 target-cpu is only selected
# inside ci-build-native.ts/build-native.ts, cargo invalidates the restored
# target/ but rust-cache sees an exact key and refuses to save the rebuilt
# artifacts, causing macOS x64 baseline to rebuild forever.
#
# Include the native source hash in the shared key as well: rust-cache's
# lockfile scan misses the workspace root Cargo.toml version that Cargo
# fingerprints for workspace crates. Without it, release version bumps can
# get an exact hit for artifacts Cargo must rebuild.
#
# sccache is still layered on top of rust-cache: target/ wins when warm,
# sccache fills the gaps when target/ is cold.
- name: Configure native Rust flags
if: inputs.target == ''
shell: bash
env:
TARGET_ARCH: ${{ inputs.arch }}
TARGET_VARIANT: ${{ inputs.variant }}
run: |
case "$TARGET_ARCH:$TARGET_VARIANT" in
x64:modern)
rustflags="-C target-cpu=x86-64-v3"
;;
x64:baseline)
rustflags="-C target-cpu=x86-64-v2"
;;
x64:*)
echo "::error::x64 native builds require variant=modern or variant=baseline"
exit 1
;;
*)
if [ -n "${RUSTFLAGS:-}" ]; then
echo "Using caller-provided RUSTFLAGS=$RUSTFLAGS"
exit 0
fi
rustflags="-C target-cpu=native"
;;
esac
echo "RUSTFLAGS=$rustflags" >> "$GITHUB_ENV"
echo "Configured RUSTFLAGS=$rustflags"
- uses: Swatinem/rust-cache@v2
with:
shared-key: native-${{ inputs.platform }}-${{ inputs.arch }}-${{ inputs.variant || 'default' }}
shared-key: native-${{ inputs.platform }}-${{ inputs.arch }}-${{ inputs.variant || 'default' }}-h${{ inputs.hash }}
cache-on-failure: true
save-if: ${{ inputs.save_cache == 'true' }}
cache-workspace-crates: true
# `Swatinem/rust-cache` keys target/ off Cargo.lock content; release
# tag pushes bump workspace versions, busting that key every time. sccache
# caches at the rustc-invocation level (source + flags), so it still hits
# across version bumps. Layered on top of rust-cache: target/ wins when
# warm, sccache fills the gaps when target/ is cold.
- name: Setup sccache
uses: mozilla-actions/sccache-action@v0.0.10
- name: Enable sccache for cargo
+102 -34
View File
@@ -3,7 +3,6 @@ name: CI
on:
push:
branches: [main]
tags: ["v*"]
pull_request:
branches: [main]
workflow_dispatch:
@@ -18,6 +17,54 @@ concurrency:
cancel-in-progress: true
jobs:
# scripts/release.ts pushes the version-bump commit and its `v*` tag
# atomically (`git push --atomic origin main refs/tags/v*`), so a release
# now arrives as a single `push` to `refs/heads/main` — we no longer trigger
# on the tag ref at all (see `on.push`). This one branch-push run is therefore
# authoritative: it runs the full build AND, when HEAD carries a release tag,
# the release/publish jobs. `gate` resolves that tag once so downstream jobs
# switch on `is-release` and address the tag by name — `github.ref` is
# `refs/heads/main` here, not the tag. A `workflow_dispatch` from a `v*` tag
# ref is also treated as a release (the manual re-publish escape hatch).
gate:
runs-on: ubuntu-22.04
outputs:
is-release: ${{ steps.check.outputs.is-release }}
release-tag: ${{ steps.check.outputs.release-tag }}
steps:
# Only a main-branch push needs tags fetched, so `git tag --points-at
# HEAD` can see the freshly-pushed `v*`. A tag-ref dispatch reads the
# tag straight from `github.ref_name`, and fetching `--tags` while
# checkout uses an explicit tag refspec makes git refuse — so scope
# fetch-tags to main pushes.
- uses: actions/checkout@v4
with:
fetch-tags: ${{ github.ref == 'refs/heads/main' }}
- name: Detect release tag at HEAD
id: check
shell: bash
run: |
is_release=false
release_tag=""
case "${{ github.ref }}" in
refs/tags/v[0-9]*)
release_tag="${{ github.ref_name }}"
;;
refs/heads/main)
if [ "${{ github.event_name }}" != "pull_request" ]; then
release_tag=$(git tag --points-at HEAD | grep -E '^v[0-9]' | head -n1 || true)
fi
;;
esac
if [ -n "$release_tag" ]; then
echo "HEAD carries release tag $release_tag; this run builds and publishes the release."
is_release=true
fi
{
echo "is-release=$is_release"
echo "release-tag=$release_tag"
} >> "$GITHUB_OUTPUT"
# Compute a stable hash of every input that affects the native cdylib output,
# then look for any prior successful main run that already uploaded the
# native artifacts for this hash. Two independent outputs:
@@ -134,10 +181,10 @@ jobs:
run: bun run ci:check:full
# Linux x64 baseline + modern: required by `test`, so it runs on every PR
# unless rust-hash found a cached run. Tags always rebuild for fresh artifacts.
# unless rust-hash found a cached run. Release pushes always rebuild for fresh artifacts.
native_linux:
needs: [rust-hash]
if: ${{ startsWith(github.ref, 'refs/tags/v') || needs.rust-hash.outputs.linux-run-id == '' }}
needs: [gate, rust-hash]
if: ${{ needs.gate.outputs.is-release == 'true' || needs.rust-hash.outputs.linux-run-id == '' }}
runs-on: ubuntu-22.04
strategy:
fail-fast: false
@@ -154,14 +201,14 @@ jobs:
arch: x64
variant: ${{ matrix.variant }}
rust_checks: ${{ matrix.rust_checks && 'true' || 'false' }}
save_cache: ${{ github.event_name == 'push' && (github.ref == 'refs/heads/main' || startsWith(github.ref, 'refs/tags/v')) }}
save_cache: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }}
# Pre-warm the cross-platform native build cache on `main`, in addition to
# building the artifacts that ship in release tags. Skipped on main when the
# rust-hash canary already found a recent run with all artifacts intact.
native_release:
needs: [rust-hash]
if: ${{ startsWith(github.ref, 'refs/tags/v') || (github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.rust-hash.outputs.release-run-id == '') }}
needs: [gate, rust-hash]
if: ${{ needs.gate.outputs.is-release == 'true' || (github.event_name == 'push' && github.ref == 'refs/heads/main' && needs.rust-hash.outputs.release-run-id == '') }}
strategy:
fail-fast: false
matrix:
@@ -180,7 +227,7 @@ jobs:
arch: ${{ matrix.arch }}
variant: ${{ matrix.variant }}
target: ${{ matrix.target }}
save_cache: ${{ github.event_name == 'push' && (github.ref == 'refs/heads/main' || startsWith(github.ref, 'refs/tags/v')) }}
save_cache: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }}
test:
runs-on: ubuntu-22.04
@@ -222,13 +269,10 @@ jobs:
run-id: ${{ steps.source.outputs.run-id }}
github-token: ${{ secrets.GITHUB_TOKEN }}
- name: Test workspace (TS)
env:
# Bun's `bun test` emits `::group::`/`::endgroup::` per file under
# GHA. `--workspaces` prefixes each output line with `<pkg> test: `,
# which breaks GHA's column-0 parsing and leaks the markers as
# literal text. Unset for this step only — the annotations would be
# equally broken by the prefix, so we lose nothing.
GITHUB_ACTIONS: ""
# `test:ts` sets GITHUB_ACTIONS=0 inline so `bun test` skips its
# per-file `::group::`/`::endgroup::` annotations. Under `--workspaces`
# every line is prefixed with `<pkg> test: `, which breaks GHA's
# column-0 parsing and would leak those markers as literal log spam.
run: bun run test:ts
- name: CLI smoke test
run: bun run ci:test:smoke
@@ -247,8 +291,7 @@ jobs:
with:
shared-key: install-methods-linux-x64
cache-on-failure: true
save-if: ${{ github.event_name == 'push' && (github.ref == 'refs/heads/main' ||
startsWith(github.ref, 'refs/tags/v')) }}
save-if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/main' }}
cache-workspace-crates: true
# Layer sccache on top of rust-cache for the same reason as the
# build-native action: tag pushes bump workspace versions and bust
@@ -279,11 +322,11 @@ jobs:
run: bun run ci:test:install-methods
release_binary:
if: ${{ startsWith(github.ref, 'refs/tags/v') && !cancelled() &&
if: ${{ needs.gate.outputs.is-release == 'true' && !cancelled() &&
needs.native_linux.result == 'success' && needs.native_release.result ==
'success' && needs.test.result == 'success' && needs.check.result ==
'success' && needs.install_methods.result == 'success' }}
needs: [check, native_linux, native_release, test, install_methods, rust-hash]
needs: [gate, check, native_linux, native_release, test, install_methods, rust-hash]
strategy:
fail-fast: false
matrix:
@@ -326,11 +369,20 @@ jobs:
runs-on: ${{ matrix.os }}
permissions:
contents: read
id-token: write
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
with:
bun-version: "1.3"
- uses: actions/setup-node@v4
with:
node-version: "24"
registry-url: "https://registry.npmjs.org"
# Trusted publishing allowed-actions flags require npm >= 11.16.0.
- name: Ensure npm supports trusted publishing
if: ${{ !inputs.skip_npm }}
run: npm install -g npm@latest
- name: Cache bun dependencies
uses: actions/cache@v4
with:
@@ -357,6 +409,14 @@ jobs:
runtime_dir="$(mktemp -d)"
HOME="$runtime_dir/home" XDG_DATA_HOME="$runtime_dir/xdg" "${{ matrix.binary_path }}" --version
HOME="$runtime_dir/home" XDG_DATA_HOME="$runtime_dir/xdg" "${{ matrix.binary_path }}" --smoke-test
- name: Publish native addon package
if: ${{ !inputs.skip_npm }}
env:
# Fallback auth: setup-node wrote an .npmrc referencing
# NODE_AUTH_TOKEN; npm uses it only when OIDC has no trusted
# publisher for the package (or on a first publish).
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
run: bun run ci:release:publish-native-leaf ${{ matrix.target_id }}
- name: Upload release binary artifact
uses: actions/upload-artifact@v4
with:
@@ -364,9 +424,9 @@ jobs:
path: ${{ matrix.binary_path }}
release-github:
if: ${{ startsWith(github.ref, 'refs/tags/v') && !cancelled() &&
if: ${{ needs.gate.outputs.is-release == 'true' && !cancelled() &&
needs.release_binary.result == 'success' }}
needs: [release_binary]
needs: [gate, release_binary]
runs-on: ubuntu-22.04
permissions:
contents: write
@@ -376,7 +436,7 @@ jobs:
with:
bun-version: "1.3"
- name: Generate release notes from CHANGELOGs
run: bun scripts/ci-release-notes.ts
run: bun scripts/ci-release-notes.ts ${{ needs.gate.outputs.release-tag }}
- name: Download release binaries
uses: actions/download-artifact@v4
with:
@@ -386,6 +446,7 @@ jobs:
- name: Create GitHub Release
uses: softprops/action-gh-release@v2
with:
tag_name: ${{ needs.gate.outputs.release-tag }}
files: |
packages/coding-agent/binaries/omp-*
body_path: release-notes.md
@@ -393,16 +454,16 @@ jobs:
release_github_verify:
if: ${{ startsWith(github.ref, 'refs/tags/v') && !cancelled() &&
if: ${{ needs.gate.outputs.is-release == 'true' && !cancelled() &&
needs['release-github'].result == 'success' }}
needs: [release-github]
needs: [gate, release-github]
runs-on: macos-14
permissions:
contents: read
steps:
- name: Download published macOS arm64 binary
run: |
curl -fsSL -o omp-darwin-arm64 "https://github.com/${{ github.repository }}/releases/download/${{ github.ref_name }}/omp-darwin-arm64"
curl -fsSL -o omp-darwin-arm64 "https://github.com/${{ github.repository }}/releases/download/${{ needs.gate.outputs.release-tag }}/omp-darwin-arm64"
chmod +x omp-darwin-arm64
- name: Verify published macOS arm64 binary
run: |
@@ -411,12 +472,19 @@ jobs:
HOME="$runtime_dir/home" XDG_DATA_HOME="$runtime_dir/xdg" ./omp-darwin-arm64 --version
release-npm:
if: ${{ startsWith(github.ref, 'refs/tags/v') && !cancelled() &&
if: ${{ needs.gate.outputs.is-release == 'true' && !cancelled() &&
needs.release_binary.result == 'success' &&
needs.release_github_verify.result == 'success' &&
!inputs.skip_npm }}
needs: [release_binary, release_github_verify, rust-hash]
needs: [gate, release_binary, release_github_verify]
runs-on: ubuntu-22.04
# `id-token: write` lets npm mint the GitHub OIDC token it exchanges for a
# short-lived publish token (trusted publishing + provenance). When a
# package has no matching trusted publisher configured, npm silently falls
# back to NODE_AUTH_TOKEN below — which also covers first-ever publishes.
permissions:
id-token: write
contents: read
steps:
- uses: actions/checkout@v4
- uses: oven-sh/setup-bun@v2
@@ -426,19 +494,19 @@ jobs:
with:
node-version: "24"
registry-url: "https://registry.npmjs.org"
# Trusted publishing (OIDC) and auto-provenance need npm >= 11.5.1.
- name: Ensure npm supports OIDC trusted publishing
run: npm install -g npm@latest
- name: Cache bun dependencies
uses: actions/cache@v4
with:
path: ~/.bun/install/cache
key: bun-${{ runner.os }}-${{ hashFiles('**/bun.lock') }}
- run: bun install --frozen-lockfile
- name: Download native addons
uses: actions/download-artifact@v4
with:
pattern: pi-natives-*-h${{ needs.rust-hash.outputs.hash }}
path: packages/natives/native
merge-multiple: true
- name: Publish to npm
env:
NPM_CONFIG_TOKEN: ${{ secrets.NPM_TOKEN }}
# Fallback auth: setup-node wrote an .npmrc referencing
# NODE_AUTH_TOKEN; npm uses it only when OIDC has no trusted
# publisher for the package (or on a first publish).
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
run: bun run ci:release:publish
+1
View File
@@ -56,6 +56,7 @@ pi-*.html
# Generated files
packages/coding-agent/src/internal-urls/docs-index.generated.ts
packages/natives/npm/
/runs/
python/omp-rpc/src/omp_rpc.egg-info/
# parallel-agent worktrees
Generated
+49 -49
View File
@@ -238,9 +238,9 @@ checksum = "bef38d45163c2f1dde094a7dfd33ccf595c92905c8f8f4fdc18d06fb1037718a"
[[package]]
name = "bitflags"
version = "2.11.1"
version = "2.12.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c4512299f36f043ab09a583e57bceb5a5aab7a73db1805848e8fef3c9e8c78b3"
checksum = "84d7ced0ae9557296835c32bf1b1e02b44c746701f898460fb000d7eaa84f00a"
[[package]]
name = "bitvec"
@@ -501,9 +501,9 @@ checksum = "ade8366b8bd5ba243f0a58f036cc0ca8a2f069cff1a2351ef1cac6b083e16fc0"
[[package]]
name = "cc"
version = "1.2.62"
version = "1.2.63"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a1dce859f0832a7d088c4f1119888ab94ef4b5d6795d1ce05afb7fe159d79f98"
checksum = "556e016178bb5662a08681bbe0f00f8e17631781a4dfc8c45e466e4b185ec27f"
dependencies = [
"find-msvc-tools",
"shlex",
@@ -769,9 +769,9 @@ dependencies = [
[[package]]
name = "ctor"
version = "1.0.6"
version = "1.0.7"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "6d765eb1c0bda10d31e0ea185f5ee15da532d60b0912d2bd1441783439e749c5"
checksum = "01334b89b69ff726750c5ce5073fc8bd860e99aa9a8fc5ca11b04730e3aee97a"
[[package]]
name = "darling"
@@ -872,7 +872,7 @@ version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e0e367e4e7da84520dedcac1901e4da967309406d1e51017ae1abfb97adbd38"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"objc2",
]
@@ -1748,9 +1748,9 @@ dependencies = [
[[package]]
name = "log"
version = "0.4.30"
version = "0.4.31"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "616ec5685824bcc94416c6d4a7a446eea774a31efd7062c8480ba6fd06d7a6e5"
checksum = "113b30b4cd05f7c06868fdb2854f66a7b9fece9a48425351cd532e810d74024f"
[[package]]
name = "lru"
@@ -1805,9 +1805,9 @@ dependencies = [
[[package]]
name = "mio"
version = "1.2.0"
version = "1.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "50b7e5b27aa02a74bac8c3f23f448f8d87ff11f92d3aac1a6ed369ee08cc56c1"
checksum = "02bd0af71c67b473010cbbc60715ee815645a4dc942899111f494b4b737d6fda"
dependencies = [
"libc",
"wasi 0.11.1+wasi-snapshot-preview1",
@@ -1830,7 +1830,7 @@ version = "3.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f1d395473824516f38dd1071a1a37bc57daa7be65b293ebba4ead5f7abb017a2"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"ctor",
"futures",
"napi-build",
@@ -1894,7 +1894,7 @@ version = "0.28.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ab2156c4fce2f8df6c499cc1c763e4394b7482525bf2a9701c9d79d215f519e4"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"cfg-if",
"cfg_aliases 0.1.1",
"libc",
@@ -1906,7 +1906,7 @@ version = "0.31.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "cf20d2fde8ff38632c426f1165ed7436270b44f199fc55284c38276f9db47c3d"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"cfg-if",
"cfg_aliases 0.2.1",
"libc",
@@ -1997,7 +1997,7 @@ version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d49e936b501e5c5bf01fda3a9452ff86dc3ea98ad5f283e1455153142d97518c"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"objc2",
"objc2-core-graphics",
"objc2-foundation",
@@ -2009,7 +2009,7 @@ version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2a180dd8642fa45cdb7dd721cd4c11b1cadd4929ce112ebd8b9f5803cc79d536"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"dispatch2",
"objc2",
]
@@ -2020,7 +2020,7 @@ version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e022c9d066895efa1345f8e33e584b9f958da2fd4cd116792e15e07e4720a807"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"dispatch2",
"objc2",
"objc2-core-foundation",
@@ -2039,7 +2039,7 @@ version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e3e0adef53c21f888deb4fa59fc59f7eb17404926ee8a6f59f5df0fd7f9f3272"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"objc2",
"objc2-core-foundation",
]
@@ -2050,7 +2050,7 @@ version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "180788110936d59bab6bd83b6060ffdfffb3b922ba1396b312ae795e1de9d81d"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"objc2",
"objc2-core-foundation",
]
@@ -2331,7 +2331,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "15.5.10"
version = "15.8.3"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -2399,7 +2399,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "15.5.10"
version = "15.8.3"
dependencies = [
"async-trait",
"libc",
@@ -2411,7 +2411,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "15.5.10"
version = "15.8.3"
dependencies = [
"anyhow",
"arboard",
@@ -2457,7 +2457,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "15.5.10"
version = "15.8.3"
dependencies = [
"anyhow",
"brush-builtins",
@@ -2497,7 +2497,7 @@ version = "0.18.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "60769b8b31b2a9f263dae2776c37b1b28ae246943cf719eb6946a1db05128a61"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"crc32fast",
"fdeflate",
"flate2",
@@ -2567,7 +2567,7 @@ version = "0.18.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "25485360a54d6861439d60facef26de713b1e126bf015ec8f98239467a2b82f7"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"chrono",
"flate2",
"procfs-core",
@@ -2580,7 +2580,7 @@ version = "0.18.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e6401bf7b6af22f78b563665d15a22e9aef27775b79b149a66ca022468a4e405"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"chrono",
"hex",
]
@@ -2735,7 +2735,7 @@ version = "0.5.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ed2bf2547551a7053d6fdfafda3f938979645c44812fbfcda098faae3f1a362d"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
]
[[package]]
@@ -2833,7 +2833,7 @@ version = "1.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"errno",
"libc",
"linux-raw-sys",
@@ -2975,9 +2975,9 @@ checksum = "dc6fe69c597f9c37bfeeeeeb33da3530379845f10be461a66d16d03eca2ded77"
[[package]]
name = "shlex"
version = "1.3.0"
version = "2.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0fda2ff0d084019ba4d7c6f371c95d8fd75ce3524c3cb8fb653a3023f6323e64"
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
[[package]]
name = "signal-hook-registry"
@@ -3033,9 +3033,9 @@ dependencies = [
[[package]]
name = "socket2"
version = "0.6.3"
version = "0.6.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3a766e1110788c36f4fa1c2b71b387a7815aa65f88ce0229841826633d93723e"
checksum = "52d1cfed4120b4d927bf7c0f86d2087a4a7d6027c906d9f9d525a80573b9be51"
dependencies = [
"libc",
"windows-sys 0.61.2",
@@ -3886,9 +3886,9 @@ dependencies = [
[[package]]
name = "tree-sitter-swift"
version = "0.7.2"
version = "0.7.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f3b98fb6bc8e6a6a10023f401aa6a1858115e849dfaf7de57dd8b8ea0f257bd9"
checksum = "fe36052155b9dd69ca82b3b8f1b4ccfb2d867125ac1a4db1dd7331829242668c"
dependencies = [
"cc",
"tree-sitter-language",
@@ -4002,9 +4002,9 @@ dependencies = [
[[package]]
name = "typenum"
version = "1.20.0"
version = "1.20.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "40ce102ab67701b8526c123c1bab5cbe42d7040ccfd0f64af1a385808d2f43de"
checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20"
[[package]]
name = "ucd-trie"
@@ -4038,9 +4038,9 @@ checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
[[package]]
name = "unicode-segmentation"
version = "1.13.2"
version = "1.13.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9629274872b2bfaf8d66f5f15725007f635594914870f65218920345aa11aa8c"
checksum = "c6f5d3c3b1bf09027a88a6bc961fc00497d651009560b5463668dc81b0fa87a8"
[[package]]
name = "unicode-width"
@@ -4130,9 +4130,9 @@ dependencies = [
[[package]]
name = "uuid"
version = "1.23.1"
version = "1.23.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ddd74a9687298c6858e9b88ec8935ec45d22e8fd5e6394fa1bd4e99a87789c76"
checksum = "d258b83ceec21034727ecee8c382cfa6c3e133699b0742c64571814fb420c9f7"
dependencies = [
"js-sys",
"wasm-bindgen",
@@ -4279,7 +4279,7 @@ version = "0.244.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"hashbrown 0.15.5",
"indexmap",
"semver",
@@ -4304,7 +4304,7 @@ version = "0.31.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "645c7c96bb74690c3189b5c9cb4ca1627062bb23693a4fad9d8c3de958260144"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"rustix",
"wayland-backend",
"wayland-scanner",
@@ -4316,7 +4316,7 @@ version = "0.32.12"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "563a85523cade2429938e790815fd7319062103b9f4a2dc806e9b53b95982d8f"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"wayland-backend",
"wayland-client",
"wayland-scanner",
@@ -4328,7 +4328,7 @@ version = "0.3.12"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "eb04e52f7836d7c7976c78ca0250d61e33873c34156a2a1fc9474828ec268234"
dependencies = [
"bitflags 2.11.1",
"bitflags 2.12.1",
"wayland-backend",
"wayland-client",
"wayland-protocols",
@@ -4836,7 +4836,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2"
dependencies = [
"anyhow",
"bitflags 2.11.1",
"bitflags 2.12.1",
"indexmap",
"log",
"serde",
@@ -4947,18 +4947,18 @@ dependencies = [
[[package]]
name = "zerocopy"
version = "0.8.49"
version = "0.8.50"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bce33a6288fa3f072a8c2c7d0f2fdbb90e28298f0135c1f99b96c3db2efcc60b"
checksum = "3b065d4f0e55f82fae73202e189638116a87c55ab6b8e6c2721e13dd9d854ad1"
dependencies = [
"zerocopy-derive",
]
[[package]]
name = "zerocopy-derive"
version = "0.8.49"
version = "0.8.50"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8fd425244944f4ab65ccff928e7323354c5a018c75838362fdce749dfad2ee1e"
checksum = "0b631b19d36a892ab55420c92dbc83ccd79274f25be714855d3074aa71cab639"
dependencies = [
"proc-macro2",
"quote",
+1 -1
View File
@@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"]
resolver = "3"
[workspace.package]
version = "15.5.10"
version = "15.8.3"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
+16 -3
View File
@@ -54,6 +54,21 @@ mise use -g github:can1357/oh-my-pi
macOS · Linux · Windows · bun ≥ 1.3.14
### Shell completions
`omp` generates its own completion scripts for **bash**, **zsh**, and **fish** from the live command/flag metadata, so they never drift from the actual CLI. Subcommands, flags, and enum values complete statically; model names (`--model`, `--smol`, `--slow`, `--plan`) resolve against the bundled model catalog and `--resume` against your on-disk sessions.
```sh
# zsh — add to ~/.zshrc (or write the output into a file on your $fpath)
eval "$(omp completions zsh)"
# bash — add to ~/.bashrc
eval "$(omp completions bash)"
# fish
omp completions fish > ~/.config/fish/completions/omp.fish
```
## Every tool, _benchmaxxed_.
Edits that land on the first attempt. Reads that summarize files instead of dumping their content. Searches that return instantly. Pick any model — omp will get it right.
@@ -198,7 +213,6 @@ Stealth's on by default, so pages see a normal user instead of a headless bot. T
- `bash` — workspace shell, with optional PTY or background-job dispatch.
- `eval` — persistent Python and JavaScript cells with shared prelude and tool re-entry.
- `recipe` — invoke a target from a detected task runner — bun, just, make, cargo.
- `ssh` — one remote command against a configured host.
**Code intelligence**
@@ -233,11 +247,10 @@ Stealth's on by default, so pages see a normal user instead of a headless bot. T
**Misc**
- `calc` — deterministic arithmetic — no model in the loop.
- `resolve` — apply or discard a queued preview action.
- `search_tool_bm25` — BM25 over the hidden tool index; activates top matches mid-session.
Setting-gated, off by default: `github`, `calc`, `inspect_image`, `render_mermaid`, `checkpoint`, `rewind`, `search_tool_bm25`, `retain`, `recall`, `reflect`. Flip them on once, scoped per project.
Setting-gated, off by default: `github`, `inspect_image`, `render_mermaid`, `checkpoint`, `rewind`, `search_tool_bm25`, `retain`, `recall`, `reflect`. Flip them on once, scoped per project.
[Full reference →](https://omp.sh/docs/tools)
+318 -260
View File
File diff suppressed because it is too large Load Diff
+49 -35
View File
@@ -617,37 +617,46 @@ pub(crate) fn execute_external_command(
// Set up process group/session state.
//
// A child we are about to `setsid()` (`DetachSession`) must NOT also be
// handed a `process_group(...)`. For a would-be new-group leader it would
// duplicate the group `setsid` already creates; for a pipeline stage joining
// an established group it is a cross-session `setpgid` that fails with EPERM
// now that the leader (and every prior stage) has moved into its own session.
// In both cases `setsid` alone gives the child its own session and process
// group. See `child_session_action` for the decision rationale.
let command_leads_session = new_pg
&& matches!(session_action, ChildSessionAction::TakeForeground)
&& context.shell.options().external_cmd_leads_session;
if new_pg {
match session_action {
ChildSessionAction::DetachSession => {
// `detach_session()` calls `setsid()`, which creates a fresh session
// and process group; requesting `process_group(0)` as well would
// conflict with that setup.
}
ChildSessionAction::TakeForeground if command_leads_session => {
// Don't set process_group(0) - setsid() in pre_exec will handle it.
cmd.lead_session();
}
ChildSessionAction::TakeForeground | ChildSessionAction::None => {
// Normal case: create new process group in current session.
match session_action {
ChildSessionAction::DetachSession => {
// setsid() creates the fresh session + process group; no process_group().
cmd.detach_session();
}
ChildSessionAction::TakeForeground if command_leads_session => {
// Don't set process_group(0) - setsid() in pre_exec will handle it.
cmd.lead_session();
}
ChildSessionAction::TakeForeground => {
// Foreground a child that is not leading its own session: create/join
// the process group in the current session, then grab the terminal.
if new_pg {
cmd.process_group(0);
} else if let Some(pgid) = process_group_id {
cmd.process_group(pgid);
}
cmd.take_foreground();
}
ChildSessionAction::None => {
// Normal case: create a new process group in the current session, or
// join an established one (later pipeline stages).
if new_pg {
cmd.process_group(0);
} else if let Some(pgid) = process_group_id {
cmd.process_group(pgid);
}
}
} else if let Some(pgid) = process_group_id {
// We need to join an established process group.
cmd.process_group(pgid);
}
// See `child_session_action` for the decision rationale and call-out about
// pipeline groups.
match session_action {
ChildSessionAction::DetachSession => cmd.detach_session(),
ChildSessionAction::TakeForeground if !command_leads_session => cmd.take_foreground(),
ChildSessionAction::TakeForeground | ChildSessionAction::None => {}
}
// When tracing is enabled, report.
@@ -964,18 +973,27 @@ pub enum ChildSessionAction {
/// child inherited the host's controlling tty, and any `/dev/tty` open or
/// `tcsetpgrp` call from the child could SIGTTIN/SIGTTOU and stop the host.
///
/// `detach_session()` is unsafe for any member of a multi-command pipeline:
/// for the first stage it puts the process-group leader in a different session,
/// causing later stages' `setpgid()` to fail with EPERM; for later stages it
/// either fails with EPERM or moves the child into a fresh session, breaking the
/// pipeline's shared process group and job-control signal propagation. Pipeline
/// stages therefore keep their pre-fix behavior (no detach).
/// A child whose stdin is **not** a terminal therefore always detaches, even
/// when it is a stage of a multi-command pipeline. An interactive program in a
/// pipeline (`zsh -i ... | awk`) would otherwise open `/dev/tty`, `tcsetpgrp`
/// itself to the foreground, and leave the host stopped on its next tty read.
/// `setsid()` puts each stage in its own session with no controlling tty, so it
/// cannot reach `/dev/tty` at all. The historical EPERM hazard — a later stage
/// `setpgid()`-joining a leader that already moved to a new session — is avoided
/// in `execute_external_command`, which skips `process_group(...)` entirely for
/// detached children; pipeline stages no longer share one process group, which
/// the embedded host does not rely on (it cancels via the descendant tree, and
/// pipes are session-independent).
///
/// `in_pipeline_group` is no longer consulted: a pipeline stage that legitimately
/// needs the shared tty group has terminal stdin and is handled by the
/// `child_stdin_is_terminal` arm before pipeline membership would ever matter.
///
/// Foregrounding remains gated on `new_pg && child_stdin_is_terminal`.
pub fn child_session_action(
new_pg: bool,
child_stdin_is_terminal: bool,
in_pipeline_group: bool,
_in_pipeline_group: bool,
) -> ChildSessionAction {
if new_pg && child_stdin_is_terminal {
return ChildSessionAction::TakeForeground;
@@ -985,9 +1003,5 @@ pub fn child_session_action(
return ChildSessionAction::None;
}
if in_pipeline_group {
return ChildSessionAction::None;
}
ChildSessionAction::DetachSession
}
+20
View File
@@ -437,6 +437,26 @@ impl Job {
}
}
/// Aborts shell-internal background tasks and drops their join handles.
///
/// External process jobs are intentionally left alone; callers that abort
/// internal tasks are still responsible for signalling any process trees
/// those tasks may have spawned.
pub fn abort_internal_tasks(&mut self) {
let mut aborted = false;
self.tasks.retain_mut(|task| {
if let JobTask::Internal(handle) = task {
handle.abort();
aborted = true;
return false;
}
true
});
if aborted && self.tasks.is_empty() {
self.state = JobState::Done;
}
}
/// Tries to retrieve a "representative" pid for the job.
pub fn representative_pid(&self) -> Option<sys::process::ProcessId> {
for task in &self.tasks {
+223
View File
@@ -0,0 +1,223 @@
//! Resolve the syntactic block that begins on a given source line.
//!
//! Powers the hashline `replace block N:` operator: given a 1-indexed line,
//! parse the source with tree-sitter and return the line span of the outermost
//! named node that *begins* on that line (excluding the whole-file root). Brace
//! languages anchor a construct's block to its opening line, so pointing at the
//! line that opens an `if` / `function` / `struct` resolves to that construct's
//! full span; pointing at a continuation line or a lone closing delimiter
//! resolves to nothing.
use anyhow::{Result, anyhow};
use ast_grep_core::tree_sitter::LanguageExt;
use serde::{Deserialize, Serialize};
use tree_sitter::{Parser, Point};
use crate::summary::{node_content_end_line, node_start_line, resolve_language};
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct BlockRangeOptions {
/// Source code to inspect.
pub code: String,
/// Language alias (e.g. "rust", "typescript") used before path inference.
pub lang: Option<String>,
/// File path used to infer language by extension when `lang` is omitted.
pub path: Option<String>,
/// 1-indexed source line the block must begin on.
pub line: u32,
}
#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
pub struct BlockRange {
/// 1-indexed inclusive first line of the resolved block.
pub start_line: u32,
/// 1-indexed inclusive last line of the resolved block.
pub end_line: u32,
}
/// Count of leading space/tab bytes on `row` (0-indexed), i.e. the byte column
/// of the first content character. Returns `None` when `row` is out of range
/// or the line is blank / whitespace-only — there is no block to resolve there.
fn first_content_column(code: &str, row: usize) -> Option<usize> {
let line = code.split('\n').nth(row)?;
for (col, byte) in line.bytes().enumerate() {
if byte != b' ' && byte != b'\t' {
return Some(col);
}
}
None
}
/// Resolve the block beginning on `options.line`.
///
/// Returns `None` (a soft "no block here", surfaced as a hard error one layer
/// up) when the language is unrecognized, the line is out of range / blank, no
/// node begins on that line, or the resolved subtree contains a syntax error.
pub fn block_range_at(options: BlockRangeOptions) -> Result<Option<BlockRange>> {
let BlockRangeOptions { code, lang, path, line } = options;
if line == 0 || code.is_empty() {
return Ok(None);
}
let Some(language) = resolve_language(lang.as_deref(), path.as_deref()) else {
return Ok(None);
};
let row = (line - 1) as usize;
let Some(col) = first_content_column(&code, row) else {
return Ok(None);
};
let mut parser = Parser::new();
parser
.set_language(&language.get_ts_language())
.map_err(|err| anyhow!("Failed to load tree-sitter language: {err}"))?;
let Some(tree) = parser.parse(&code, None) else {
return Ok(None);
};
let root = tree.root_node();
let point = Point::new(row, col);
let Some(leaf) = root.named_descendant_for_point_range(point, point) else {
return Ok(None);
};
// A leaf whose own start row is earlier than `row` means `point` landed on
// a continuation line or a closing delimiter of a block that opened earlier
// — there is no block *beginning* on line N.
if leaf.start_position().row != row {
return Ok(None);
}
// Climb to the outermost named ancestor that still begins on `row`,
// excluding the whole-file root. Ancestors can only begin on an earlier
// row, so the first parent that starts before `row` stops the climb.
let mut node = leaf;
while let Some(parent) = node.parent() {
if parent.id() == root.id() {
break;
}
if parent.start_position().row != row {
break;
}
node = parent;
}
// Refuse degenerate error-recovery spans: a missing brace can make
// tree-sitter wrap a huge region in an ERROR node. Checking only the
// resolved node's subtree (not the whole file) keeps an unrelated syntax
// error elsewhere from disabling the feature.
if node.has_error() {
return Ok(None);
}
Ok(Some(BlockRange {
start_line: node_start_line(node),
end_line: node_content_end_line(node),
}))
}
#[cfg(test)]
mod tests {
use super::*;
fn resolve(code: &str, path: &str, line: u32) -> Option<BlockRange> {
block_range_at(BlockRangeOptions {
code: code.to_string(),
lang: None,
path: Some(path.to_string()),
line,
})
.expect("block resolution succeeds")
}
const TS_EXAMPLE: &str = "function x() {\n if (y) {\n }\n}\n";
#[test]
fn resolves_inner_if_block() {
assert_eq!(resolve(TS_EXAMPLE, "x.ts", 2), Some(BlockRange { start_line: 2, end_line: 3 }));
}
#[test]
fn resolves_enclosing_function_block() {
assert_eq!(resolve(TS_EXAMPLE, "x.ts", 1), Some(BlockRange { start_line: 1, end_line: 4 }));
}
#[test]
fn lone_closing_brace_resolves_to_nothing() {
// Line 3 is ` }` — the closing delimiter of a block that opened on an
// earlier line, so no block *begins* there.
assert_eq!(resolve(TS_EXAMPLE, "x.ts", 3), None);
}
#[test]
fn blank_line_resolves_to_nothing() {
let code = "function x() {\n\n return 1;\n}\n";
assert_eq!(resolve(code, "x.ts", 2), None);
}
#[test]
fn out_of_range_line_resolves_to_nothing() {
assert_eq!(resolve(TS_EXAMPLE, "x.ts", 99), None);
assert_eq!(resolve(TS_EXAMPLE, "x.ts", 0), None);
}
#[test]
fn unrecognized_extension_resolves_to_nothing() {
assert_eq!(resolve(TS_EXAMPLE, "x.unknownext", 2), None);
}
#[test]
fn resolves_top_level_python_def() {
let code = "x = 1\ndef greet():\n return 1\n";
assert_eq!(resolve(code, "g.py", 2), Some(BlockRange { start_line: 2, end_line: 3 }));
}
#[test]
fn resolves_inner_python_block() {
// Point at the `for` loop inside the function body. The suite's first
// statement is `total = 0` (line 2), so the `for` at line 3 is not the
// suite's first child and climbs only to the `for_statement`, not the
// whole function suite.
let code =
"def f(xs):\n total = 0\n for x in xs:\n total += x\n return total\n";
assert_eq!(resolve(code, "f.py", 3), Some(BlockRange { start_line: 3, end_line: 4 }));
}
#[test]
fn resolves_nested_block_to_outermost_on_line() {
// Point at the inner `if` line; it resolves the whole `if` block
// (header through its closing brace), not just the call inside it.
let code = "function f() {\n if (a) {\n g();\n }\n}\n";
assert_eq!(resolve(code, "f.ts", 2), Some(BlockRange { start_line: 2, end_line: 4 }));
}
#[test]
fn multi_statement_line_resolves_first_statement_node() {
// `let a = 1; let b = 2;` — pointing at the line resolves the first
// statement that begins at the line's first content column.
let code = "let a = 1; let b = 2;\n";
let range = resolve(code, "m.ts", 1);
assert!(range.is_some(), "expected a block on a single-statement-bearing line");
assert_eq!(range.unwrap().start_line, 1);
}
#[test]
fn continuation_line_resolves_to_nothing() {
// A bare argument-continuation line whose first content does not open a
// new named node beginning on that row.
let code = "foo(\n a,\n b,\n);\n";
// Line 2 (` a,`) is an argument — `a` is an identifier beginning on the
// row, so it DOES resolve. Use the closing `);` line instead, which is
// a continuation/closer of the call begun earlier.
assert_eq!(resolve(code, "c.ts", 4), None);
}
#[test]
fn error_subtree_resolves_to_nothing() {
// Missing closing brace: the function's subtree carries an ERROR, so we
// refuse to resolve a degenerate recovery span.
let code = "function broken() {\n if (y) {\n}\n";
assert_eq!(resolve(code, "b.ts", 1), None);
}
#[test]
fn resolves_rust_struct_block() {
let code = "struct A;\nstruct B {\n x: u32,\n}\n";
assert_eq!(resolve(code, "r.rs", 2), Some(BlockRange { start_line: 2, end_line: 4 }));
}
}
+1
View File
@@ -1,3 +1,4 @@
pub mod block;
pub mod language;
pub mod ops;
pub mod summary;
+3 -3
View File
@@ -208,7 +208,7 @@ pub fn summarize_code(options: SummaryOptions) -> Result<SummaryResult> {
})
}
fn resolve_language(lang: Option<&str>, path: Option<&str>) -> Option<SupportLang> {
pub(crate) fn resolve_language(lang: Option<&str>, path: Option<&str>) -> Option<SupportLang> {
if let Some(lang) = lang.map(str::trim).filter(|lang| !lang.is_empty()) {
return SupportLang::from_alias(lang);
}
@@ -354,7 +354,7 @@ fn flush_groupable_run(
}
}
fn node_start_line(node: Node<'_>) -> u32 {
pub(crate) fn node_start_line(node: Node<'_>) -> u32 {
node
.start_position()
.row
@@ -376,7 +376,7 @@ fn node_end_line(node: Node<'_>) -> u32 {
/// When that byte is a newline, the resulting position lands at column 0 of
/// the next row, which makes the naive `row + 1` answer one greater than the
/// row of the last visible content. This helper subtracts that off.
fn node_content_end_line(node: Node<'_>) -> u32 {
pub(crate) fn node_content_end_line(node: Node<'_>) -> u32 {
let pos = node.end_position();
let row = if pos.column == 0 && pos.row > 0 {
pos.row - 1
+47
View File
@@ -0,0 +1,47 @@
//! Resolve the syntactic block beginning on a source line (tree-sitter).
use napi::bindgen_prelude::*;
use napi_derive::napi;
#[napi(object)]
pub struct BlockRangeOptions {
/// Source code to inspect.
pub code: String,
/// Language alias (e.g. "rust", "typescript") used before path inference.
pub lang: Option<String>,
/// File path used to infer language by extension when `lang` is omitted.
pub path: Option<String>,
/// 1-indexed source line the block must begin on.
pub line: u32,
}
#[napi(object)]
pub struct BlockRange {
/// 1-indexed inclusive first line of the resolved block.
pub start_line: u32,
/// 1-indexed inclusive last line of the resolved block.
pub end_line: u32,
}
impl From<pi_ast::block::BlockRange> for BlockRange {
fn from(value: pi_ast::block::BlockRange) -> Self {
Self { start_line: value.start_line, end_line: value.end_line }
}
}
/// Find the outermost named tree-sitter node that begins on `options.line`.
///
/// Returns its 1-indexed inclusive line span, or `null` when the language is
/// unrecognized, the line is out of range / blank, no node begins on that line,
/// or the resolved subtree contains a syntax error.
#[napi]
pub fn block_range_at(options: BlockRangeOptions) -> Result<Option<BlockRange>> {
pi_ast::block::block_range_at(pi_ast::block::BlockRangeOptions {
code: options.code,
lang: options.lang,
path: options.path,
line: options.line,
})
.map(|range| range.map(Into::into))
.map_err(|error| Error::from_reason(error.to_string()))
}
+82 -32
View File
@@ -200,8 +200,6 @@ static LEGACY_SEQUENCES: phf::Map<&'static [u8], &'static str> = phf_map! {
b"\x1b[[A" => "f1", b"\x1b[[B" => "f2", b"\x1b[[C" => "f3", b"\x1b[[D" => "f4", b"\x1b[[E" => "f5",
b"\x1b[15~" => "f5", b"\x1b[17~" => "f6", b"\x1b[18~" => "f7", b"\x1b[19~" => "f8",
b"\x1b[20~" => "f9", b"\x1b[21~" => "f10", b"\x1b[23~" => "f11", b"\x1b[24~" => "f12",
// Alt+arrow (legacy)
b"\x1bb" => "alt+left", b"\x1bf" => "alt+right", b"\x1bp" => "alt+up", b"\x1bn" => "alt+down",
};
/// Pre-allocated single ASCII printable characters (33-126)
@@ -701,9 +699,9 @@ fn matches_key_inner(bytes: &[u8], key_id: &str, kitty_protocol_active: bool) ->
}
if key.eq_ignore_ascii_case("enter") || key.eq_ignore_ascii_case("return") {
// alt+enter is commonly ESC + CR even when kitty disambiguation is on (Enter is
// an exception).
if modifier == MOD_ALT && bytes == b"\x1b\r" {
// alt+enter is commonly ESC + CR/LF even when kitty disambiguation is on
// (Enter is an exception).
if modifier == MOD_ALT && (bytes == b"\x1b\r" || bytes == b"\x1b\n") {
return true;
}
@@ -799,7 +797,7 @@ fn matches_key_inner(bytes: &[u8], key_id: &str, kitty_protocol_active: bool) ->
if key.eq_ignore_ascii_case("up") {
if modifier == MOD_ALT {
return bytes == b"\x1bp" || kitty_matches(ARROW_UP, MOD_ALT);
return kitty_matches(ARROW_UP, MOD_ALT);
}
if modifier == 0 {
return matches_legacy_key(bytes, "up") || kitty_matches(ARROW_UP, 0);
@@ -810,7 +808,7 @@ fn matches_key_inner(bytes: &[u8], key_id: &str, kitty_protocol_active: bool) ->
if key.eq_ignore_ascii_case("down") {
if modifier == MOD_ALT {
return bytes == b"\x1bn" || kitty_matches(ARROW_DOWN, MOD_ALT);
return kitty_matches(ARROW_DOWN, MOD_ALT);
}
if modifier == 0 {
return matches_legacy_key(bytes, "down") || kitty_matches(ARROW_DOWN, 0);
@@ -823,7 +821,6 @@ fn matches_key_inner(bytes: &[u8], key_id: &str, kitty_protocol_active: bool) ->
if modifier == MOD_ALT {
return bytes == b"\x1b[1;3D"
|| (!kitty_protocol_active && bytes == b"\x1bB")
|| bytes == b"\x1bb"
|| kitty_matches(ARROW_LEFT, MOD_ALT);
}
if modifier == MOD_CTRL {
@@ -842,7 +839,6 @@ fn matches_key_inner(bytes: &[u8], key_id: &str, kitty_protocol_active: bool) ->
if modifier == MOD_ALT {
return bytes == b"\x1b[1;3C"
|| (!kitty_protocol_active && bytes == b"\x1bF")
|| bytes == b"\x1bf"
|| kitty_matches(ARROW_RIGHT, MOD_ALT);
}
if modifier == MOD_CTRL {
@@ -883,14 +879,15 @@ fn matches_key_inner(bytes: &[u8], key_id: &str, kitty_protocol_active: bool) ->
let codepoint = ch as i32;
let is_letter = ch.is_ascii_lowercase();
// ctrl+alt+letter in legacy mode
// Legacy: ctrl+alt+letter is ESC followed by the control character.
// If that legacy form does not match, continue so CSI-u and
// Legacy ctrl+alt+letter is ESC followed by the control character.
// tmux extkeys/CSI-u and Kitty mixed modes can still pass these legacy Meta
// pairs through, so accept them even when enhanced keyboard reporting is
// active. If that legacy form does not match, continue so CSI-u and
// modifyOtherKeys sequences from tmux can still be recognized.
// Legacy ESC+ctrl-char would also match Alt+Enter/Alt+Backspace/etc;
// skip the legacy fast-path for those bytes and let kitty/modifyOtherKeys
// disambiguate.
if modifier == (MOD_CTRL | MOD_ALT) && !kitty_protocol_active && is_letter {
if modifier == (MOD_CTRL | MOD_ALT) && is_letter {
let ctrl_char = raw_ctrl_char(ch);
if bytes.len() == 2
&& bytes[0] == 0x1b
@@ -901,14 +898,22 @@ fn matches_key_inner(bytes: &[u8], key_id: &str, kitty_protocol_active: bool) ->
}
}
// alt+letter in legacy mode
if modifier == MOD_ALT && !kitty_protocol_active && is_letter {
return bytes.len() == 2 && bytes[0] == 0x1b && bytes[1] == ch;
// alt+letter can remain ESC+letter inside tmux/Kitty mixed modes. If that
// legacy form does not match, fall through so CSI-u and modifyOtherKeys
// encodings still match.
if modifier == MOD_ALT && is_letter && bytes.len() == 2 && bytes[0] == 0x1b && bytes[1] == ch
{
return true;
}
// alt+shift+letter in legacy mode (ESC + UPPERCASE letter)
if modifier == (MOD_ALT | MOD_SHIFT) && !kitty_protocol_active && is_letter {
return bytes.len() == 2 && bytes[0] == 0x1b && bytes[1] == ch.to_ascii_uppercase();
// alt+shift+letter can remain ESC+UPPERCASE inside tmux/Kitty mixed modes.
if modifier == (MOD_ALT | MOD_SHIFT)
&& is_letter
&& bytes.len() == 2
&& bytes[0] == 0x1b
&& bytes[1] == ch.to_ascii_uppercase()
{
return true;
}
// ctrl+key
@@ -1031,6 +1036,15 @@ fn parse_key_inner(bytes: &[u8], kitty_protocol_active: bool) -> Option<Cow<'sta
return None;
}
// Two-byte ESC sequences are legacy Meta/Alt keypresses. Handle them before
// the legacy table so ESC+p from Ghostty/tmux is parsed as Alt+P rather than
// the historical ESC+p Alt+Up compatibility alias.
if bytes.len() == 2
&& let Some(key) = parse_esc_pair(bytes[1], kitty_protocol_active)
{
return Some(key);
}
// O(1) lookup in perfect hash map for legacy sequences
if let Some(&key_id) = LEGACY_SEQUENCES.get(bytes) {
return Some(Cow::Borrowed(key_id));
@@ -1068,12 +1082,6 @@ fn parse_key_inner(bytes: &[u8], kitty_protocol_active: bool) -> Option<Cow<'sta
return Some(Cow::Owned(format!("alt+{inner_key}")));
}
// Two-byte ESC sequences (legacy ALT prefix, with exceptions even in kitty
// mode)
if bytes.len() == 2 {
return parse_esc_pair(bytes[1], kitty_protocol_active);
}
// Fixed CSI / SS3 sequences not covered by LEGACY_SEQUENCES
match bytes {
b"\x1b[Z" => Some(Cow::Borrowed("shift+tab")),
@@ -1108,26 +1116,29 @@ fn parse_esc_pair(code: u8, kitty_protocol_active: bool) -> Option<Cow<'static,
// terminals.
match code {
0x7f | 0x08 => return Some(Cow::Borrowed("alt+backspace")),
b'\r' => return Some(Cow::Borrowed("alt+enter")),
b'\r' | b'\n' => return Some(Cow::Borrowed("alt+enter")),
b'\t' => return Some(Cow::Borrowed("alt+tab")),
_ => {},
}
// Legacy ALT-prefix parsing only when kitty protocol isn't expected to
// disambiguate.
// Historical cursor-key aliases used by some legacy terminals. Keep them in
// legacy mode only; in mixed modes (tmux extkeys/CSI-u, Kitty, etc.) ESC+B/F
// are real Alt+Shift+B/F keypresses.
if !kitty_protocol_active {
match code {
b' ' => return Some(Cow::Borrowed("alt+space")),
b'B' => return Some(Cow::Borrowed("alt+left")),
b'F' => return Some(Cow::Borrowed("alt+right")),
1..=26 => return Some(Cow::Borrowed(CTRL_ALT_LETTERS[(code - 1) as usize])),
b'a'..=b'z' => return Some(Cow::Borrowed(ALT_LETTERS[(code - b'a') as usize])),
b'A'..=b'Z' => return Some(Cow::Borrowed(ALT_SHIFT_LETTERS[(code - b'A') as usize])),
_ => {},
}
}
None
match code {
1..=26 => Some(Cow::Borrowed(CTRL_ALT_LETTERS[(code - 1) as usize])),
b'a'..=b'z' => Some(Cow::Borrowed(ALT_LETTERS[(code - b'a') as usize])),
b'A'..=b'Z' => Some(Cow::Borrowed(ALT_SHIFT_LETTERS[(code - b'A') as usize])),
_ => None,
}
}
// =============================================================================
@@ -1519,6 +1530,45 @@ mod tests {
assert_eq!(parse_key_inner(b"\x1b\x1b", true).as_deref(), None);
}
#[test]
fn esc_pair_alt_letters_mixed_mode() {
// tmux 3.6 with `extended-keys-format csi-u` can enable enhanced keyboard
// handling while still forwarding Alt+letter as the legacy ESC+letter form.
for active in [false, true] {
assert_eq!(parse_key_inner(b"\x1bp", active).as_deref(), Some("alt+p"));
assert_eq!(parse_key_inner(b"\x1bh", active).as_deref(), Some("alt+h"));
assert_eq!(parse_key_inner(b"\x1bP", active).as_deref(), Some("alt+shift+p"));
assert_eq!(parse_key_inner(b"\x1b\x10", active).as_deref(), Some("ctrl+alt+p"));
assert!(matches_key_inner(b"\x1bp", "alt+p", active));
assert!(matches_key_inner(b"\x1bh", "alt+h", active));
assert!(matches_key_inner(b"\x1bP", "alt+shift+p", active));
assert!(matches_key_inner(b"\x1b\x10", "ctrl+alt+p", active));
assert!(!matches_key_inner(b"\x1bp", "alt+up", active));
assert!(!matches_key_inner(b"\x1bn", "alt+down", active));
assert!(!matches_key_inner(b"\x1bb", "alt+left", active));
assert!(!matches_key_inner(b"\x1bf", "alt+right", active));
}
assert!(matches_key_inner(b"\x1b[1;3A", "alt+up", true));
assert!(matches_key_inner(b"\x1b[112;3u", "alt+p", true));
assert!(matches_key_inner(b"\x1b[27;3;112~", "alt+p", false));
for active in [false, true] {
assert_eq!(parse_key_inner(b"\x1b\n", active).as_deref(), Some("alt+enter"));
assert!(matches_key_inner(b"\x1b\n", "alt+enter", active));
}
}
#[test]
fn uppercase_meta_b_f_stay_legacy_arrow_aliases_only_without_kitty() {
assert_eq!(parse_key_inner(b"\x1bB", false).as_deref(), Some("alt+left"));
assert_eq!(parse_key_inner(b"\x1bF", false).as_deref(), Some("alt+right"));
assert_eq!(parse_key_inner(b"\x1bB", true).as_deref(), Some("alt+shift+b"));
assert_eq!(parse_key_inner(b"\x1bF", true).as_deref(), Some("alt+shift+f"));
assert!(matches_key_inner(b"\x1bB", "alt+left", false));
assert!(matches_key_inner(b"\x1bF", "alt+right", false));
assert!(!matches_key_inner(b"\x1bB", "alt+left", true));
assert!(!matches_key_inner(b"\x1bF", "alt+right", true));
}
#[test]
fn esc_prefix_csi_only() {
// Only CSI and SS3 inner sequences parse as Alt; other double-ESC does not
+2 -1
View File
@@ -23,6 +23,7 @@
pub mod appearance;
pub mod ast;
pub mod block;
pub mod clipboard;
pub mod fd;
pub mod fs_cache;
@@ -67,5 +68,5 @@ use napi_derive::napi;
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV15_5_10")]
#[napi(js_name = "__piNativesV15_8_3")]
pub const fn pi_natives_version_sentinel() {}
+9 -4
View File
@@ -356,9 +356,11 @@ mod tests {
}
#[test]
fn non_terminal_stdin_leading_new_pgroup_detaches_unless_pipeline() {
fn non_terminal_stdin_detaches_regardless_of_pipeline() {
assert_eq!(child_session_action(true, false, false), ChildSessionAction::DetachSession);
assert_eq!(child_session_action(true, false, true), ChildSessionAction::None);
// A leading-new-pgroup stage of a pipeline still detaches: setsid keeps
// it off the host's controlling tty.
assert_eq!(child_session_action(true, false, true), ChildSessionAction::DetachSession);
}
#[test]
@@ -377,8 +379,11 @@ mod tests {
}
#[test]
fn pipeline_stage_does_not_detach() {
assert_eq!(child_session_action(false, false, true), ChildSessionAction::None);
fn pipeline_stage_with_non_terminal_stdin_detaches() {
// Regression: an interactive child inside a pipeline (`zsh -i | awk`)
// must not stay in the host session and seize its tty. Pre-fix this
// returned `None`, leaving the stage attached and able to SIGTTIN the host.
assert_eq!(child_session_action(false, false, true), ChildSessionAction::DetachSession);
}
}
+58 -8
View File
@@ -378,14 +378,34 @@ const fn ascii_cell_width_u16(u: u16, tab_width: usize) -> usize {
}
}
const MACOS_HANGUL_COMPAT_JAMO_WIDTH: usize = 1;
#[inline]
const fn is_macos_hangul_compat_jamo(c: char) -> bool {
let cp = c as u32;
cfg!(target_os = "macos") && cp >= 0x3131 && cp <= 0x318e
}
#[inline]
fn apply_macos_hangul_compat_jamo_delta(width: usize, c: char) -> usize {
if !is_macos_hangul_compat_jamo(c) {
return width;
}
let unicode_width = UnicodeWidthChar::width(c).unwrap_or(0);
if unicode_width > MACOS_HANGUL_COMPAT_JAMO_WIDTH {
width.saturating_sub(unicode_width - MACOS_HANGUL_COMPAT_JAMO_WIDTH)
} else {
width.saturating_add(MACOS_HANGUL_COMPAT_JAMO_WIDTH - unicode_width)
}
}
#[inline]
fn char_width_corrected(c: char) -> Option<usize> {
// Hangul Compatibility Jamo U+3131..=U+318E render as 1 cell on macOS
// terminals (Ghostty, Terminal.app, iTerm2), but follow UAX#11 at 2
// cells on WezTerm and most Linux terminals. Only force 1 on macOS.
let cp = c as u32;
if cfg!(target_os = "macos") && (0x3131..=0x318e).contains(&cp) {
return Some(1);
if is_macos_hangul_compat_jamo(c) {
return Some(MACOS_HANGUL_COMPAT_JAMO_WIDTH);
}
UnicodeWidthChar::width(c)
}
@@ -402,13 +422,18 @@ fn grapheme_width_str(g: &str, tab_width: usize) -> usize {
if it.next().is_none() {
return char_width_corrected(c0).unwrap_or(0);
}
// Multi-char grapheme: keep UnicodeWidthStr as the source of truth for
// sequence-level width rules (VS16 emoji presentation, keycaps, ZWJ emoji,
// CRLF, script ligatures). A per-char sum is not equivalent. On macOS,
// apply only the same local Compatibility Jamo delta that
// char_width_corrected applies to standalone code points.
let mut width = UnicodeWidthStr::width(g);
if cfg!(target_os = "macos") {
g.chars()
.map(|c| char_width_corrected(c).unwrap_or(0))
.sum()
} else {
UnicodeWidthStr::width(g)
for c in g.chars() {
width = apply_macos_hangul_compat_jamo_delta(width, c);
}
}
width
}
thread_local! {
@@ -1311,6 +1336,31 @@ mod tests {
assert_eq!(visible_width_u16(&to_u16("a\tb"), DEFAULT_TAB_WIDTH), 1 + DEFAULT_TAB_WIDTH + 1);
}
#[test]
fn test_visible_width_vs16_emoji_presentation() {
// Variation-selector-16 (U+FE0F) promotes a default-text-presentation
// symbol to emoji presentation, which renders as 2 cells. A naive
// per-char sum would count U+26A0 (1) + U+FE0F (0) = 1 and shift table
// borders one column. Guards the regression where ⚠️ measured as 1.
assert_eq!(visible_width_u16(&to_u16("\u{26A0}\u{FE0F}"), DEFAULT_TAB_WIDTH), 2); // ⚠️
assert_eq!(visible_width_u16(&to_u16("\u{2139}\u{FE0F}"), DEFAULT_TAB_WIDTH), 2); // ℹ️
assert_eq!(visible_width_u16(&to_u16("\u{2764}\u{FE0F}"), DEFAULT_TAB_WIDTH), 2); // ❤️
assert_eq!(visible_width_u16(&to_u16("0\u{FE0F}\u{20E3}"), DEFAULT_TAB_WIDTH), 2); // 0️⃣ keycap
// Bare symbol without VS16 keeps text-presentation width (1 cell).
assert_eq!(visible_width_u16(&to_u16("\u{26A0}"), DEFAULT_TAB_WIDTH), 1);
// Intrinsically wide emoji are unaffected.
assert_eq!(visible_width_u16(&to_u16("\u{2705}"), DEFAULT_TAB_WIDTH), 2); // ✅
assert_eq!(visible_width_u16(&to_u16("\u{274C}"), DEFAULT_TAB_WIDTH), 2); // ❌
}
#[test]
fn test_visible_width_jamo_correction_inside_combining_cluster() {
let jamo_cells = if cfg!(target_os = "macos") { 1 } else { 2 };
let filler_cells = if cfg!(target_os = "macos") { 1 } else { 0 };
assert_eq!(visible_width_u16(&to_u16("\u{3141}\u{0301}"), DEFAULT_TAB_WIDTH), jamo_cells);
assert_eq!(visible_width_u16(&to_u16("\u{3164}\u{0301}"), DEFAULT_TAB_WIDTH), filler_cells);
}
#[test]
fn test_ansi_detection() {
let data = to_u16("\x1b[31mred\x1b[0m");
+205 -23
View File
@@ -663,7 +663,7 @@ async fn run_shell_command(
.await;
if cancel_token.is_cancelled() {
terminate_background_jobs(&session.shell);
terminate_background_jobs(&mut session.shell);
}
if env_scope_pushed {
@@ -833,7 +833,7 @@ async fn run_shell_command_streams(
.await;
if cancel_token.is_cancelled() {
terminate_background_jobs(&session.shell);
terminate_background_jobs(&mut session.shell);
}
if env_scope_pushed {
@@ -998,9 +998,10 @@ async fn terminate_new_descendants<S: std::hash::BuildHasher + Sync>(baseline: &
}
}
}
fn terminate_background_jobs(shell: &BrushShell) {
fn terminate_background_jobs(shell: &mut BrushShell) {
let mut targets = process::TerminationTargets::new();
for job in &shell.jobs().jobs {
for job in &mut shell.jobs_mut().jobs {
job.abort_internal_tasks();
if let Some(pgid) = job.process_group_id() {
targets.add_pgid(pgid);
}
@@ -1009,11 +1010,9 @@ fn terminate_background_jobs(shell: &BrushShell) {
}
}
if targets.is_empty() {
// Pure descendant cleanup is handled by `process_cancel_bridge` while
// the cancel was still in flight. Here we only signal brush's own
// job-tracked targets — pgids of background-group leaders that may have
// already exited (so the descendant walk would no longer find them as
// new descendants, but their group still holds live grandchildren).
// Shell-internal jobs were aborted above. Pure descendant cleanup is
// handled by `process_cancel_bridge` while the cancel was in flight;
// without job-tracked pgids or pids there is nothing else to signal here.
return;
}
@@ -1634,13 +1633,16 @@ mod tests {
assert_eq!(child_session_action(true, true, true), ChildSessionAction::TakeForeground,);
}
/// Brush leading a new pgroup with non-terminal stdin detaches only when
/// it is not part of a multi-command pipeline. Pipeline leaders must stay
/// in the parent session so later stages can join their process group.
/// Brush leading a new pgroup with non-terminal stdin always detaches —
/// including the first stage of a pipeline. `setsid()` keeps the child
/// off the host's controlling tty; the spawn path skips
/// `process_group(...)` for detached children, so later stages no
/// longer try to `setpgid`-join a leader that has moved sessions (the
/// historical EPERM hazard).
#[test]
fn non_terminal_stdin_leading_new_pgroup_detaches_unless_pipeline() {
fn non_terminal_stdin_detaches_regardless_of_pipeline() {
assert_eq!(child_session_action(true, false, false), ChildSessionAction::DetachSession,);
assert_eq!(child_session_action(true, false, true), ChildSessionAction::None,);
assert_eq!(child_session_action(true, false, true), ChildSessionAction::DetachSession,);
}
/// Non-interactive brush, terminal stdin, no pipeline: nothing to do.
@@ -1665,16 +1667,16 @@ mod tests {
assert_eq!(child_session_action(false, false, false), ChildSessionAction::DetachSession,);
}
/// **Pipeline carve-out.** Non-interactive brush, non-terminal stdin
/// (pipe), and a multi-command pipeline: MUST NOT detach. For the first
/// external stage, `setsid()` puts the process-group leader into a
/// different session, so later stages fail to join its group with
/// EPERM. For later stages, `setsid()` would either fail with EPERM or
/// move the child into a new session, breaking the pipeline's shared
/// process group and job-control signal propagation.
/// **Pipeline tty-safety.** Non-interactive brush, non-terminal stdin
/// (pipe), and a multi-command pipeline: detach. An interactive child in
/// a pipeline (`zsh -i ... | awk`) would otherwise open `/dev/tty`,
/// `tcsetpgrp` itself to the foreground, and leave the host stopped on
/// its next tty read (`suspended (tty input)`). Each stage gets its own
/// session instead; the embedded host cancels via the descendant tree,
/// not a shared pgroup, and pipes are session-independent.
#[test]
fn pipeline_stage_does_not_detach() {
assert_eq!(child_session_action(false, false, true), ChildSessionAction::None,);
fn pipeline_stage_with_non_terminal_stdin_detaches() {
assert_eq!(child_session_action(false, false, true), ChildSessionAction::DetachSession,);
}
}
@@ -1796,6 +1798,126 @@ mod tests {
);
}
/// Regression for the `suspended (tty input)` bug: an **interactive child
/// inside a pipeline** (`zsh -i ... | awk`) used to stay in the host
/// session, open `/dev/tty`, `tcsetpgrp` itself to the foreground, and
/// leave the embedded host (OMP) stopped on its next tty read. The earlier
/// embedded-host fix carved pipelines out of `detach_session` because a
/// later stage that `setpgid`-joined a detached leader failed with EPERM.
///
/// This test boots a real embedded `BrushShell` and runs a two-stage
/// pipeline whose first stage prints its PID then sleeps (forwarded to us
/// by `cat`). It asserts two contracts at once:
/// 1. the first stage runs in its **own session** (`getsid == own pid`),
/// so it can never reach the host's controlling tty — guards the
/// decision; and
/// 2. the pipeline still exits **successfully**, proving the second stage
/// spawned without the cross-session `setpgid` EPERM — guards the
/// wiring that skips `process_group(...)` for detached children.
#[cfg(unix)]
#[tokio::test(flavor = "multi_thread")]
async fn embedded_pipeline_stage_runs_in_its_own_session() {
use std::io::Read as _;
// SAFETY: `getsid(0)` only queries the current process session; checked below.
let host_sid = unsafe { libc::getsid(0) };
assert!(host_sid > 0, "getsid(0) failed: {}", std::io::Error::last_os_error());
let config = ShellConfig { session_env: None, snapshot_path: None, minimizer: None };
let mut session = create_session(&config).await.expect("create_session");
let (mut reader, writer) = pipe_to_files("e2e-pipe").expect("pipe");
let stdout_file = OpenFile::from(writer.try_clone().expect("clone"));
let stderr_file = OpenFile::from(writer);
let mut params = session.shell.default_exec_params();
params.set_fd(OpenFiles::STDIN_FD, null_file().expect("null stdin"));
params.set_fd(OpenFiles::STDOUT_FD, stdout_file);
params.set_fd(OpenFiles::STDERR_FD, stderr_file);
let (pid_tx, pid_rx) = tokio::sync::oneshot::channel::<i32>();
let reader_handle = tokio::task::spawn_blocking(move || {
let mut buf = Vec::new();
let mut chunk = [0u8; 64];
let mut pid_tx = Some(pid_tx);
while let Ok(n) = reader.read(&mut chunk)
&& n > 0
{
buf.extend_from_slice(&chunk[..n]);
if pid_tx.is_some()
&& let Some(line_end) = buf.iter().position(|&byte| byte == b'\n')
&& let Ok(line) = std::str::from_utf8(&buf[..line_end])
&& let Ok(pid) = line.trim().parse::<i32>()
{
let _ = pid_tx
.take()
.expect("pid sender should be present")
.send(pid);
}
}
buf
});
let shell_handle = tokio::spawn(async move {
let source_info = SourceInfo::from("pi-natives:test");
// First stage prints its own PID and sleeps; `cat` forwards the PID
// line to our reader and exits on EOF. The first stage leads the
// pipeline's process group, the second (`cat`) is the join-or-detach
// stage that would EPERM without the wiring fix.
let exec = session
.shell
.run_string(
"/bin/sh -c 'printf \"%d\\n\" \"$$\"; sleep 1' | /bin/cat",
&source_info,
&params,
)
.await
.expect("run_string");
drop(params);
(session, exec)
});
let child_pid = time::timeout(Duration::from_secs(5), pid_rx)
.await
.expect("timed out waiting for first-stage PID")
.expect("reader closed pid channel without sending");
assert!(child_pid > 0, "got non-positive child pid: {child_pid}");
// SAFETY: `child_pid` is a live positive PID (still in `sleep`); the return
// value is checked.
let child_sid = unsafe { libc::getsid(child_pid) };
assert!(
child_sid > 0,
"getsid({child_pid}) failed: {} (child may have already exited)",
std::io::Error::last_os_error(),
);
let (_session, exec) = time::timeout(Duration::from_secs(5), shell_handle)
.await
.expect("shell timed out")
.expect("shell task panicked");
// Guards the wiring: the second stage spawned without a cross-session
// `setpgid` EPERM, so the whole pipeline succeeded.
assert!(
matches!(exec.exit_code, ExecutionExitCode::Success),
"pipeline did not succeed (second stage may have hit setpgid EPERM): {}",
exit_code(&exec),
);
let _ = time::timeout(Duration::from_secs(2), reader_handle).await;
// Guards the decision: a pipeline stage must not share the host session,
// or it could seize the controlling tty and SIGTTIN the host.
assert_ne!(
child_sid, host_sid,
"pipeline stage PID {child_pid} inherited host session {host_sid}; it could seize the \
controlling tty — the pipeline tty-suspend bug is back",
);
assert_eq!(
child_sid, child_pid,
"pipeline stage PID {child_pid} should be its own session leader after setsid",
);
}
#[tokio::test]
async fn abort_state_signals_cancel_token() {
let abort_state = ShellAbortState::default();
@@ -1811,6 +1933,66 @@ mod tests {
assert!(matches!(reason, AbortReason::Signal));
}
#[tokio::test(flavor = "multi_thread")]
async fn cancellation_aborts_internal_background_jobs() {
let unique = std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.expect("system clock before epoch")
.as_nanos();
let dir =
std::env::temp_dir().join(format!("pi-shell-bg-cancel-{}-{unique}", std::process::id()));
std::fs::create_dir(&dir).expect("create temp dir");
let started = dir.join("started");
let release = dir.join("release");
let marker = dir.join("marker");
let config = ShellConfig { session_env: None, snapshot_path: None, minimizer: None };
let mut session = create_session(&config).await.expect("create session");
session
.shell
.set_working_dir(dir.to_string_lossy().as_ref())
.expect("set cwd");
let mut params = session.shell.default_exec_params();
params.set_fd(OpenFiles::STDIN_FD, null_file().expect("null stdin"));
params.set_fd(OpenFiles::STDOUT_FD, null_file().expect("null stdout"));
params.set_fd(OpenFiles::STDERR_FD, null_file().expect("null stderr"));
let source_info = SourceInfo::from("pi-shell:test");
let result = session
.shell
.run_string(
"{ echo started > started; while [ ! -f release ]; do sleep 0.05; done; echo done > \
marker; } &",
&source_info,
&params,
)
.await
.expect("spawn background job");
assert_eq!(exit_code(&result), 0);
let mut background_started = false;
for _ in 0..200 {
if started.exists() {
background_started = true;
break;
}
time::sleep(Duration::from_millis(10)).await;
}
assert!(background_started, "background job did not reach its wait loop");
terminate_background_jobs(&mut session.shell);
std::fs::write(&release, b"").expect("release marker");
time::sleep(Duration::from_millis(250)).await;
let marker_exists = marker.exists();
std::fs::remove_dir_all(&dir).expect("cleanup temp dir");
assert!(
!marker_exists,
"internal background job survived cancellation and wrote marker after release",
);
}
#[cfg(unix)]
#[tokio::test]
async fn read_output_stops_when_cancelled_before_pipe_eof() {
+33 -29
View File
@@ -1,5 +1,9 @@
# ERRATA — GPT-5 Harmony-Header Leakage
Historical research note, not a current runtime contract. The statistics below
come from the named local stats database snapshot, not from checked-in tests or
runtime code.
## 1. The problem
OpenAI frames tool calls in the Harmony chat protocol:
@@ -50,22 +54,22 @@ Source: `~/.omp/stats.db` (`ss_tool_calls`, `ss_assistant_msgs`), through
### 2.1 Rate
| Model | Leaks in tool args | Calls | per million |
|------------------|-------------------:|--------:|------------:|
| gpt-5.4 | 37 | 226,957 | 163 |
| gpt-5.3-codex | 17 | 112,243 | 151 |
| gpt-5.5 | 2 | 80,750 | 25 |
| gpt-5.2-codex | 0 | — | — |
| Model | Leaks in tool args | Calls | per million |
| ------------- | -----------------: | ------: | ----------: |
| gpt-5.4 | 37 | 226,957 | 163 |
| gpt-5.3-codex | 17 | 112,243 | 151 |
| gpt-5.5 | 2 | 80,750 | 25 |
| gpt-5.2-codex | 0 | — | — |
Plus 15 hits in assistant visible text / thinking blobs.
### 2.2 Tool distribution
| Tool | Hits |
|---------------------|-----:|
| `edit` | 38 |
| `eval` | 11 |
| `report_tool_issue` | 3 |
| Tool | Hits |
| ------------------------------ | -----: |
| `edit` | 38 |
| `eval` | 11 |
| `report_tool_issue` | 3 |
| `grep`/`read`/`search`/`yield` | 1 each |
Concentrated in tools with free-form (non-JSON-schema) argument formats.
@@ -83,8 +87,8 @@ JUNK_PREFIX ::= (GLITCH_TOKEN | CHANNEL_WORD | NON_LATIN_RUN | "}" | "】【")+
records, 39 contain ≥2 markers and 7 contain ≥3 — the model emits
multiple fake `to=functions.X code …` blocks back-to-back, often with
fake `code_output\nCell N:\n…` framing between them. Once the
plain-text scaffolding is in the residual stream, the prefix now *looks
like* a fresh tool envelope start, so the macro prior over continuations
plain-text scaffolding is in the residual stream, the prefix now _looks
like_ a fresh tool envelope start, so the macro prior over continuations
keeps voting for more scaffolding. Self-amplifying.
### 2.4 Glitch tokens
@@ -93,13 +97,13 @@ Single-token identifiers in `o200k_base` whose embeddings appear to be
near-init from underrepresentation in post-training. ASCII residue
immediately before the marker in the natural corpus:
| Surface string | Single-token | Token ID | Hits in corpus |
|-------------------|:-:|---------:|---:|
| `Japgolly` | ✅ | 199,745 | 1 |
| `Jsii` | ✅ | 114,318 | (subtoken of `Jsii_commentary`) |
| `Jsii_commentary` | — (3 toks) | — | 2 |
| `changedFiles` | — (2 toks) | — | 8 |
| `RTLU` | — (2 toks) | — | 3 |
| Surface string | Single-token | Token ID | Hits in corpus |
| ----------------- | :----------: | -------: | ------------------------------: |
| `Japgolly` | ✅ | 199,745 | 1 |
| `Jsii` | ✅ | 114,318 | (subtoken of `Jsii_commentary`) |
| `Jsii_commentary` | — (3 toks) | — | 2 |
| `changedFiles` | — (2 toks) | — | 8 |
| `RTLU` | — (2 toks) | — | 3 |
`Japgolly` is in the last 0.13% of the vocabulary — the same family of
GitHub-corpus residue that produced `SolidGoldMagikarp` in the 2023
@@ -136,17 +140,17 @@ reproduction (§7.3), independent of the prompt's natural language.
The `edit` tool exists in two variants in the corpus:
| Variant | Calls | Recovery |
|--------------------------|------:|----------|
| Patch-DSL (`§PATH`/anchor/`«»≔` ops) | 27 | **Recoverable** by op-truncation (§3.3) |
| JSON-schema (`{path,edits:[…]}`) | 11 | **Not recoverable** — contamination is escaped *inside* JSON strings, parser accepts it cleanly, content would be written verbatim into source files |
| Variant | Calls | Recovery |
| ------------------------------------ | ----: | ---------------------------------------------------------------------------------------------------------------------------------------------------- |
| Patch-DSL (`§PATH`/anchor/`«»≔` ops) | 27 | **Recoverable** by op-truncation (§3.3) |
| JSON-schema (`{path,edits:[…]}`) | 11 | **Not recoverable** — contamination is escaped _inside_ JSON strings, parser accepts it cleanly, content would be written verbatim into source files |
For Patch-DSL leaks specifically:
- 20/27 cases: contamination on the last input line; nothing follows.
- 7/27 cases: contamination mid-input; what follows is one of: a
duplicate replay of an earlier file/anchor, intended content for a
*different* tool call (the model started its next call inline), or
_different_ tool call (the model started its next call inline), or
pure hallucination. Post-contamination content is never trustworthy.
### 2.8 Mechanism (confirmed)
@@ -167,7 +171,7 @@ Step by step:
merge corpus but barely in LM/RL training, so its **input embedding
`e_g` ≈ near-init noise of small norm**.
3. At position t+1, the residual update `h_{t+1} ≈ LN(h_t + e_g + Attn +
MLP)` is dominated by the prefix-derived terms; the just-emitted-token
MLP)` is dominated by the prefix-derived terms; the just-emitted-token
signal is effectively absent. Generation diversity normally comes
from `e_x` steering the residual into different sub-regions —
stripped here.
@@ -179,7 +183,7 @@ Step by step:
5. The mask zeros the control-token IDs. Mass redistributes onto the
**next-best continuation**: the un-bracketed surface-form spelling of
the same protocol (`analysis`, `commentary`, ` to=functions.X`,
` code `). This spelling is unmasked because those characters are
`code`). This spelling is unmasked because those characters are
ordinary tokens.
6. Once a few tokens of plain-text scaffolding land in the residual
stream, the prefix now resembles a fresh envelope start. The macro
@@ -194,7 +198,7 @@ explained:**
- **The brackets never appear** (§1, §2.5). The mask is what makes the
leak land in plain text instead of as a real envelope-close.
- **Counterintuitive grammar dependency** (§7.4). The leak is *worse* in
- **Counterintuitive grammar dependency** (§7.4). The leak is _worse_ in
formats closest to OpenAI's training distribution. Off-distribution
custom grammars dampen the macro-prior basin; the official
`*** Begin Patch` format is the strongest collapse target.
@@ -202,4 +206,4 @@ explained:**
The 2023 SolidGoldMagikarp paper documented mechanism (1)+(2)+(4). The
new piece is (5): when constrained decoding masks the natural collapse
target, the mass laundered through the un-masked plain-text shadow
becomes a structurally-invisible exfiltration channel.
becomes a structurally-invisible exfiltration channel.
+21 -22
View File
@@ -43,14 +43,14 @@ Removed in the unified-flow refactor:
## Dispatcher mapping
| Provider transport(s) | Dispatcher |
| -------------------------------------------------------------------- | -------------------------------------------- |
| `openai-completions`, `openai-responses`, `openai-codex-responses` | `adaptSchemaForStrict` (sanitize + enforce) |
| `openai-responses` family (`oneOf` → `anyOf` only) | `normalizeSchemaForOpenAIResponses` |
| `google-generative-ai`, `google-vertex`, Gemini CLI | `normalizeSchemaForGoogle` |
| Cloud Code Assist Claude (Antigravity + GCA, `claude-*` model ids) | `normalizeSchemaForCCA` |
| MCP `inputSchema` ingestion | `normalizeSchemaForMCP` |
| `anthropic-messages` (native, not CCA) | per-provider whitelist in `anthropic.ts` |
| Provider transport(s) | Dispatcher |
| ------------------------------------------------------------------ | ------------------------------------------- |
| `openai-completions`, `openai-responses`, `openai-codex-responses` | `adaptSchemaForStrict` (sanitize + enforce) |
| `openai-responses` family (`oneOf` → `anyOf` only) | `normalizeSchemaForOpenAIResponses` |
| `google-generative-ai`, `google-vertex`, Gemini CLI | `normalizeSchemaForGoogle` |
| Cloud Code Assist Claude (Antigravity + GCA, `claude-*` model ids) | `normalizeSchemaForCCA` |
| MCP `inputSchema` ingestion | `normalizeSchemaForMCP` |
| `anthropic-messages` (native, not CCA) | per-provider whitelist in `anthropic.ts` |
Gemini CLI / Antigravity CCA MUST run the full `normalizeSchemaForCCA`
pipeline (not just the first keyword-stripping pass) to keep parity with the
@@ -58,25 +58,25 @@ shared Google Claude path.
## Walk semantics
`normalizeSchema` first upgrades the input to JSON Schema 2020-12, then
walks the tree with the option set pinned by the dispatcher. Each node:
`normalizeSchema` first detoxifies serialized Zod-instance-shaped inputs, upgrades them to
JSON Schema 2020-12, dereferences the tree, then walks it with the option set
pinned by the dispatcher. Each node:
1. Inlines `$ref` (see "Edge cases" below).
2. Renames `snake_case` combinator/property keys to camelCase
1. Renames `snake_case` combinator/property keys to camelCase
(`any_of` → `anyOf`, etc.; collisions follow python-genai
`pop(from)`/`set(to)` semantics — snake_case wins).
3. Applies the `handle_null_fields` collapse for nullable unions before
2. Applies the `handle_null_fields` collapse for nullable unions before
recursing into children.
4. Strips keys the target provider does not support, optionally lifting
3. Strips keys the target provider does not support, optionally lifting
human-meaningful keys (`pattern`, `format`, min/max, `default`,
`examples`, ...) into the sibling `description` via the spill formatter
(`spill.ts`). Structural/meta keys (`$ref`, `$defs`,
`additionalProperties`) are not spilled.
5. Normalizes type unions (`type: ["T", "null"]` → `type: "T"` + nullable
4. Normalizes type unions (`type: ["T", "null"]` → `type: "T"` + nullable
marker on Google, plain `type: "T"` on CCA).
6. Collapses object-only / same-type combiners, optionally lossy-collapses
5. Collapses object-only / same-type combiners, optionally lossy-collapses
mixed-type combiners (CCA only), and runs the residual-combiner fixpoint.
7. Validates against AJV 2020 when `validateAndFallback` is set (CCA path)
6. Validates against AJV 2020 when `validateAndFallback` is set (CCA path)
and emits the per-tool fallback `{ "type": "object", "properties": {} }`
on residual incompatibility — `type` array, `type: "null"`, `nullable`
key, or any remaining `anyOf`/`oneOf`/`allOf`.
@@ -99,11 +99,10 @@ which composes:
(`anyOf: [<original>, { "type": "null" }]`). Tuple `prefixItems` are
strictified recursively.
The two passes share node-level caches and the same epoch-based cycle
guard, so a single walk on the wire path normalizes refs, allOf, and
nullable wrapping consistently. `tryEnforceStrictSchema` is fail-open:
if anything throws, it returns `{ strict: false, schema: original }` so
callers MUST emit `strict: true` only when enforcement actually succeeded.
The two passes use cache/cycle guards, so refs, `allOf`, and nullable wrapping
stay deterministic without recursing forever. `tryEnforceStrictSchema` is
fail-open: if anything throws, it returns `{ strict: false, schema: upgraded }`
so callers MUST emit `strict: true` only when enforcement actually succeeded.
### Edge cases the strict-mode normalizer handles
+18 -20
View File
@@ -6,7 +6,7 @@ Tool approval has two independent inputs:
- `read`: reads data or updates UI-only session metadata.
- `write`: mutates workspace/session state but does not execute arbitrary code.
- `exec`: executes code, shells out, drives a browser, spawns agents, or performs similarly broad actions.
2. **User policy** — `tools.approval.<toolName>: allow | deny | prompt` overrides the mode for that tool.
2. **User policy** — `tools.approval.<toolName>: allow | deny | prompt` overrides the mode for that tool unless a non-yolo safety override forces a prompt.
Tools without an `approval` declaration are treated as `exec`. This is the safe default for MCP and unknown custom tools.
@@ -14,13 +14,11 @@ Tools without an `approval` declaration are treated as `exec`. This is the safe
Configure with `tools.approvalMode`:
## Modes
| Mode | Auto-approves | Prompts for |
| --- | --- | --- |
| `always-ask` | `read` | `write`, `exec` |
| `write` | `read`, `write` | `exec` |
| `yolo` (default) | `read`, `write`, `exec` | none |
| Mode | Auto-approves | Prompts for |
| ---------------- | ----------------------- | --------------- |
| `always-ask` | `read` | `write`, `exec` |
| `write` | `read`, `write` | `exec` |
| `yolo` (default) | `read`, `write`, `exec` | none |
`--auto-approve` and `--yolo` force `tools.approvalMode: yolo` for the session.
@@ -40,12 +38,11 @@ tools:
Resolution per tool call:
1. Compute the tool's approval decision from `tool.approval(args)`; omitted means `exec`.
2. A user policy in `tools.approval.<tool>` is always applied.
3. In `yolo` mode, with no user policy, the call is auto-approved.
4. In non-yolo modes, if the tool sets `override: true`, `deny` is blocked and all other cases prompt.
5. Otherwise, the active mode auto-approves or prompts by tier.
Invalid policy values are ignored and fall back to the tool tier/mode decision.
2. Normalize `tools.approval.<tool>` if present; invalid values are ignored.
3. In `yolo` mode, the user policy is used when present; otherwise the call is allowed. Safety `override` reasons do not force a prompt in `yolo`.
4. In non-yolo modes, if the tool sets `override: true`, `deny` is blocked and all other cases prompt, even if user policy says `allow`.
5. Otherwise, a valid user policy wins.
6. Otherwise, the active mode auto-approves or prompts by tier.
## Safety overrides
@@ -55,7 +52,7 @@ A tool can force a prompt with object-form approval:
approval: { tier: "exec", override: true, reason: "Critical pattern detected" }
```
`bash` uses this for critical destructive patterns such as `rm -rf /`, fork bombs, remote-fetch-then-execute, writes to `/etc/passwd`, and host shutdown commands. These surface as `reason` in the approval prompt, but in `yolo` mode they are auto-approved unless a user policy for the tool is set to `prompt` or `deny`.
`bash` uses this for critical destructive patterns such as `rm -rf /`, fork bombs, remote-fetch-then-execute, writes to `/etc/passwd`, and host shutdown commands. These surface as `reason` in the approval prompt, but in `yolo` mode they are auto-approved unless a user policy for the tool is set to `prompt` or `deny`.
## Per-tool prompt details
@@ -82,13 +79,14 @@ formatApprovalDetails?: (args: unknown) => string | string[] | undefined;
Examples:
```ts
approval: "read"
approval: "read";
approval: args => LSP_READONLY_ACTIONS.has(args.action) ? "read" : "write"
approval: (args) => (LSP_READONLY_ACTIONS.has(args.action) ? "read" : "write");
approval: args => isCritical(args.command)
? { tier: "exec", override: true, reason: "Critical pattern detected" }
: "exec"
approval: (args) =>
isCritical(args.command)
? { tier: "exec", override: true, reason: "Critical pattern detected" }
: "exec";
```
## Subagents
+48 -43
View File
@@ -2,8 +2,8 @@
The auth broker and auth gateway are two cooperating HTTP services that move OAuth refresh tokens and provider access tokens off developer laptops and into a single broker host.
- **`omp auth-broker serve`** holds the canonical SQLite credential vault, performs OAuth refreshes, and exposes a small REST API (`/v1/snapshot`, `/v1/credential/:id/refresh`, `/v1/credential/:id/disable`, `/v1/credential`, `/v1/usage`, `/v1/healthz`).
- **`omp auth-gateway serve`** is a forward-proxy. It accepts OpenAI Chat Completions, Anthropic Messages, and OpenAI Responses requests, injects the broker-resolved access token, and forwards the bytes to the real provider. Clients (containerised omp, llm-git, the macOS usage widget, …) never see the access token.
- **`omp auth-broker serve`** holds the canonical SQLite credential vault, performs OAuth refreshes, and exposes a small REST API (`/v1/snapshot`, `/v1/snapshot/stream`, `/v1/credential/:id/refresh`, `/v1/credential/:id/disable`, `/v1/credential`, `/v1/usage`, `/v1/healthz`).
- **`omp auth-gateway serve`** is a forward-proxy. It accepts OpenAI Chat Completions, Anthropic Messages, OpenAI Responses, and pi-native stream requests, resolves the broker-backed credential, and dispatches through `pi-ai` provider logic. Clients (containerised omp, llm-git, the macOS usage widget, …) never see the access token.
Transport security between operator, broker, and gateway is delegated to the operator (Tailscale / Wireguard / reverse proxy + TLS). Every endpoint except `/v1/healthz` (broker) and `/healthz` (gateway) requires a bearer token.
@@ -25,20 +25,21 @@ Source: `packages/ai/src/auth-broker/`, `packages/ai/src/auth-gateway/`, `packag
│ ▼ │
│ ┌──────────────────────────┐ │
│ │ omp auth-gateway serve │ RemoteAuthCredentialStore │
│ │ /v1/{chat,messages,…} │ pulls /v1/snapshot at boot, │
│ │ /v1/usage, /v1/models │ refreshes credentials by id │
│ └─────────┬────────────────┘ via the broker on expiry │
│ │ /v1/{chat,messages,…} │ receives snapshot stream, │
│ │ /v1/usage,/v1/models │ refreshes credentials by id │
│ │ /v1/credentials/check │ via the broker on expiry │
│ └─────────┬────────────────┘ │
└────────────┼───────────────────────────────────────────────┘
│ bearer ($CONFIG_DIR/auth-gateway.token)
▼
unauthenticated clients
gateway clients
(llm-git, macOS widget, robomp containers, IDE plugins, …)
│
▼ same path is forwarded with Authorization
▼ provider request with broker-resolved credential
api.anthropic.com / api.openai.com / …
```
The broker is the only writer of OAuth refresh tokens. Clients (including the gateway itself) load a redacted snapshot in which every `refresh` field has been replaced with `REMOTE_REFRESH_SENTINEL`; when an access token expires the client calls `POST /v1/credential/:id/refresh` and the broker performs the refresh server-side. `RemoteAuthCredentialStore` rejects any local code path that tries to write through it, with an error pointing at `omp auth-broker login` / `omp auth-broker logout`.
The broker is the only writer of OAuth refresh tokens. Clients (including the gateway itself) load a redacted snapshot in which every `refresh` field has been replaced with `REMOTE_REFRESH_SENTINEL`; when an access token expires the client calls `POST /v1/credential/:id/refresh` and the broker performs the refresh server-side. `RemoteAuthCredentialStore` rejects local replace/upsert/delete-by-provider mutations, with errors pointing at `omp auth-broker login` / `omp auth-broker logout`.
## auth-broker
@@ -51,7 +52,7 @@ omp auth-broker login [<provider>] [--via=user@host] [--dry-run]
omp auth-broker logout [<provider>]
omp auth-broker list [--json]
omp auth-broker import <file|dir> [--provider=<id>] [--include-disabled] [--dry-run] [--json]
omp auth-broker migrate --from-local [--dry-run] [--json]
omp auth-broker migrate --from-local [--include-oauth] [--include-env] [--dry-run] [--json]
omp auth-broker status [--json]
```
@@ -61,19 +62,20 @@ omp auth-broker status [--json]
- `logout [<provider>]` deletes every credential row for `<provider>`. With no argument it shows an interactive numbered picker of currently-stored providers.
- `list` enumerates every registered OAuth provider id/name (the union of built-ins + `registerOAuthProvider` custom providers). `--json` emits a machine-readable array.
- `import <file|dir>` imports CLIProxyAPI-style JSON credentials into the local SQLite store. Maps `type` field → omp provider (`claude → anthropic`, `codex → openai-codex`, `gemini → google-gemini-cli`, `antigravity → google-antigravity`, `gemini-cli → google-gemini-cli`).
- `migrate --from-local` walks the local SQLite store + env-derived credentials and idempotently uploads them to the configured broker (`POST /v1/credential`).
- `migrate --from-local` uploads local SQLite credentials to the configured broker (`POST /v1/credential`). Local API keys are included by default; local OAuth rows are skipped unless `--include-oauth` is set; environment-derived API keys are skipped unless `--include-env` is set. Re-runs are idempotent against the broker snapshot.
- `status` health-pings the configured remote broker.
### Endpoints
| Method | Path | Auth | Purpose |
| ------ | ---- | ---- | ------- |
| `GET` | `/v1/healthz` | none | Liveness + version |
| `GET` | `/v1/snapshot` | bearer | Redacted snapshot (refresh tokens replaced by sentinel) |
| `POST` | `/v1/credential` | bearer | Upsert one OAuth or API-key credential |
| `POST` | `/v1/credential/:id/refresh` | bearer | Force-refresh one OAuth credential |
| `POST` | `/v1/credential/:id/disable` | bearer | Disable one credential with a recorded cause |
| `GET` | `/v1/usage` | bearer | Aggregate `UsageReport[]` across credentials |
| Method | Path | Auth | Purpose |
| ------ | ---------------------------- | ------ | ------------------------------------------------------- |
| `GET` | `/v1/healthz` | none | Liveness + version |
| `GET` | `/v1/snapshot` | bearer | Redacted snapshot (refresh tokens replaced by sentinel) |
| `GET` | `/v1/snapshot/stream` | bearer | SSE snapshot stream with delta events and keepalives |
| `POST` | `/v1/credential` | bearer | Upsert one OAuth or API-key credential |
| `POST` | `/v1/credential/:id/refresh` | bearer | Force-refresh one OAuth credential |
| `POST` | `/v1/credential/:id/disable` | bearer | Disable one credential with a recorded cause |
| `GET` | `/v1/usage` | bearer | Aggregate `UsageReport[]` across credentials |
Requests use `Authorization: Bearer <token>`. The server compares against an in-memory token allow-list; the gateway’s implementation uses a timing-safe comparison.
@@ -92,26 +94,29 @@ Requests use `Authorization: Bearer <token>`. The server compares against an in-
omp auth-gateway serve [--bind=host:port] [--no-auth]
omp auth-gateway token [--regenerate] [--json]
omp auth-gateway status [--json]
omp auth-gateway check [--strict] [--json]
```
- `serve` requires `OMP_AUTH_BROKER_URL` (or `auth.broker.url` in `config.yml`) — the gateway is itself a broker client. It calls `AuthBrokerClient.fetchSnapshot()`, wraps it in `RemoteAuthCredentialStore`, and constructs an `AuthStorage` that resolves access tokens through the broker. Default bind is `127.0.0.1:4000`. The gateway token is stored at `<config-dir>/auth-gateway.token` (`0600`); `--no-auth` disables the bearer check entirely (loopback-only use).
- `token` / `status` mirror the broker’s equivalents.
- `token` / `status` manage and inspect the gateway bearer token and upstream broker readiness.
- `check` probes broker-backed credentials through the gateway store. Without `--strict` it uses provider usage probes; `--strict` also exercises each credential against its chat-completion endpoint and can consume a small amount of quota.
### Endpoints
| Method | Path | Auth | Purpose |
| ------ | ---- | ---- | ------- |
| `GET` | `/healthz` | none | Liveness + version |
| `GET` | `/v1/usage` | bearer | Aggregate `UsageReport[]` (proxied through `AuthStorage`) |
| `GET` | `/v1/models` | bearer | Bundled-model catalog filtered to providers with credentials |
| `POST` | `/v1/chat/completions` | bearer | OpenAI Chat Completions wire format |
| `POST` | `/v1/messages` | bearer | Anthropic Messages wire format |
| `POST` | `/v1/responses` | bearer | OpenAI Responses wire format |
| Method | Path | Auth | Purpose |
| ------ | ----------------------- | ------ | ------------------------------------------------------------ |
| `GET` | `/healthz` | none | Liveness + version |
| `GET` | `/v1/usage` | bearer | Aggregate `UsageReport[]` (proxied through `AuthStorage`) |
| `GET` | `/v1/models` | bearer | Bundled-model catalog filtered to providers with credentials |
| `GET` | `/v1/credentials/check` | bearer | Per-credential auth health probe |
| `POST` | `/v1/chat/completions` | bearer | OpenAI Chat Completions wire format |
| `POST` | `/v1/messages` | bearer | Anthropic Messages wire format |
| `POST` | `/v1/responses` | bearer | OpenAI Responses wire format |
| `POST` | `/v1/pi/stream` | bearer | Native `pi-ai` stream wire format |
The model id is read from the top-level `model` field. The gateway picks the first bundled `Model<Api>` matching that id and:
The model id is read from the top-level `model` field for foreign wire formats and from the pi-native request body for `/v1/pi/stream`. The gateway picks the first bundled `Model<Api>` matching that id, parses the inbound wire format into an omp `Context`, resolves the provider credential from broker-backed `AuthStorage`, dispatches through `streamSimple()`, and re-encodes the result to the inbound format (SSE for streamed responses).
- **Passthrough fast-path** — when the inbound wire format matches the model’s native API (`openai-chat → openai-completions`, `anthropic-messages → anthropic-messages`, `openai-responses → openai-responses`), the request body is forwarded byte-for-byte with the client `Authorization`/`x-api-key` stripped and replaced by `Authorization: Bearer <resolved-access-token>`. Provider-specific fields (`cache_control`, `service_tier`, tool-choice extensions, …) flow through unmodified. Hop-by-hop headers (RFC 7230) plus `Content-Encoding`/`Content-Length` are stripped from the upstream response.
- **Translate path** — when the inbound format and the resolved model’s API differ (e.g. `/v1/chat/completions` targeting an Anthropic model, or `/v1/responses` targeting `openai-codex-responses` which runs over a websocket transport), the request is parsed against the wire schema, rebuilt into an omp `Context`, dispatched through `streamSimple()`, and re-encoded back to the inbound format (SSE for streamed responses).
There is no raw provider passthrough path. All supported routes go through `pi-ai` provider logic so credential-specific request shaping, OAuth refresh-on-auth-error, and provider quirks stay centralized.
`idleTimeout` on the underlying `Bun.serve` is set to `255 s` so long thinking-budget calls do not get killed by Bun’s default idle timeout.
@@ -141,14 +146,14 @@ The broker is **off** unless `OMP_AUTH_BROKER_URL` (or `auth.broker.url` in `con
### Environment variables
| Variable | Purpose | Required when |
| -------- | ------- | ------------- |
| `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`). Selecting this puts the client in broker mode — local SQLite is bypassed. | Any time the omp client should resolve credentials through a broker (and required by `omp auth-gateway serve`). |
| `OMP_AUTH_BROKER_TOKEN` | Bearer token used for every broker endpoint except `/v1/healthz`. | When `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `<config-dir>/auth-broker.token`. |
| Variable | Purpose | Required when |
| ----------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- |
| `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`). Selecting this puts the client in broker mode — local SQLite is bypassed. | Any time the omp client should resolve credentials through a broker (and required by `omp auth-gateway serve`). |
| `OMP_AUTH_BROKER_TOKEN` | Bearer token used for every broker endpoint except `/v1/healthz`. | When `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `<config-dir>/auth-broker.token`. |
Resolution order in `resolveAuthBrokerConfig()`:
1. `OMP_AUTH_BROKER_URL` env (else `auth.broker.url` from `config.yml`, with `$ENV_NAME` resolution);
1. `OMP_AUTH_BROKER_URL` env (else `auth.broker.url` from `config.yml`, resolved through `resolveConfigValue`);
2. `OMP_AUTH_BROKER_TOKEN` env (else `auth.broker.token` from `config.yml`, else `<config-dir>/auth-broker.token`);
3. URL set but no token resolvable → hard error pointing at the token file path.
@@ -156,16 +161,16 @@ The gateway has no dedicated env vars — it inherits `OMP_AUTH_BROKER_*` becaus
### `config.yml` keys
| Key | Default | Purpose |
| --- | ------- | ------- |
| `auth.broker.url` | unset | Same as `OMP_AUTH_BROKER_URL`; env wins. Hidden from the settings UI. |
| `auth.broker.token` | unset | Same as `OMP_AUTH_BROKER_TOKEN`; env wins. Values may be the literal token or `$ENV_NAME` to indirect through env. |
| Key | Default | Purpose |
| ------------------- | ------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `auth.broker.url` | unset | Same as `OMP_AUTH_BROKER_URL`; env wins. Hidden from the settings UI. Values are resolved as a literal, an environment variable name, or `!<shell command>` to use trimmed stdout. |
| `auth.broker.token` | unset | Same as `OMP_AUTH_BROKER_TOKEN`; env wins. Values are resolved the same way. |
### Token files
| Path | Owner | Mode |
| ---- | ----- | ---- |
| `<config-dir>/auth-broker.token` | `omp auth-broker serve` (created at first start) | `0600` in a `0700` parent dir |
| Path | Owner | Mode |
| --------------------------------- | ---------------------------------------------------- | ----------------------------- |
| `<config-dir>/auth-broker.token` | `omp auth-broker serve` (created at first start) | `0600` in a `0700` parent dir |
| `<config-dir>/auth-gateway.token` | `omp auth-gateway serve` (skipped under `--no-auth`) | `0600` in a `0700` parent dir |
`<config-dir>` resolves to `~/.omp/` (respecting `PI_CONFIG_DIR`).
@@ -178,6 +183,6 @@ The broker only owns OAuth credentials and provider-API-key credentials that wer
## See also
- [`secrets.md`](./secrets.md) — secret obfuscation around tokens that *do* leak through (e.g. `OMP_AUTH_BROKER_TOKEN` in shell output).
- [`secrets.md`](./secrets.md) — secret obfuscation around tokens that _do_ leak through (e.g. `OMP_AUTH_BROKER_TOKEN` in shell output).
- [`models.md`](./models.md) — provider auth resolution order; the broker plugs in at layers 2–3 (stored credentials).
- [`environment-variables.md`](./environment-variables.md) — full env reference including `OMP_AUTH_BROKER_URL` / `OMP_AUTH_BROKER_TOKEN`.
+22 -19
View File
@@ -10,7 +10,7 @@ There are two different bash execution surfaces in coding-agent:
1. **Tool-call surface** (`toolName: "bash"`): used when the model calls the bash tool.
- Entry point: `BashTool.execute()`.
- Parameters include `command`, optional `env`, `timeout`, `cwd`, `head`, `tail`, `pty`, and, when `async.enabled` is true, `async`.
- Parameters include `command`, optional `env`, `timeout`, `cwd`, `pty`, and, when `async.enabled` is true, `async`.
2. **User bang-command surface** (`!cmd` from interactive input or RPC `bash` command): session-level helper path.
- Entry point: `AgentSession.executeBash()`.
@@ -23,11 +23,11 @@ Both eventually use `executeBash()` in `src/exec/bash-executor.ts` for non-PTY e
`BashTool.execute()` currently handles input before execution as follows:
- validates optional `env` names against shell-variable syntax,
- extracts a leading `cd <path> && ...` into `cwd` when `cwd` was not supplied,
- rejects `async: true` when `async.enabled` is false,
- uses only explicit `head`/`tail` tool args for post-run filtering.
- when `bash.stripTrailingHeadTail` is enabled (default), applies conservative native fixups that remove safe trailing `| head` / `| tail` pipes and redundant trailing `2>&1`,
- extracts a leading single-line `cd <path> && ...` into `cwd` when `cwd` was not supplied,
- rejects `async: true` when `async.enabled` is false.
`normalizeBashCommand()` was previously in `src/tools/bash-normalize.ts` but has been removed. Trailing shell pipes such as `| head -n 50` remain part of the shell command unless the caller uses the structured `head`/`tail` args.
There are no structured `head` or `tail` tool parameters in the current schema. Output limiting is handled by `OutputSink` truncation/artifacts, and the optional trailing-pipe fixup exists to avoid hiding output before the harness can capture it.
## 2) Optional interception (blocked-command path)
@@ -173,6 +173,10 @@ Both PTY and non-PTY paths use `OutputSink`.
Runtime truncation is byte-threshold based in `OutputSink` (50KB default). It does not enforce a hard 2000-line cap in this code path.
### Shell output minimizer
Non-PTY execution also passes shell-minimizer settings into the native `Shell` session. When the minimizer rewrites verbose output, the executor replaces the sink's visible text with the minimized text and, when possible, saves the raw original capture as a separate `bash-original` artifact referenced by a `[raw output: artifact://<id>]` footer.
## Live tool updates and async jobs
For non-PTY foreground execution, `BashTool` uses a separate `TailBuffer` for partial updates and emits `onUpdate` snapshots while command is running.
@@ -189,12 +193,11 @@ After execution:
- if abort signal is aborted -> throw `ToolAbortError` (abort semantics),
- else -> throw `ToolError` (treated as tool failure).
2. PTY `timedOut` -> throw `ToolError`.
3. apply head/tail filters to final output text (`applyHeadTail`, head then tail).
4. empty output becomes `(no output)`.
5. attach truncation metadata via `toolResult(...).truncationFromSummary(result, { direction: "tail" })`.
6. exit-code mapping:
- missing exit code -> `ToolError("... missing exit status")`
- non-zero exit -> `ToolError("... Command exited with code N")`
3. empty output becomes `(no output)`.
4. attach truncation metadata via `toolResult(...).truncationFromSummary(result, { direction: "tail" })`.
5. exit-code mapping:
- missing exit code -> throw `ToolError("... missing exit status")`
- non-zero exit -> error result with `"Command exited with code N"` and `details.exitCode`
- zero exit -> success result.
Success payload structure:
@@ -236,13 +239,13 @@ This component is wired by `CommandController.handleBashCommand()` and fed from
## Mode-specific behavior differences
| Surface | Entry path | PTY eligible | Live output UX | Error surfacing |
| ------------------------------ | ----------------------------------------------------- | -------------------------------------------------------------------- | ------------------------------------------------------------------------ | ------------------------------------------------ |
| Interactive tool call | `BashTool.execute` | Yes, when `pty=true` and UI exists and `PI_NO_PTY!=1` | PTY overlay (interactive) or streamed tail updates | Tool errors become `toolResult.isError` |
| Print mode tool call | `BashTool.execute` | No (no UI context) | No TUI overlay; output appears in event stream/final assistant text flow | Same tool error mapping |
| RPC tool call (agent tooling) | `BashTool.execute` | Usually no UI -> non-PTY | Structured tool events/results | Same tool error mapping |
| Interactive bang command (`!`) | `AgentSession.executeBash` + `BashExecutionComponent` | No (uses executor directly) | Dedicated bash execution component | Controller catches exceptions and shows UI error |
| RPC `bash` command | `rpc-mode` -> `session.executeBash` | No | Returns `BashResult` directly | Consumer handles returned fields |
| Surface | Entry path | PTY eligible | Live output UX | Error surfacing |
| ------------------------------ | ----------------------------------------------------- | ----------------------------------------------------- | ------------------------------------------------------------------------ | ------------------------------------------------ |
| Interactive tool call | `BashTool.execute` | Yes, when `pty=true` and UI exists and `PI_NO_PTY!=1` | PTY overlay (interactive) or streamed tail updates | Tool errors become `toolResult.isError` |
| Print mode tool call | `BashTool.execute` | No (no UI context) | No TUI overlay; output appears in event stream/final assistant text flow | Same tool error mapping |
| RPC tool call (agent tooling) | `BashTool.execute` | Usually no UI -> non-PTY | Structured tool events/results | Same tool error mapping |
| Interactive bang command (`!`) | `AgentSession.executeBash` + `BashExecutionComponent` | No (uses executor directly) | Dedicated bash execution component | Controller catches exceptions and shows UI error |
| RPC `bash` command | `rpc-mode` -> `session.executeBash` | No | Returns `BashResult` directly | Consumer handles returned fields |
## Operational caveats
@@ -256,7 +259,7 @@ This component is wired by `CommandController.handleBashCommand()` and fed from
## Implementation files
- [`src/tools/bash.ts`](../packages/coding-agent/src/tools/bash.ts) — tool entrypoint, input handling/interception, async and PTY/non-PTY selection, result/error mapping, bash tool renderer.
- ~~`src/tools/bash-normalize.ts`~~ — removed (post-run head/tail filtering is now handled inline).
- [`src/tools/bash-command-fixup.ts`](../packages/coding-agent/src/tools/bash-command-fixup.ts) — native-backed conservative cleanup for trailing `head`/`tail` pipes and redundant `2>&1`.
- [`src/tools/bash-interceptor.ts`](../packages/coding-agent/src/tools/bash-interceptor.ts) — interceptor rule matching and blocked-command messages.
- [`src/exec/bash-executor.ts`](../packages/coding-agent/src/exec/bash-executor.ts) — non-PTY executor, shell session reuse, cancellation wiring, output sink integration.
- [`src/tools/bash-interactive.ts`](../packages/coding-agent/src/tools/bash-interactive.ts) — PTY runtime, overlay UI, input normalization, non-interactive env defaults.
+73 -64
View File
@@ -16,9 +16,9 @@ They are intentionally separate:
## Storage boundaries and on-disk layout
## Blob store boundary (global)
### Blob store boundary (global)
`SessionManager` constructs `BlobStore(getBlobsDir())`, so blob files live in a shared global blob directory (not in a session folder).
`SessionManager` constructs `BlobStore(getBlobsDir())`, so blob files live in a shared global blob directory, not in a session folder.
Blob file naming:
@@ -43,12 +43,15 @@ Artifact types share this directory:
- truncated tool output files: `<numericId>.<toolType>.log` (for `artifact://`)
- subagent output files: `<outputId>.md` (for `agent://`)
- subagent session JSONL sidecars: `<outputId>.jsonl` when task execution receives an artifacts directory
Subagents can adopt the parent `ArtifactManager`; in that case parent and subagent tree share one artifact directory and numeric artifact ID space.
## ID and name allocation schemes
## Blob IDs: content hash
### Blob IDs: content hash
`BlobStore.put()` computes SHA-256 over the bytes it is given and returns:
`BlobStore.put()` / `putSync()` computes SHA-256 over the bytes it is given and returns:
- `hash`: hex digest,
- `path`: `<blobsDir>/<hash>`,
@@ -56,27 +59,30 @@ Artifact types share this directory:
No session-local counter is used.
## Artifact IDs: session-local monotonic integer
### Artifact IDs: session-local monotonic integer
`ArtifactManager` scans existing `*.log` artifact files on first use to find max existing numeric ID and sets `nextId = max + 1`.
`ArtifactManager` scans existing `*.log` artifact files on first directory-backed allocation to find max existing numeric ID and sets `nextId = max + 1`.
Allocation behavior:
- file format: `{id}.{toolType}.log`
- IDs are sequential strings (`"0"`, `"1"`, ...)
- resume does not overwrite existing artifacts because scan happens before allocation.
- resume does not overwrite existing artifacts because scan happens before allocation
- the directory is created lazily on first save/allocation
If artifact directory is missing, scanning yields empty list and allocation starts from `0`.
If the artifact directory is missing, scanning yields an empty list and allocation starts from `0`.
## Agent output IDs (`agent://`)
Non-persistent sessions without an adopted manager can store `saveArtifact(...)` content in memory under numeric IDs, but `artifact://` resolution is file-backed through registered artifact directories.
`AgentOutputManager` allocates IDs for subagent outputs as `<index>-<requestedId>` (optionally nested under parent prefix, e.g. `0-Parent.1-Child`). It scans existing `.md` files on initialization to continue from the next index on resume.
### Agent output IDs (`agent://`)
`AgentOutputManager` allocates IDs for subagent outputs from the requested name, used verbatim the first time and suffixed (`-2`, `-3`, …) only when the same name repeats (e.g. `Anna`, `Anna-2`). Nested outputs are grouped under the parent prefix (e.g. `Parent.Child`). It scans existing `.md` files on initialization so a resumed session never reuses a name that would clobber a prior output.
## Persistence dataflow
## 1) Session entry persistence rewrite path
### 1) Session entry persistence rewrite path
Before session entries are written (`#rewriteFile` / incremental persist), `SessionManager` calls `prepareEntryForPersistence()` (via `truncateForPersistence`).
Before session entries are written (`#rewriteFile` / incremental persist), `SessionManager` calls `prepareEntryForPersistence()` / `prepareEntryForPersistenceSync()` through the truncation pipeline.
Key behaviors:
@@ -91,7 +97,7 @@ Key behaviors:
This keeps session JSONL compact while preserving recoverability.
## 2) Session load rehydration path
### 2) Session load rehydration path
When opening a session (`setSessionFile`), after migrations, `SessionManager` runs `resolveBlobRefsInEntries()`.
@@ -102,122 +108,125 @@ For message/custom-message image blocks with `blob:sha256:<hash>` and for persis
- converts provider `image_url` blobs back to the original string,
- mutates in-memory entry fields for runtime consumers.
If blob is missing:
If a blob is missing:
- `resolveImageData()` logs warning,
- returns original ref string unchanged,
- load continues (no hard crash).
- image-block resolution logs a warning and keeps the original `blob:sha256:` ref string in memory,
- provider `image_url` resolution logs a warning and keeps the original ref string,
- load continues.
## 3) Tool output spill/truncation path
### 3) Tool output spill/truncation path
`OutputSink` powers streaming output in bash/python/ssh and related executors.
Behavior:
1. Every chunk is sanitized and appended to in-memory tail buffer.
2. When in-memory bytes exceed spill threshold (`DEFAULT_MAX_BYTES`, 50KB), sink marks output truncated.
3. If an artifact path is available, sink opens a file writer and writes:
- existing buffered content once,
- all subsequent chunks.
4. In-memory buffer is always trimmed to tail window for display.
5. `dump()` returns summary including `artifactId` only when file sink was successfully created.
1. Every chunk is sanitized with `sanitizeWithOptionalSixelPassthrough(..., sanitizeText)` and appended to in-memory accounting.
2. Optional live `onChunk` receives sanitized pre-column-cap chunks, throttled if configured.
3. A per-line column cap can drop bytes from long lines in the LLM-facing buffer; when this happens, artifact mirroring starts so the on-disk file keeps the full sanitized stream.
4. When the in-memory tail buffer would exceed spill threshold (`DEFAULT_MAX_BYTES`, 50KB), sink marks output truncated and starts artifact mirroring if an artifact path is available.
5. If a file sink is opened, it first writes the current buffer, then all queued/subsequent sanitized chunks.
6. In-memory buffer is trimmed to a tail window, or to head + elision marker + tail when head retention is configured.
7. `dump()` returns summary including `artifactId` only when file sink creation succeeded.
Practical effect:
- UI/tool return shows truncated tail,
- full output is preserved in artifact file and referenced as `artifact://<id>`.
- UI/tool return shows bounded output,
- full sanitized output is preserved in artifact file and referenced as `artifact://<id>` when file-backed artifact mirroring succeeded.
If file sink creation fails (I/O error, missing path, etc.), sink silently falls back to in-memory truncation only; full output is not persisted.
If file sink creation fails (I/O error, missing path, etc.), sink falls back to in-memory truncation only; full output is not persisted.
## URL access model
## `blob:` references
### `blob:` references
`blob:sha256:<hash>` is a persistence reference inside session entry payloads, not an internal URL scheme handled by the router. Resolution is done by `SessionManager` during session load.
## `artifact://<id>`
### `artifact://<id>`
Handled by `ArtifactProtocolHandler`:
Handled by `ArtifactProtocolHandler` over registered active session artifact directories:
- requires active session artifact directory,
- ID must be numeric,
- resolves by matching filename prefix `<id>.`,
- requires a numeric ID,
- searches each registered artifacts directory for filename prefix `<id>.`,
- returns raw text (`text/plain`) from the matched `.log` file,
- when missing, error includes list of available artifact IDs.
- when missing, error includes available numeric artifact IDs from existing artifact files.
Missing directory behavior:
Failure behavior:
- if artifacts directory does not exist, throws `No artifacts directory found`.
- if no artifact directories are registered: throws `No session - artifacts unavailable`,
- if registered directories exist but none are present on disk: throws `No artifacts directory found`,
- if ID is not numeric: throws `artifact:// ID must be numeric, got: <id>`.
## `agent://<id>`
### `agent://<id>`
Handled by `AgentProtocolHandler` over `<artifactsDir>/<id>.md`:
Handled by `AgentProtocolHandler` over registered active session artifact directories and `<artifactsDir>/<id>.md`:
- plain form returns markdown text,
- `/path` or `?q=` forms perform JSON extraction,
- path and query extraction cannot be combined,
- if extraction requested, file content must parse as JSON.
Missing directory behavior:
Failure behavior:
- throws `No artifacts directory found`.
Missing output behavior:
- throws `Not found: <id>` with available IDs from existing `.md` files.
- if no artifact directories are registered: throws `No session - agent outputs unavailable`,
- if registered directories exist but none are present on disk: throws `No artifacts directory found`,
- missing output throws `Not found: <id>` with available `.md` output IDs when directory listing succeeds.
Read tool integration:
- `read` supports offset/limit pagination for non-extraction internal URL reads,
- rejects `offset/limit` when `agent://` extraction is used.
- rejects offset/limit when `agent://` extraction is used.
## Resume, fork, and move semantics
## Resume
### Resume
- `ArtifactManager` scans existing `{id}.*.log` files on first allocation and continues numbering.
- `AgentOutputManager` scans existing `.md` output IDs and continues numbering.
- `SessionManager` rehydrates blob refs to base64 on load.
- `SessionManager` rehydrates blob refs to base64/data URLs on load.
## Fork
### Fork
`SessionManager.fork()` creates a new session file with new session ID and `parentSession` link, then returns old/new file paths. Artifact copying is handled by `AgentSession.fork()`:
- flushes current session first,
- attempts recursive copy of old artifact directory to new artifact directory,
- missing old directory is tolerated,
- non-ENOENT copy errors are logged as warnings and fork still completes.
ID implications after fork:
- if copy succeeded, artifact counters in new session continue after max copied ID,
- if copy succeeded, artifact counters in the new session continue after max copied ID when the new `ArtifactManager` first scans,
- if copy failed/skipped, new session artifact IDs start from `0`.
Blob implications after fork:
- blobs are global and content-addressed, so no blob directory copy is required.
## Move to new cwd
### Move to new cwd
`SessionManager.moveTo()` renames both session file and artifact directory to the new default session directory, with rollback logic if a later step fails. This preserves artifact identity while relocating session scope.
## Failure handling and fallback paths
| Case | Behavior |
| -------------------------------------------------------- | --------------------------------------------------------------------- |
| Blob file missing during rehydration | Warn and keep `blob:sha256:` ref string in-memory |
| Blob read ENOENT via `BlobStore.get` | Returns `null` |
| Artifact directory missing (`ArtifactManager.listFiles`) | Returns empty list (allocation can start fresh) |
| Artifact directory missing (`artifact://` / `agent://`) | Throws explicit `No artifacts directory found` |
| Artifact ID not found | Throws with available IDs listing |
| OutputSink artifact writer init fails | Continues with tail-only truncation (no full-output artifact) |
| No session file (some task paths) | Task tool falls back to temp artifacts directory for subagent outputs |
| Case | Behavior |
| --------------------------------------------------------- | -------------------------------------------------------------------- |
| Blob file missing during image-block rehydration | Warn and keep `blob:sha256:` ref string in memory |
| Blob file missing during provider `image_url` rehydration | Warn and keep `blob:sha256:` ref string in memory |
| Blob read ENOENT via `BlobStore.get` | Returns `null` |
| Artifact directory missing (`ArtifactManager.listFiles`) | Returns empty list (allocation can start fresh) |
| No registered artifact dirs (`artifact://`) | Throws `No session - artifacts unavailable` |
| No registered artifact dirs (`agent://`) | Throws `No session - agent outputs unavailable` |
| Registered artifact dirs missing on disk | Throws explicit `No artifacts directory found` |
| Artifact ID not found | Throws with available IDs listing |
| OutputSink artifact writer init fails | Continues with bounded in-memory output only |
| Non-persistent `saveArtifact` | Stores text in `SessionManager` memory map; not file-backed URL data |
## Binary blob externalization vs text-output artifacts
- **Blob externalization** is for image payloads inside persisted session entry content and provider image data URLs; it replaces inline payload strings in JSONL with stable content refs.
- **Artifacts** are plain text files for execution output and subagent output; they are addressable by session-local IDs through internal URLs.
- **Artifacts** are plain text files for execution output and subagent output; file-backed artifacts are addressable by session-local IDs through internal URLs.
The two systems intersect only indirectly (both reduce session JSONL bloat) but have different identity, lifetime, and retrieval paths.
The two systems intersect only indirectly: both reduce session JSONL bloat, but they have different identity, lifetime, and retrieval paths.
## Implementation files
@@ -228,6 +237,6 @@ The two systems intersect only indirectly (both reduce session JSONL bloat) but
- [`src/session/agent-session.ts`](../packages/coding-agent/src/session/agent-session.ts) — artifact directory copy during interactive fork.
- [`src/internal-urls/artifact-protocol.ts`](../packages/coding-agent/src/internal-urls/artifact-protocol.ts) — `artifact://` resolver.
- [`src/internal-urls/agent-protocol.ts`](../packages/coding-agent/src/internal-urls/agent-protocol.ts) — `agent://` resolver + JSON extraction.
- [`src/sdk.ts`](../packages/coding-agent/src/sdk.ts) — internal URL router wiring and artifacts-dir resolver.
- [`src/internal-urls/router.ts`](../packages/coding-agent/src/internal-urls/router.ts) — internal URL router wiring.
- [`src/task/output-manager.ts`](../packages/coding-agent/src/task/output-manager.ts) — session-scoped agent output ID allocation for `agent://`.
- [`src/task/executor.ts`](../packages/coding-agent/src/task/executor.ts) — subagent output artifact writes (`<id>.md`) and temp artifact directory fallback.
- [`src/task/executor.ts`](../packages/coding-agent/src/task/executor.ts) — subagent output artifact writes (`<id>.md`) and session JSONL sidecars.
+25 -12
View File
@@ -53,12 +53,13 @@ Those custom roles are then transformed into LLM-facing user messages in `conver
### Triggers
Compaction/context maintenance can run in four ways:
Compaction/context maintenance can run in five ways:
1. **Manual context compaction**: `/compact [instructions]` calls `AgentSession.compact(...)`.
2. **Automatic overflow recovery**: after a same-model assistant error that matches context overflow.
3. **Automatic threshold maintenance**: after a successful turn when context exceeds the resolved threshold.
4. **Idle maintenance**: `runIdleCompaction()` can invoke the same auto-maintenance path with reason `"idle"`.
3. **Automatic incomplete-output recovery**: after a same-model assistant message ends with `stopReason === "length"` (OpenAI/Codex `response.incomplete`).
4. **Automatic threshold maintenance**: after a successful turn when context exceeds the resolved threshold.
5. **Idle maintenance**: `runIdleCompaction()` can invoke the same auto-maintenance path with reason `"idle"`.
### Compaction shape (visual)
@@ -94,7 +95,7 @@ What the LLM sees:
prompt from cmp messages from firstKeptEntryId
```
### Overflow-retry vs threshold/idle maintenance
### Overflow/incomplete recovery vs threshold/idle maintenance
The automatic paths are intentionally different:
@@ -102,15 +103,23 @@ The automatic paths are intentionally different:
- Trigger: current-model assistant error is detected as context overflow and the error is not older than the latest compaction.
- The failing assistant error message is removed from active agent state before retry.
- Context promotion is tried first; if a configured larger model is available, the agent switches model and retries without compacting.
- If promotion is unavailable and compaction is enabled, context-full compaction runs with `reason: "overflow"` and `willRetry: true`; handoff strategy is not used for overflow.
- On success, agent auto-continues (`agent.continue()`) after compaction.
- If promotion is unavailable and compaction is enabled, context-full compaction runs with `reason: "overflow"` and `willRetry: true`; handoff strategy is not used for overflow because the handoff request would reuse the overflowing input.
- On success, `agent.continue()` is scheduled to retry the turn.
- **Incomplete-output recovery**
- Trigger: same-model assistant message ends with `stopReason === "length"` and the message is not older than the latest compaction.
- The incomplete assistant message is removed from active agent state before recovery.
- Context promotion is tried first.
- If promotion is unavailable and compaction is enabled, auto maintenance runs with `reason: "incomplete"` and `willRetry: true`.
- Unlike overflow, `compaction.strategy: "handoff"` is allowed for incomplete-output recovery because the input context is still usable.
- On context-full success, `agent.continue()` is scheduled to retry the turn.
- **Threshold maintenance**
- Trigger: successful, non-error assistant message whose adjusted context tokens exceed `resolveThresholdTokens(...)`.
- Tool-output pruning can reduce the measured token count before threshold comparison.
- Context promotion is tried before compaction.
- If promotion is unavailable, auto maintenance runs with `reason: "threshold"` and `willRetry: false`.
- With `compaction.strategy: "handoff"`, threshold maintenance starts a new handoff session instead of writing a compaction entry; if handoff returns no document without aborting, it falls back to context-full compaction.
- With `compaction.strategy: "handoff"`, threshold maintenance normally schedules a post-prompt auto-handoff task instead of writing a compaction entry; pre-prompt checks run it inline to avoid racing the next turn. If handoff returns no document without aborting, it falls back to context-full compaction.
- On success, if `compaction.autoContinue !== false`, schedules an agent-authored developer auto-continue prompt from `prompts/system/auto-continue.md`.
- **Idle maintenance**
@@ -188,7 +197,7 @@ Final stored summary is merged as:
2. Serialize with `serializeConversation()`.
3. Wrap in `<conversation>...</conversation>`.
4. Optionally include `<previous-summary>...</previous-summary>`.
5. Optionally inject hook context as `<additional-context>` list.
5. Optionally inject extension hook context and active memory-backend compaction context as `<additional-context>` entries.
6. Execute summarization prompt with `SUMMARIZATION_SYSTEM_PROMPT`.
Prompt selection:
@@ -244,7 +253,8 @@ After summary generation (or hook-provided summary), agent session:
1. Appends `CompactionEntry` with `appendCompaction(...)` for context-full maintenance; handoff strategy creates a new session and injects a handoff `custom_message` instead.
2. Rebuilds display context from the active leaf via `buildDisplaySessionContext()`.
3. Replaces live agent messages with rebuilt context.
4. Emits `session_compact` hook event.
4. Synchronizes active todo phases from the rebuilt branch and closes provider sessions whose history was rewritten.
5. Emits `session_compact` hook event.
## Branch summarization pipeline
@@ -348,13 +358,14 @@ Post-navigation event exposing new/old leaf and optional summary entry.
## Runtime behavior and failure semantics
- Manual compaction aborts current agent operation first.
- `abortCompaction()` cancels both manual and auto-compaction controllers.
- `abortCompaction()` cancels manual compaction, auto-compaction, and handoff generation controllers.
- Auto compaction emits start/end session events for UI/state updates.
- Auto compaction can try multiple model candidates and retry transient failures; long retry delays prefer the next candidate when one is available.
- Overflow errors are excluded from generic retry path because they are handled by context promotion/compaction.
- If auto-compaction fails:
- overflow path emits `Context overflow recovery failed: ...`
- threshold path emits `Auto-compaction failed: ...`
- incomplete-output path emits `Incomplete response recovery failed: ...`
- threshold/idle paths emit `Auto-compaction failed: ...`
- Branch summarization can be cancelled via abort signal (e.g., Escape), returning canceled/aborted navigation result.
## Settings and defaults
@@ -369,7 +380,9 @@ From `settings-schema.ts`:
- `compaction.remoteEnabled` = `true`
- `compaction.remoteEndpoint` = `undefined`
- `compaction.thresholdPercent` = `-1` and `compaction.thresholdTokens` = `-1`; when no positive override is set, the threshold is `contextWindow - max(15% of contextWindow, reserveTokens)`
- `compaction.idleEnabled` = `true`
- `compaction.idleEnabled` = `false`
- `compaction.idleThresholdTokens` = `200000`
- `compaction.idleTimeoutSeconds` = `300`
- `branchSummary.enabled` = `false`
- `branchSummary.reserveTokens` = `16384`
+5 -5
View File
@@ -137,7 +137,7 @@ Legacy migration still supported:
The runtime settings model is layered:
1. Global settings: `~/.omp/agent/config.yml`
2. Project settings: discovered via settings capability (`settings.json` from providers)
2. Project settings: discovered via settings capability (`settings.json` and `config.yml` from providers)
3. Runtime overrides: in-memory, non-persistent
4. Schema defaults: from `SETTINGS_SCHEMA`
@@ -217,7 +217,7 @@ Native provider (`id: native`) reads native config from:
- Slash commands, rules, prompts, instructions, hooks, tools, extensions, extension modules, and settings use a project/user root only when the root directory exists and is non-empty.
- Skills scan `<ancestor>/.omp/skills` for each ancestor from the current working directory up to the repo root/home boundary, plus `~/.omp/agent/skills`, without requiring the root `.omp` directory itself to be non-empty.
- `SYSTEM.md` and `AGENTS.md` read user-level files directly and use nearest-ancestor project `.omp` lookup for project files, but the project `.omp` directory must be non-empty.
- `SYSTEM.md` and `AGENTS.md` read user-level files directly and use nearest-ancestor project `.omp` lookup for project files, but the project `.omp` directory must be non-empty. See [`docs/system-prompt-customization.md`](./system-prompt-customization.md) for the full `SYSTEM.md` / `APPEND_SYSTEM.md` contract (replace vs. append, templating).
### Scope-specific loading
@@ -230,7 +230,7 @@ Native provider (`id: native`) reads native config from:
- Tools: `tools/*.{json,md,ts,js,sh,bash,py}` and `tools/<name>/index.ts`
- Extension modules: discovered under `extensions/` (+ legacy `settings.json.extensions` string array)
- Extensions: `extensions/<name>/gemini-extension.json`
- Settings capability: `settings.json`
- Settings capability: `settings.json`, then `config.yml`
### Nearest-project lookup nuance
@@ -240,7 +240,7 @@ Native provider (`id: native`) reads native config from:
## Settings subsystem
- `Settings.init()` loads global `config.yml` + discovered project `settings.json` capability items.
- `Settings.init()` loads global `config.yml` + discovered project settings capability items.
- Only capability items with `level === "project"` are merged into project layer.
## Skills subsystem
@@ -285,7 +285,7 @@ Settings capability items are not deduplicated; `Settings.#loadProjectSettings()
- `ConfigFile` JSON -> YAML migration for YAML-targeted files.
- Settings migration from `settings.json` and `agent.db` to `config.yml`.
- Settings key migrations (`queueMode`, `ask.timeout`, flat `theme`, `task.isolation.enabled`, `statusLine.plan_mode`).
- Settings key migrations include `queueMode`, `ask.timeout`, flat `theme`, `task.isolation.enabled`, legacy `task.isolation.mode` values, removed edit modes, `statusLine.plan_mode`, `memories.enabled`, and hindsight scoping/name fields.
- Legacy setting names `skills.enablePiUser` / `skills.enablePiProject` are still active gates for native skill source.
If these compatibility paths are removed in code, update this document immediately; several runtime behaviors still depend on them today.
+40 -34
View File
@@ -67,39 +67,43 @@ A custom tool module must export a function (default export preferred):
import type { CustomToolFactory } from "@oh-my-pi/pi-coding-agent";
const factory: CustomToolFactory = (pi) => ({
name: "repo_stats",
label: "Repo Stats",
description: "Counts tracked TypeScript files",
parameters: pi.zod.object({
glob: pi.zod.string().optional().default("**/*.ts"),
}),
name: "repo_stats",
label: "Repo Stats",
description: "Counts tracked TypeScript files",
parameters: pi.zod.object({
glob: pi.zod.string().optional().default("**/*.ts"),
}),
async execute(toolCallId, params, onUpdate, ctx, signal) {
onUpdate?.({
content: [{ type: "text", text: "Scanning files..." }],
details: { phase: "scan" },
});
async execute(toolCallId, params, onUpdate, ctx, signal) {
onUpdate?.({
content: [{ type: "text", text: "Scanning files..." }],
details: { phase: "scan" },
});
const result = await pi.exec("git", ["ls-files", params.glob ?? "**/*.ts"], { signal, cwd: pi.cwd });
if (result.killed) {
throw new Error("Scan was cancelled");
}
if (result.code !== 0) {
throw new Error(result.stderr || "git ls-files failed");
}
const result = await pi.exec(
"git",
["ls-files", params.glob ?? "**/*.ts"],
{ signal, cwd: pi.cwd },
);
if (result.killed) {
throw new Error("Scan was cancelled");
}
if (result.code !== 0) {
throw new Error(result.stderr || "git ls-files failed");
}
const files = result.stdout.split("\n").filter(Boolean);
return {
content: [{ type: "text", text: `Found ${files.length} files` }],
details: { count: files.length, sample: files.slice(0, 10) },
};
},
const files = result.stdout.split("\n").filter(Boolean);
return {
content: [{ type: "text", text: `Found ${files.length} files` }],
details: { count: files.length, sample: files.slice(0, 10) },
};
},
onSession(event) {
if (event.reason === "shutdown") {
// cleanup resources if needed
}
},
onSession(event) {
if (event.reason === "shutdown") {
// cleanup resources if needed
}
},
});
export default factory;
@@ -122,11 +126,11 @@ From `types.ts` and `loader.ts`:
- `ui`: UI context (can be no-op in headless modes)
- `hasUI`: `false` in non-interactive flows
- `logger`: shared file logger
- `zod`: injected `zod` module (use `pi.zod.object`, `pi.zod.string`, …)
- `typebox`: zod-backed compatibility shim for legacy TypeBox-style schemas
- `zod`: injected `zod/v4` module (canonical for new schemas)
- `pi`: injected `@oh-my-pi/pi-coding-agent` exports
- `pushPendingAction(action)`: register a preview action for hidden `resolve` tool (`docs/resolve-tool-runtime.md`)
Loader starts with a no-op UI context and requires host code to call `setUIContext(...)` when real UI is ready.
Loader starts with a no-op UI context and requires host code to call `setUIContext(...)` when real UI is ready.
## Execution contract and typing
@@ -136,14 +140,16 @@ Loader starts with a no-op UI context and requires host code to call `setUIConte
execute(toolCallId, params, onUpdate, ctx, signal);
```
- `params` is statically typed from your Zod schema via `z.infer<typeof schema>` (`Static<TParams>` in API types).
- `params` is statically typed from your Zod/TypeBox schema via `Static<TParams>`.
- Runtime argument validation happens before execution in the agent loop.
- `onUpdate` emits partial results for UI streaming.
- `ctx` includes session/model state and an `abort()` helper.
- `ctx` includes `sessionManager`, `modelRegistry`, current `model`, `isIdle()`, `hasQueuedMessages()`, `abort()`, and optional `settings` / `autoApprove`.
- `signal` carries cancellation.
`CustomToolAdapter` bridges this to the agent tool interface and forwards calls in the correct argument order.
Tool definitions may also declare `strict`, `hidden`, `deferrable`, `mcpServerName`, `mcpToolName`, `approval`, and `formatApprovalDetails`.
## How tools are exposed to the model
- Tools are wrapped into `AgentTool` instances (`CustomToolAdapter` or extension wrappers).
+86 -64
View File
@@ -41,6 +41,7 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
| `GROQ_API_KEY` | Groq auth | Using Groq models | |
| `CEREBRAS_API_KEY` | Cerebras auth | Using Cerebras models | |
| `FIREWORKS_API_KEY` | Fireworks auth | Using Fireworks models | |
| `FIREPASS_API_KEY` | Fire Pass auth | Using Fire Pass models | |
| `TOGETHER_API_KEY` | Together auth | Using `together` provider | |
| `HUGGINGFACE_HUB_TOKEN` | Hugging Face auth | Using `huggingface` provider | Primary Hugging Face token env var |
| `HF_TOKEN` | Hugging Face auth | Using `huggingface` provider | Fallback when `HUGGINGFACE_HUB_TOKEN` is unset |
@@ -54,10 +55,12 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
| `LLAMA_CPP_API_KEY` | llama.cpp auth (optional) | Using `llama.cpp` provider with authenticated hosts | Local llama.cpp usually runs without auth; any non-empty token works when a key is configured |
| `XIAOMI_API_KEY` | Xiaomi MiMo auth | Using `xiaomi` provider | |
| `MOONSHOT_API_KEY` | Moonshot auth | Using `moonshot` provider | |
| `XAI_API_KEY` | xAI auth | Using xAI models | |
| `XAI_API_KEY` | xAI auth | Using xAI models or as fallback for `xai-oauth` | |
| `XAI_OAUTH_TOKEN` | xAI OAuth/SuperGrok auth | Using `xai-oauth` provider | Takes precedence over `XAI_API_KEY` for `xai-oauth` |
| `OPENROUTER_API_KEY` | OpenRouter auth | Using OpenRouter models | Also used by image tool when preferred/auto provider is OpenRouter |
| `MISTRAL_API_KEY` | Mistral auth | Using Mistral models | |
| `ZAI_API_KEY` | z.ai auth | Using z.ai models | Also used by z.ai web search provider |
| `ZHIPU_API_KEY` | Zhipu Coding Plan auth | Using `zhipu-coding-plan` provider | |
| `MINIMAX_API_KEY` | MiniMax auth | Using `minimax` provider | |
| `MINIMAX_CODE_API_KEY` | MiniMax Code auth | Using `minimax-code` provider | |
| `MINIMAX_CODE_CN_API_KEY` | MiniMax Code CN auth | Using `minimax-code-cn` provider | |
@@ -90,10 +93,10 @@ These are consumed via `getEnvApiKey()` (`packages/ai/src/stream.ts`) unless not
When the broker is enabled, the local SQLite credential store is bypassed and all OAuth refresh / access tokens live on the broker host. See [`auth-broker-gateway.md`](./auth-broker-gateway.md) for the full protocol, CLI surface, and 5-min/15-s usage cache layering.
| Variable | Used for | Required when | Notes / precedence |
| ----------------------- | ------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`); selects broker mode | Resolving credentials through a broker; also required by `omp auth-gateway serve` (the gateway is itself a broker client) | Wins over `auth.broker.url` in `config.yml`. When set with no resolvable token, `resolveAuthBrokerConfig()` hard-errors instead of falling back to local SQLite. |
| `OMP_AUTH_BROKER_TOKEN` | Bearer token sent on every broker endpoint except `/v1/healthz` | `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `<config-dir>/auth-broker.token` | Resolution: this env → `auth.broker.token` (`$ENV_NAME` indirection supported) → `<config-dir>/auth-broker.token` (mode `0600`). `<config-dir>` is `~/.omp/` (respecting `PI_CONFIG_DIR`). |
| Variable | Used for | Required when | Notes / precedence |
| ----------------------- | -------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
| `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`); selects broker mode | Resolving credentials through a broker; also required by `omp auth-gateway serve` (the gateway is itself a broker client) | Wins over `auth.broker.url` in `config.yml`. When set with no resolvable token, `resolveAuthBrokerConfig()` hard-errors instead of falling back to local SQLite. |
| `OMP_AUTH_BROKER_TOKEN` | Bearer token sent on every broker endpoint except `/v1/healthz` | `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `<config-dir>/auth-broker.token` | Resolution: this env → `auth.broker.token` (`$ENV_NAME` indirection supported) → `<config-dir>/auth-broker.token` (mode `0600`). `<config-dir>` is `~/.omp/` (respecting `PI_CONFIG_DIR`). |
The gateway has no dedicated env vars — it inherits `OMP_AUTH_BROKER_*`. Its own inbound bearer token lives at `<config-dir>/auth-gateway.token` and is managed via `omp auth-gateway token`.
@@ -108,7 +111,11 @@ When `CLAUDE_CODE_USE_FOUNDRY` is enabled, Anthropic requests switch to Foundry
- Base URL resolves from `FOUNDRY_BASE_URL` (fallback remains model/default base URL if unset).
- API key resolution for provider `anthropic` becomes:
`ANTHROPIC_FOUNDRY_API_KEY` → `ANTHROPIC_OAUTH_TOKEN` → `ANTHROPIC_API_KEY`.
- `ANTHROPIC_CUSTOM_HEADERS` is parsed as comma/newline-separated `key: value` pairs and merged into request headers.
- `ANTHROPIC_CUSTOM_HEADERS` is parsed as comma/newline-separated `key: value`
pairs and merged into request headers. They are also forwarded when
`ANTHROPIC_BASE_URL` points to a non-Anthropic host (e.g. a corporate API
gateway), so enterprise gateways requiring proprietary auth headers work
without enabling Foundry mode.
- TLS client/server material can be injected from env values:
`NODE_EXTRA_CA_CERTS`, `CLAUDE_CODE_CLIENT_CERT`, `CLAUDE_CODE_CLIENT_KEY`.
Each accepts either:
@@ -120,7 +127,7 @@ When `CLAUDE_CODE_USE_FOUNDRY` is enabled, Anthropic requests switch to Foundry
| `CLAUDE_CODE_USE_FOUNDRY` | Boolean-like string (`1`, `true`, `yes`, `on`) | Enables Foundry mode for Anthropic provider |
| `FOUNDRY_BASE_URL` | URL string | Anthropic endpoint base URL in Foundry mode |
| `ANTHROPIC_FOUNDRY_API_KEY` | Token string | Used for `Authorization: Bearer <token>` |
| `ANTHROPIC_CUSTOM_HEADERS` | Header list string | Extra headers; format `header-a: value, header-b: value` or newline-separated |
| `ANTHROPIC_CUSTOM_HEADERS` | Header list string | Extra headers; format `header-a: value, header-b: value` or newline-separated. Also forwarded outside Foundry whenever `ANTHROPIC_BASE_URL` is non-Anthropic. |
| `NODE_EXTRA_CA_CERTS` | PEM path or inline PEM | Extra CA chain for server certificate validation |
| `CLAUDE_CODE_CLIENT_CERT` | PEM path or inline PEM | mTLS client certificate |
| `CLAUDE_CODE_CLIENT_KEY` | PEM path or inline PEM | mTLS client private key (must be paired with cert) |
@@ -133,7 +140,7 @@ When `CLAUDE_CODE_USE_FOUNDRY` is enabled, Anthropic requests switch to Foundry
| `AWS_DEFAULT_REGION` | Fallback if `AWS_REGION` unset |
| `AWS_PROFILE` | Enables named profile auth path |
| `AWS_ACCESS_KEY_ID` + `AWS_SECRET_ACCESS_KEY` | Enables IAM key auth path |
| `AWS_BEARER_TOKEN_BEDROCK` | Highest-precedence bearer token auth path; skips AWS profile/credential-chain lookup when set |
| `AWS_BEARER_TOKEN_BEDROCK` | Highest-precedence bearer token auth path; skips AWS profile/credential-chain lookup when set |
| `AWS_CONTAINER_CREDENTIALS_RELATIVE_URI` / `AWS_CONTAINER_CREDENTIALS_FULL_URI` | Enables ECS task credential path |
| `AWS_WEB_IDENTITY_TOKEN_FILE` + `AWS_ROLE_ARN` | Enables web identity auth path |
| `AWS_BEDROCK_SKIP_AUTH` | If `1`, injects dummy credentials (proxy/non-auth scenarios) |
@@ -159,10 +166,13 @@ Base URL resolution: option `azureBaseUrl` → env `AZURE_OPENAI_BASE_URL` → o
| Variable | Required? | Notes |
| -------------------------------- | ------------------------------ | ------------------------------------------------------------------------------------------------------------------------- |
| `GOOGLE_CLOUD_PROJECT` | Yes (unless passed in options) | Fallback: `GCLOUD_PROJECT` |
| `GCLOUD_PROJECT` | Fallback | Used as alternate project ID source |
| `GOOGLE_CLOUD_PROJECT` | Yes (unless passed in options) | Primary project ID source |
| `GCP_PROJECT` | Fallback | Alternate project ID source |
| `GCLOUD_PROJECT` | Fallback | Alternate project ID source |
| `GOOGLE_CLOUD_PROJECT_ID` | OAuth login helper only | Used by Gemini CLI OAuth project discovery |
| `GOOGLE_CLOUD_LOCATION` | Yes (unless passed in options) | No default in provider |
| `GOOGLE_VERTEX_LOCATION` | Yes (unless passed in options) | Primary Vertex location source |
| `GOOGLE_CLOUD_LOCATION` | Fallback | Alternate Vertex location source |
| `VERTEX_LOCATION` | Fallback | Alternate Vertex location source |
| `GOOGLE_CLOUD_API_KEY` | Conditional | Direct Vertex API-key auth; otherwise ADC fallback can authenticate when project and location are set |
| `GOOGLE_APPLICATION_CREDENTIALS` | Conditional | If set, file must exist; otherwise ADC fallback path is checked (`~/.config/gcloud/application_default_credentials.json`) |
@@ -184,15 +194,16 @@ OAuth host chain: `KIMI_CODE_OAUTH_HOST` → `KIMI_OAUTH_HOST` → `https://auth
### OpenAI Codex responses (feature/debug controls)
| Variable | Behavior |
| ------------------------------------ | ---------------------------------------------------- |
| `PI_CODEX_DEBUG` | `1`/`true` enables Codex provider debug logging |
| `PI_CODEX_WEBSOCKET` | `1`/`true` enables websocket transport preference |
| `PI_CODEX_WEBSOCKET_V2` | `1`/`true` enables websocket v2 path |
| `PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS` | Positive integer override (default 300000) |
| `PI_CODEX_WEBSOCKET_RETRY_BUDGET` | Non-negative integer override (default 5) |
| `PI_CODEX_WEBSOCKET_RETRY_DELAY_MS` | Positive integer base backoff override (default 500) |
| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override |
| Variable | Behavior |
| ------------------------------------------ | ---------------------------------------------------- |
| `PI_CODEX_DEBUG` | `1`/`true` enables Codex provider debug logging |
| `PI_CODEX_WEBSOCKET` | `1`/`true` enables websocket transport preference |
| `PI_CODEX_WEBSOCKET_V2` | `1`/`true` enables websocket v2 path |
| `PI_CODEX_WEBSOCKET_IDLE_TIMEOUT_MS` | Positive integer override (default 300000) |
| `PI_CODEX_WEBSOCKET_RETRY_BUDGET` | Non-negative integer override (default 5) |
| `PI_CODEX_WEBSOCKET_RETRY_DELAY_MS` | Positive integer base backoff override (default 500) |
| `PI_OPENAI_STREAM_FIRST_EVENT_TIMEOUT_MS` | Positive integer OpenAI first-event timeout override |
| `PI_OPENAI_STREAM_IDLE_TIMEOUT_MS` | Positive integer OpenAI stream idle timeout override |
### Cursor provider debug
@@ -235,22 +246,28 @@ SearXNG also reads the equivalent `searxng.endpoint`, `searxng.token`, `searxng.
### Anthropic web search auth chain
Anthropic web search uses `findAnthropicAuth()` from `packages/ai/src/utils/anthropic-auth.ts` in this order:
`searchAnthropic()` resolves credentials in this order:
1. `ANTHROPIC_SEARCH_API_KEY` (+ optional `ANTHROPIC_SEARCH_BASE_URL`)
2. `ANTHROPIC_FOUNDRY_API_KEY` when `CLAUDE_CODE_USE_FOUNDRY` is enabled
3. Anthropic OAuth credentials from `agent.db` (must not expire within 5-minute buffer)
4. Anthropic API-key credentials from `agent.db`
5. Generic Anthropic env fallback: provider key (`ANTHROPIC_FOUNDRY_API_KEY` in Foundry mode, otherwise `ANTHROPIC_OAUTH_TOKEN`/`ANTHROPIC_API_KEY`) + optional `ANTHROPIC_BASE_URL` (`FOUNDRY_BASE_URL` when Foundry mode is enabled)
1. `ANTHROPIC_SEARCH_API_KEY`
2. `authStorage.getApiKey("anthropic")` fallback credentials (runtime/config overrides, stored API-key credentials, stored OAuth credentials, then generic Anthropic env fallback: `ANTHROPIC_FOUNDRY_API_KEY` in Foundry mode, otherwise `ANTHROPIC_OAUTH_TOKEN` / `ANTHROPIC_API_KEY`)
For either credential path, base URL resolution is:
1. `ANTHROPIC_SEARCH_BASE_URL`
2. `FOUNDRY_BASE_URL` when `CLAUDE_CODE_USE_FOUNDRY` is enabled
3. `ANTHROPIC_BASE_URL`
4. `https://api.anthropic.com`
Related vars:
| Variable | Default / behavior |
| --------------------------- | ---------------------------------------------------- |
| `ANTHROPIC_SEARCH_API_KEY` | Highest-priority explicit search key |
| `ANTHROPIC_SEARCH_BASE_URL` | Defaults to `https://api.anthropic.com` when omitted |
| `ANTHROPIC_SEARCH_MODEL` | Defaults to `claude-haiku-4-5` |
| `ANTHROPIC_BASE_URL` | Generic fallback base URL for tier-4 auth path |
| Variable | Default / behavior |
| --------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `ANTHROPIC_SEARCH_API_KEY` | API key used exclusively for the Anthropic web search provider. Highest-priority search auth; overrides `ANTHROPIC_API_KEY` / OAuth / Foundry for search calls without affecting chat completions. |
| `ANTHROPIC_SEARCH_BASE_URL` | Base URL used exclusively for the Anthropic web search provider. Applied to either `ANTHROPIC_SEARCH_API_KEY` or fallback Anthropic credentials; overrides `ANTHROPIC_BASE_URL` (and `FOUNDRY_BASE_URL` in Foundry mode) for search calls. |
| `ANTHROPIC_SEARCH_MODEL` | Search model override. Defaults to `claude-haiku-4-5`. |
| `ANTHROPIC_BASE_URL` | Generic fallback base URL for Anthropic requests when no search-specific base URL is set. |
Use `ANTHROPIC_SEARCH_BASE_URL` (optionally with `ANTHROPIC_SEARCH_API_KEY`) to keep chat routed through an enterprise gateway (`ANTHROPIC_BASE_URL` or `CLAUDE_CODE_USE_FOUNDRY=true`) while pointing web search at a direct Anthropic endpoint, or vice versa.
### Perplexity OAuth flow behavior flag
@@ -262,13 +279,13 @@ Related vars:
## 4) Python tooling and kernel runtime
| Variable | Default / behavior |
| ------------------------- | ------------------------------------------------------------------------------------------------------------------- |
| `PI_PY` | Eval backend override: `0`/`bash`=JavaScript only, `1`/`py`=Python only, `mix`/`both`=both; invalid values ignored |
| `PI_PYTHON_SKIP_CHECK` | If `1`, skips Python interpreter availability checks (subprocess runner still starts on demand) |
| `PI_PYTHON_INTEGRATION` | If `1`, opts gated integration tests in (e.g. `python-runner.integration.test.ts`) into running against real Python |
| `PI_PYTHON_IPC_TRACE` | If `1`, logs NDJSON frames exchanged with the Python runner subprocess |
| `VIRTUAL_ENV` | Highest-priority venv path for Python runtime resolution |
| Variable | Default / behavior |
| ----------------------- | ------------------------------------------------------------------------------------------------------------------- |
| `PI_PY` | Eval backend override: `0`/`bash`=JavaScript only, `1`/`py`=Python only, `mix`/`both`=both; invalid values ignored |
| `PI_PYTHON_SKIP_CHECK` | If `1`, skips Python interpreter availability checks (subprocess runner still starts on demand) |
| `PI_PYTHON_INTEGRATION` | If `1`, opts gated integration tests in (e.g. `python-runner.integration.test.ts`) into running against real Python |
| `PI_PYTHON_IPC_TRACE` | If `1`, logs NDJSON frames exchanged with the Python runner subprocess |
| `VIRTUAL_ENV` | Highest-priority venv path for Python runtime resolution |
Extra conditional behavior:
@@ -279,32 +296,36 @@ Extra conditional behavior:
## 5) Agent/runtime behavior toggles
| Variable | Default / behavior |
| ---------------------------- | -------------------------------------------------------------------------------------------------- |
| `PI_SMOL_MODEL` | Ephemeral model-role override for `smol` (CLI `--smol` takes precedence) |
| `PI_SLOW_MODEL` | Ephemeral model-role override for `slow` (CLI `--slow` takes precedence) |
| `PI_PLAN_MODEL` | Ephemeral model-role override for `plan` (CLI `--plan` takes precedence) |
| `PI_NO_TITLE` | If set (any non-empty value), disables auto session title generation on first user message |
| `NULL_PROMPT` | If `true`, system prompt builder returns empty string |
| `PI_BLOCKED_AGENT` | Blocks a specific subagent type in task tool |
| `PI_SUBPROCESS_CMD` | Overrides subagent spawn command (`omp` / `omp.cmd` resolution bypass) |
| `PI_TASK_MAX_OUTPUT_BYTES` | Max captured output bytes per subagent (default `500000`) |
| `PI_TASK_MAX_OUTPUT_LINES` | Max captured output lines per subagent (default `5000`) |
| Variable | Default / behavior |
| ---------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `PI_SMOL_MODEL` | Ephemeral model-role override for `smol` (CLI `--smol` takes precedence) |
| `PI_SLOW_MODEL` | Ephemeral model-role override for `slow` (CLI `--slow` takes precedence) |
| `PI_PLAN_MODEL` | Ephemeral model-role override for `plan` (CLI `--plan` takes precedence) |
| `PI_NO_TITLE` | If set (any non-empty value), disables auto session title generation on first user message |
| `PI_TINY_DEVICE` | ONNX execution provider for local tiny models; overrides the `providers.tinyModelDevice` setting (default: CPU; supports `cpu`, `gpu`, `metal`/`webgpu`, `auto`, `cuda`, `dml`, `coreml`, `wasm`, `webnn`, `webnn-gpu`, `webnn-cpu`, `webnn-npu`) |
| `PI_TINY_DTYPE` | ONNX quantization/precision for local tiny models; overrides the `providers.tinyModelDtype` setting (default: each model's shipped dtype, currently `q4`; supports `auto`, `fp32`, `fp16`, `q8`, `int8`, `uint8`, `q4`, `bnb4`, `q4f16`, `q2`, `q2f16`, `q1`, `q1f16`) |
| `PI_NO_INTERLEAVED_THINKING` | If `1`, disables Anthropic interleaved thinking budget behavior and uses output-token inflation for older thinking mode |
| `NULL_PROMPT` | If `true`, system prompt builder returns empty string |
| `PI_BLOCKED_AGENT` | Blocks a specific subagent type in task tool |
| `PI_SUBPROCESS_CMD` | Overrides subagent spawn command (`omp` / `omp.cmd` resolution bypass) |
| `PI_TASK_MAX_OUTPUT_BYTES` | Max captured output bytes per subagent (default `500000`) |
| `PI_TASK_MAX_OUTPUT_LINES` | Max captured output lines per subagent (default `5000`) |
| `PI_TIMING` | If set (any non-empty value), prints a hierarchical timing-span tree to **stderr** via `logger.printTimings()`. In interactive mode the tree prints once the agent is ready (before the TUI starts); in print mode it prints after the whole prompt batch completes. Print-mode prompts are wrapped in `print:prompt:initial` / `print:prompt:next` spans so each user message shows up as its own row. `PI_TIMING=x` exits the process with code 0 right after printing in interactive mode (use to measure cold startup only). `PI_TIMING=full` lists every module-load entry instead of just the top N. |
| `PI_PACKAGE_DIR` | Overrides package asset base dir resolution (docs/examples/changelog path lookup) |
| `PI_DISABLE_LSPMUX` | If `1`, disables lspmux detection/integration and forces direct LSP server spawning |
| `PI_RPC_EMIT_TITLE` | Boolean-like flag enabling title events in RPC mode |
| `SMITHERY_URL` | Smithery web URL override (default `https://smithery.ai`) |
| `SMITHERY_API_URL` | Smithery API base URL override (default `https://api.smithery.ai`) |
| `PUPPETEER_EXECUTABLE_PATH` | Browser tool Chromium executable override |
| `LM_STUDIO_BASE_URL` | Default implicit LM Studio discovery base URL override (`http://127.0.0.1:1234/v1` if unset) |
| `OLLAMA_BASE_URL` | Default implicit Ollama discovery base URL override (`http://127.0.0.1:11434` if unset) |
| `LLAMA_CPP_BASE_URL` | Default implicit Llama.cpp discovery base URL override (`http://127.0.0.1:8080` if unset) |
| `PI_EDIT_VARIANT` | Forces edit tool variant when valid (`patch`, `replace`, `hashline`, `apply_patch`) |
| `PI_FORCE_IMAGE_PROTOCOL` | Forces supported image protocol (`kitty`, `iterm2`/`iterm`, `sixel`, `none`) where used |
| `PI_ALLOW_SIXEL_PASSTHROUGH` | Allows SIXEL passthrough when `PI_FORCE_IMAGE_PROTOCOL=sixel` |
| `PI_NO_PTY` | If `1`, disables interactive PTY path for bash tool |
| `OMP_MCP_TIMEOUT_MS` | Overrides MCP client request timeout (ms) for every MCP server. `0` disables client-side timeouts (`AbortSignal` never fires). Invalid (negative or non-numeric) values are ignored with a warning and the per-server config or default (`30000`) is used. |
| `PI_PACKAGE_DIR` | Overrides package asset base dir resolution (`docs/`, `examples/`, `CHANGELOG.md`) |
| `PI_DISABLE_LSPMUX` | If `1`, disables lspmux detection/integration and forces direct LSP server spawning |
| `PI_RPC_EMIT_TITLE` | Boolean-like flag enabling title events in RPC mode |
| `SMITHERY_URL` | Smithery web URL override (default `https://smithery.ai`) |
| `SMITHERY_API_URL` | Smithery API base URL override (default `https://api.smithery.ai`) |
| `SMITHERY_API_KEY` | Smithery API key for managed MCP auth lookup |
| `PUPPETEER_EXECUTABLE_PATH` | Browser tool Chromium executable override |
| `LM_STUDIO_BASE_URL` | Default implicit LM Studio discovery base URL override (`http://127.0.0.1:1234/v1` if unset) |
| `OLLAMA_BASE_URL` | Default implicit Ollama discovery base URL override (`http://127.0.0.1:11434` if unset) |
| `LLAMA_CPP_BASE_URL` | Default implicit Llama.cpp discovery base URL override (`http://127.0.0.1:8080` if unset) |
| `PI_EDIT_VARIANT` | Forces edit tool variant when valid (`patch`, `replace`, `hashline`, `apply_patch`) |
| `PI_FORCE_IMAGE_PROTOCOL` | Forces supported image protocol (`kitty`, `iterm2`/`iterm`, `sixel`, `none`) where used |
| `PI_ALLOW_SIXEL_PASSTHROUGH` | Allows SIXEL passthrough when `PI_FORCE_IMAGE_PROTOCOL=sixel` |
| `PI_NO_PTY` | If `1`, disables interactive PTY path for bash tool |
| `OMP_MCP_TIMEOUT_MS` | Overrides MCP client request timeout (ms) for every MCP server. `0` disables client-side timeouts (`AbortSignal` never fires). Invalid (negative or non-numeric) values are ignored with a warning and the per-server config or default (`30000`) is used. |
`PI_NO_PTY` is also set internally when CLI `--no-pty` is used.
@@ -366,6 +387,7 @@ These are read as runtime signals; they are usually set by the terminal/OS rathe
| `PI_TUI_WRITE_LOG` | If set, logs TUI writes to file |
| `PI_HARDWARE_CURSOR` | If `1`, enables hardware cursor mode |
| `PI_CLEAR_ON_SHRINK` | If `1`, clears empty rows when content shrinks |
| `PI_NO_SYNC_OUTPUT` | If `1`, disables DEC 2026 synchronized-output wrappers while keeping TUI autowrap guards |
| `PI_DEBUG_REDRAW` | If `1`, enables redraw debug logging |
| `PI_TUI_DEBUG` | If `1`, enables deep TUI debug dump path |
| `PI_FORCE_IMAGE_PROTOCOL` | Forces terminal image protocol detection (`kitty`, `iterm2`/`iterm`, `sixel`, `none`) |
+12 -9
View File
@@ -28,17 +28,18 @@ Extension loading builds a list of module entry files, imports each module with
`discoverAndLoadExtensions()` first asks discovery providers for `extension-module` capability items, then keeps only provider `native` items.
Effective native locations:
Native `extension-module` discovery comes from:
- Project: `<cwd>/.omp/extensions`
- User: `~/.omp/agent/extensions`
- Project directory: `<cwd>/.omp/extensions`
- User directory: `~/.omp/agent/extensions`
- Native legacy/settings JSON entries: `<cwd>/.omp/settings.json#extensions` and `~/.omp/agent/settings.json#extensions`
Path roots come from the native provider (`SOURCE_PATHS.native`).
Path roots come from the native provider (`SOURCE_PATHS.native`). Project lookup is cwd-only for these native roots; it does not walk ancestors.
Notes:
- Native auto-discovery is currently `.omp` based.
- Legacy `.pi` is still accepted in `package.json` manifest keys (`pi.extensions`), but not as a native root here.
- Legacy `.pi` is still accepted in package manifests (`pi.extensions`) and project override lookup, but `.pi/extensions` is not a native root here.
### 2) Installed plugin extension entries
@@ -53,14 +54,16 @@ After plugin extension entries, configured paths are appended and resolved.
Configured path sources in the main session startup path (`sdk.ts`):
1. CLI-provided paths (`--extension/-e`, and `--hook` is also treated as an extension path)
2. Settings `extensions` array (merged global + project settings)
2. Merged settings `extensions` array
Global settings file:
Settings files:
- `~/.omp/agent/config.yml` (or custom agent dir via `PI_CODING_AGENT_DIR`)
- User: `~/.omp/agent/config.yml` (or custom agent dir via `PI_CODING_AGENT_DIR`)
- Project/native settings capability: `<cwd>/.omp/config.yml` and `<cwd>/.omp/settings.json`
Project settings file:
Native extension-module discovery also reads legacy JSON extension lists from:
- `~/.omp/agent/settings.json`
- `<cwd>/.omp/settings.json`
Examples:
+31 -5
View File
@@ -10,7 +10,7 @@ This document covers the current extension runtime in:
- `src/extensibility/extensions/index.ts`
- `src/modes/controllers/extension-ui-controller.ts`
For discovery paths and filesystem loading rules, see `docs/extension-loading.md`.
For discovery paths and filesystem loading rules, see [`extension-loading.md`](./extension-loading.md).
## What an extension is
@@ -112,9 +112,11 @@ Core methods:
- `on(event, handler)`
- `registerTool`, `registerCommand`, `registerShortcut`, `registerFlag`
- `registerMessageRenderer`
- `sendMessage`, `sendUserMessage`, `appendEntry`
- `registerMessageRenderer`, `registerAssistantThinkingRenderer`
- `setLabel`, `getFlag`
- `sendMessage`, `sendUserMessage`, `appendEntry`, `exec`
- `getActiveTools`, `getAllTools`, `setActiveTools`
- `getCommands`
- `getSessionName`, `setSessionName`
- `setModel`, `getThinkingLevel`, `setThinkingLevel`
- `registerProvider`
@@ -125,7 +127,8 @@ In interactive mode, `input` handlers run before the built-in first-message auto
Also exposed:
- `pi.logger`
- `pi.zod` (injected `zod` module — use for tool parameter schemas)
- `pi.typebox` (zod-backed compatibility shim for legacy TypeBox-style schemas)
- `pi.zod` (injected `zod/v4` module — canonical for tool parameter schemas)
- `pi.pi` (package exports)
### Message delivery semantics
@@ -191,6 +194,8 @@ Cancelable pre-events:
- `input`
- `before_agent_start`
- `before_provider_request` (may replace provider request payload)
- `after_provider_response`
- `context`
- `agent_start` / `agent_end`
- `turn_start` / `turn_end`
@@ -210,6 +215,8 @@ Cancelable pre-events:
- `auto_retry_start` / `auto_retry_end`
- `ttsr_triggered`
- `todo_reminder`
- `goal_updated`
- `credential_disabled`
### User command interception
@@ -247,6 +254,9 @@ pi.registerTool({
label: "My Tool",
description: "...",
parameters: z.object({}),
hidden: false,
defaultInactive: false,
deferrable: false,
async execute(_id, _params, signal, onUpdate, ctx) {
if (signal?.aborted) {
return { content: [{ type: "text", text: "Cancelled" }] };
@@ -266,7 +276,7 @@ pi.registerTool({
});
```
`tool_call`/`tool_result` intercept all tools once the registry is wrapped in `sdk.ts`, including built-ins and extension/custom tools.
`tool_call`/`tool_result` intercept all tools once the registry is wrapped in `sdk.ts`, including built-ins and extension/custom tools. `ToolDefinition` also supports optional `hidden`, `defaultInactive`, `deferrable`, `mcpServerName`, `mcpToolName`, `renderCall`, and `renderResult` fields.
## UI integration points
@@ -277,6 +287,8 @@ pi.registerTool({
Supported:
- dialogs: `select`, `confirm`, `input`, `editor`
- input editing: `setEditorText`, `getEditorText`, `pasteToEditor`, `editor`
- terminal title and working message (`setTitle`, `setWorkingMessage`)
- notifications/status/editor text/terminal input/custom overlays
- theme listing/loading by name (`setTheme` supports string names)
- tools expanded toggle
@@ -347,6 +359,20 @@ pi.registerMessageRenderer("my-type", (message, { expanded }, theme) => {
Used by interactive rendering when custom messages are displayed.
## Assistant thinking renderer
```ts
import { Container, Text } from "@oh-my-pi/pi-tui";
pi.registerAssistantThinkingRenderer((context, theme) => {
const container = new Container();
container.addChild(new Text(theme.fg("dim", `thinking chars: ${context.text.length}`), 1, 0));
return container;
});
```
Used by interactive rendering to add display-only supplemental UI below each visible assistant thinking block. The renderer receives the already-visible thinking text, content/thinking indexes, theme, and a `requestRender()` callback for async renderers. All registered renderers that return a component are appended in registration order. Renderers must not mutate messages; the original thinking block remains the provider/session source of truth.
## Tool call/result renderer
Provide `renderCall` / `renderResult` on `registerTool` definitions for custom tool visualization in TUI.
+38 -31
View File
@@ -4,12 +4,12 @@ This document defines the current contract for the shared filesystem scan cache
## What this cache is
The cache stores full directory-scan entry lists (`GlobMatch[]`) keyed by scan scope and traversal policy, then lets higher-level operations (glob filtering, fuzzy scoring, grep file selection) run against those cached entries.
The cache stores full directory-scan entry lists (`GlobMatch[]`) keyed by scan scope, traversal policy, and requested metadata detail. Higher-level operations (`glob` filtering, `fuzzyFind` scoring, and cached `grep` candidate selection) run against those cached entries.
Primary goals:
- avoid repeated filesystem walks for repeated discovery/search calls
- keep consistency across `glob`, `fuzzyFind`, and `grep` when they share the same scan policy
- keep consistency across native discovery/search flows when they share the same scan policy
- allow explicit staleness recovery for empty results and explicit invalidation after file mutations
## Ownership and public surface
@@ -18,11 +18,10 @@ Primary goals:
- Native consumers:
- `crates/pi-natives/src/glob.rs`
- `crates/pi-natives/src/fd.rs` (`fuzzyFind`)
- `crates/pi-natives/src/grep.rs`
- `crates/pi-natives/src/grep.rs` (cached directory mode only)
- JS binding/export:
- `packages/natives/src/glob/index.ts` (`invalidateFsScanCache`)
- `packages/natives/src/glob/types.ts`
- `packages/natives/src/grep/types.ts`
- `packages/natives/native/index.d.ts` (`invalidateFsScanCache`)
- `packages/natives/native/index.js`
- Coding-agent mutation invalidation helpers:
- `packages/coding-agent/src/tools/fs-cache-invalidation.ts`
@@ -34,25 +33,30 @@ Each entry is keyed by:
- `include_hidden` boolean
- `use_gitignore` boolean
- `skip_node_modules` boolean
- `detail` (`ScanDetail::Minimal` or `ScanDetail::Full`)
Implications:
- Hidden and non-hidden scans do **not** share entries.
- Gitignore-respecting and ignore-disabled scans do **not** share entries.
- Scans that prune `node_modules` do **not** share entries with scans that include it.
- Consumers must pass stable semantics for hidden/gitignore/node_modules behavior; changing any flag creates a different cache partition.
- Minimal scans (path + file type only) do **not** share entries with full scans (mtime + regular-file size metadata).
- `follow_links` is part of `ScanOptions` used to build the walker, but is not currently part of `CacheKey`; calls that differ only by `follow_links` can share a cache entry.
Consumers must pass stable semantics for hidden/gitignore/node_modules/detail behavior; changing any keyed flag creates a different cache partition.
## Scan collection behavior
Cache population uses a deterministic walker (`ignore::WalkBuilder`) configured by `include_hidden`, `use_gitignore`, and `skip_node_modules`:
Cache population uses `ignore::WalkBuilder` configured by `include_hidden`, `use_gitignore`, `skip_node_modules`, and `follow_links`:
- `follow_links(false)`
- sorted by file path
- `.git` is always skipped
- `.git` is always pruned
- `node_modules` is pruned at traversal time when `skip_node_modules=true`
- entry file type + `mtime` are captured via `symlink_metadata`
- cancellation is checked before the walk and every 128 visited entries per parallel visitor
- `ScanDetail::Minimal` records normalized relative path and file type only
- `ScanDetail::Full` also records mtime and regular-file size
Search roots are resolved by `resolve_search_path`:
Search roots for cache scans are resolved by `fs_cache::resolve_search_path`:
- relative paths are resolved against current cwd
- target must be an existing directory
@@ -70,9 +74,11 @@ Behavior:
- `get_or_scan(...)`
- if TTL is `0`: bypass cache entirely, always fresh scan (`cache_age_ms = 0`)
- on cache hit within TTL: return cached entries + non-zero `cache_age_ms`
- on cache hit within TTL: return cloned cached entries + non-zero `cache_age_ms`
- on expired hit: evict key, rescan, store fresh entry
- max entry enforcement is oldest-first eviction by `created_at`
- `force_rescan(..., store=false)`: remove any matching key, scan fresh, and do not repopulate cache
- `force_rescan(..., store=true)`: remove any matching key, scan fresh, then store the new entry
- max entry enforcement is oldest-first eviction by `created_at` after insert
## Empty-result fast recheck (separate from normal hits)
@@ -83,46 +89,45 @@ Normal cache hit:
Empty-result fast recheck:
- this is a **caller-side** policy using `ScanResult.cache_age_ms`
- if filtered/query result is empty and cached scan age is at least `empty_recheck_ms()`, caller performs one `force_rescan(...)` and retries
- intended to reduce stale-negative results when files were recently added but cache is still within TTL
- if filtered/query result is empty and cached scan age is at least `empty_recheck_ms()`, caller performs one `force_rescan(..., store=true)` and retries
- intended to reduce stale-negative results when files were added while the cache is still inside TTL
Current consumers:
- `glob`: rechecks when filtered matches are empty and scan age exceeds threshold
- `fuzzyFind` (`fd.rs`): rechecks only when query is non-empty and scored matches are empty
- `grep`: rechecks when selected candidate file list is empty
- `grep`: rechecks when cached directory candidate file list is empty
## Consumer defaults and cache usage
Cache is opt-in on all exposed APIs (`cache?: boolean`, default `false`).
Cache is opt-in on exposed scan/search APIs (`cache?: boolean`, default `false`).
Current defaults in native APIs:
- `glob`: `hidden=false`, `gitignore=true`, `cache=false`, and `node_modules` included only when the pattern mentions `node_modules`
- `fuzzyFind`: `hidden=false`, `gitignore=true`, `cache=false`, and `node_modules` is skipped
- `grep`: `hidden=true`, `gitignore=true`, `cache=false`, and `node_modules` included only when the glob mentions `node_modules`
- `glob`: `hidden=false`, `gitignore=true`, `cache=false`; `node_modules` is included only when `includeNodeModules=true` or the pattern mentions `node_modules`; full detail is used only when `sortByMtime=true`
- `fuzzyFind`: `hidden=false`, `gitignore=true`, `cache=false`, `node_modules` is skipped, `follow_links=true`, minimal detail
- `grep`: `hidden=true`, `gitignore=true`, `cache=false`; cached directory mode skips `node_modules` unless the glob mentions `node_modules`; minimal detail
Coding-agent callers today:
- High-volume mention candidate discovery enables cache:
- `packages/coding-agent/src/utils/file-mentions.ts`
- profile: `hidden=true`, `gitignore=true`, `includeNodeModules=true`, `cache=true`
- Tool-level `grep` integration currently disables scan cache (`cache: false`):
- `packages/coding-agent/src/tools/grep.ts`
- Mutation flows invalidate through `packages/coding-agent/src/tools/fs-cache-invalidation.ts`.
- Tool-level search integration (`packages/coding-agent/src/tools/search.ts`) currently calls native `grep` with `cache: false`.
## Invalidation contract
Native invalidation entrypoint:
- `invalidateFsScanCache(path?: string)`
- with `path`: remove cache entries whose root is a prefix of target path
- with `path`: remove cache entries whose root is a prefix of the target path
- without path: clear all scan cache entries
Path handling details:
- relative invalidation paths are resolved against cwd
- invalidation attempts canonicalization
- if target does not exist (e.g., delete), fallback canonicalizes parent and reattaches filename when possible
- if target does not exist (for example after delete), fallback canonicalizes the parent and reattaches the filename when possible
- this preserves invalidation behavior for create/delete/rename where one side may not exist
## Coding-agent mutation flow responsibilities
@@ -135,10 +140,12 @@ Central helpers:
- `invalidateFsScanAfterDelete(path)`
- `invalidateFsScanAfterRename(oldPath, newPath)` (invalidates both sides when paths differ)
Current mutation tool callsites:
Current mutation callsites include:
- `packages/coding-agent/src/tools/write.ts`
- `packages/coding-agent/src/patch/index.ts` (hashline/patch/replace flows)
- `packages/coding-agent/src/edit/hashline/filesystem.ts`
- `packages/coding-agent/src/edit/modes/patch.ts`
- `packages/coding-agent/src/edit/modes/replace.ts`
Rule: if a flow mutates filesystem content or location and bypasses these helpers, cache staleness bugs are expected.
@@ -147,7 +154,7 @@ Rule: if a flow mutates filesystem content or location and bypasses these helper
When introducing cache use in a new scanner/search path:
1. **Use stable scan policy inputs**
- decide hidden/gitignore/node_modules semantics first
- decide hidden/gitignore/node_modules/detail semantics first
- pass them consistently to `get_or_scan`/`force_rescan` so cache partitions are intentional
2. **Treat cache data as pre-filtered only by traversal policy**
@@ -160,7 +167,7 @@ When introducing cache use in a new scanner/search path:
- keep this path separate from normal cache-hit logic
4. **Respect no-cache mode explicitly**
- when caller disables cache, call `force_rescan(..., store=false, ...)`
- when caller disables cache, call `force_rescan(..., store=false, ...)` or use an uncached streaming walker
- do not populate shared cache in a no-cache request path
5. **Wire mutation invalidation for any new write path**
@@ -174,5 +181,5 @@ When introducing cache use in a new scanner/search path:
- Cache scope is process-local in-memory (`DashMap`), not persisted across process restarts.
- Cache stores scan entries, not final tool results.
- `glob`/`fuzzyFind`/`grep` share scan entries only when key dimensions (`root`, `hidden`, `gitignore`, `skip_node_modules`) match.
- `glob`/`fuzzyFind`/cached `grep` share scan entries only when key dimensions (`root`, `hidden`, `gitignore`, `skip_node_modules`, `detail`) match.
- `.git` is always excluded at scan collection time regardless of caller options.
+7 -7
View File
@@ -6,12 +6,12 @@ It does **not** cover TypeScript/JavaScript extension module loading (`extension
## Implementation files
- [`../src/discovery/gemini.ts`](../packages/coding-agent/src/discovery/gemini.ts)
- [`../src/discovery/builtin.ts`](../packages/coding-agent/src/discovery/builtin.ts)
- [`../src/discovery/helpers.ts`](../packages/coding-agent/src/discovery/helpers.ts)
- [`../src/capability/extension.ts`](../packages/coding-agent/src/capability/extension.ts)
- [`../src/capability/index.ts`](../packages/coding-agent/src/capability/index.ts)
- [`../src/extensibility/extensions/loader.ts`](../packages/coding-agent/src/extensibility/extensions/loader.ts)
- [`packages/coding-agent/src/discovery/gemini.ts`](../packages/coding-agent/src/discovery/gemini.ts)
- [`packages/coding-agent/src/discovery/builtin.ts`](../packages/coding-agent/src/discovery/builtin.ts)
- [`packages/coding-agent/src/discovery/helpers.ts`](../packages/coding-agent/src/discovery/helpers.ts)
- [`packages/coding-agent/src/capability/extension.ts`](../packages/coding-agent/src/capability/extension.ts)
- [`packages/coding-agent/src/capability/index.ts`](../packages/coding-agent/src/capability/index.ts)
- [`packages/coding-agent/src/extensibility/extensions/loader.ts`](../packages/coding-agent/src/extensibility/extensions/loader.ts)
---
@@ -169,7 +169,7 @@ For Gemini manifests specifically:
`gemini-extension.json` discovery currently feeds capability metadata (`Extension` items). It does **not** directly load runnable TS/JS extension modules.
Runtime module loading (`discoverAndLoadExtensions()` / `loadExtensions()`) uses `extension-modules` and explicit paths, and currently filters auto-discovered modules to provider `native` only.
Runtime module loading (`discoverAndLoadExtensions()` / `loadExtensions()`) uses the `extension-module` capability and explicit paths, and currently filters auto-discovered modules to provider `native` only.
Practical implication:
+11 -5
View File
@@ -54,7 +54,7 @@ The same minimum-content guard exists again inside `AgentSession.handoff()` and
- the live tool array (`agent.state.tools`),
- optional focus instructions,
- coding-agent message conversion (`convertToLlm`),
- provider metadata and `initiatorOverride: "agent"`.
- provider metadata, current thinking level, and `initiatorOverride: "agent"`.
`generateHandoff(...)` lives in `packages/agent/src/compaction/compaction.ts` next to summarization. It renders `packages/agent/src/compaction/prompts/handoff-document.md` via `renderHandoffPrompt(...)` with optional `additionalFocus`.
@@ -75,7 +75,7 @@ await completeSimple(
{
apiKey,
signal,
reasoning: Effort.High,
reasoning: resolveCompactionEffort(model, options.thinkingLevel),
toolChoice: "none",
initiatorOverride,
metadata,
@@ -113,7 +113,7 @@ If text was generated and not aborted:
3. Start a brand-new session with `parentSession` pointing at the previous session file when one exists.
4. Reset in-memory agent state (`agent.reset()`).
5. Rebind `agent.sessionId` to the new session id.
6. Rekey/reset hindsight state for the new session.
6. Rekey/reset Hindsight and Mnemopi memory session tracking for the new session.
7. Clear queued context arrays (`#steeringMessages`, `#followUpMessages`, `#pendingNextTurnMessages`) and any scheduled hidden next-turn generation.
8. Reset todo reminder counter.
@@ -132,7 +132,13 @@ The above is a handoff document from a previous session. Use this context to con
Insertion call:
```ts
this.sessionManager.appendCustomMessageEntry("handoff", handoffContent, true, undefined, "agent");
this.sessionManager.appendCustomMessageEntry(
"handoff",
handoffContent,
true,
undefined,
"agent",
);
```
Semantics:
@@ -233,7 +239,7 @@ High-level state flow:
1. Interactive slash command intercepted.
2. Preflight message-count guard.
3. `#handoffAbortController` created (`isGeneratingHandoff = true`).
4. `generateHandoff(...)` issues one `completeSimple(...)` request with live system prompt, tools, message history, and trailing handoff prompt.
4. `generateHandoff(...)` issues one `completeSimple(...)` request with live system prompt, tools, message history, current thinking level, and trailing handoff prompt.
5. Assistant response text blocks are joined; tool-call blocks are discarded.
6. If missing text → return `undefined`; if aborted → cancellation error path.
7. If present:
+2 -1
View File
@@ -47,6 +47,7 @@ The factory can:
- register slash commands via `pi.registerCommand(...)`
- register custom message renderers via `pi.registerMessageRenderer(...)`
- run shell commands via `pi.exec(...)`
- author schemas/helpers with injected `pi.zod`, `pi.typebox`, and package exports via `pi.pi`
## Discovery and loading
@@ -218,7 +219,7 @@ Command/renderer conflicts:
- `setEditorText`, `getEditorText`
- `theme` getter
`ctx.hasUI` indicates whether interactive UI is available.
`ctx` includes `hasUI`, `cwd`, `sessionManager`, `modelRegistry`, current `model`, `isIdle()`, `abort()`, and `hasQueuedMessages()`.
When running with no UI, the default no-op context behavior is:
+5 -5
View File
@@ -6,12 +6,12 @@ A persistent per-install UUID that identifies a single oh-my-pi installation acr
Exported from `@oh-my-pi/pi-utils` (`packages/utils/src/dirs.ts`):
| Symbol | Purpose |
| --- | --- |
| `getInstallId(): string` | Returns the install ID, generating and persisting one on first call. Result is cached in-process for the lifetime of the runtime. |
| `__resetInstallIdCacheForTests(): void` | Clears the in-process cache. Test-only — MUST NOT be called from production code. |
| Symbol | Purpose |
| --------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------- |
| `getInstallId(): string` | Returns the install ID, generating and persisting one on first call. Result is cached in-process for the lifetime of the runtime. |
| `__resetInstallIdCacheForTests(): void` | Clears the in-process cache. Test-only — MUST NOT be called from production code. |
The returned value is a canonical lowercase RFC 4122 UUID matching `^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$`.
Generated IDs are lowercase RFC 4122 UUIDs. Existing persisted values are accepted case-insensitively when they match `^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$` with the regex `i` flag, and are returned exactly as stored.
## Storage
+28 -29
View File
@@ -4,44 +4,43 @@ Run `/hotkeys` inside an `omp` session to see the active chords for your current
## Customize keybindings
User remaps live in `~/.omp/agent/keybindings.json`. The file is a JSON object whose keys are keybinding action IDs and whose values are either one chord string or an array of chord strings. It is not read from `~/.omp/agent/config.yml`, and there is no nested `keybindings` object.
User remaps live in `~/.omp/agent/keybindings.yml`. The file is a YAML mapping whose keys are keybinding action IDs and whose values are either one chord string or an array of chord strings. It is not read from `~/.omp/agent/config.yml`, and there is no nested `keybindings` object.
```json
{
"app.model.cycleForward": "Ctrl+P",
"app.model.selectTemporary": "Alt+P",
"app.plan.toggle": "Alt+Shift+P"
}
```yaml
app.model.cycleForward: Ctrl+P
app.model.selectTemporary: Alt+P
app.plan.toggle: Alt+Shift+P
```
Chord names are case-insensitive and use the same notation shown in the UI, such as `Ctrl+P`, `Alt+Shift+P`, `Shift+Enter`, and `Ctrl+Backspace`.
Set an action to an empty array to disable it:
```json
{
"app.stt.toggle": []
}
```yaml
app.stt.toggle: []
```
## Common action IDs
| Action ID | Default | Meaning |
| --- | --- | --- |
| `app.model.cycleForward` | `Ctrl+P` | Cycle role models forward |
| `app.model.cycleBackward` | `Shift+Ctrl+P` | Cycle role models backward |
| `app.model.selectTemporary` | `Alt+P` | Pick a model temporarily for this session |
| `app.model.select` | `Ctrl+L` | Open the model selector and set roles |
| `app.plan.toggle` | `Alt+Shift+P` | Toggle plan mode |
| `app.history.search` | `Ctrl+R` | Search prompt history |
| `app.tools.expand` | `Ctrl+O` | Toggle tool-output expansion |
| `app.thinking.toggle` | `Ctrl+T` | Toggle thinking-block visibility |
| `app.thinking.cycle` | `Shift+Tab` | Cycle thinking level |
| `app.editor.external` | `Ctrl+G` | Edit the draft in `$VISUAL` / `$EDITOR` |
| `app.message.followUp` | `Ctrl+Enter` | Queue a follow-up message |
| `app.message.dequeue` | `Alt+Up` | Dequeue a queued message back into the editor |
| `app.clipboard.copyLine` | `Alt+Shift+L` | Copy the current line |
| `app.clipboard.copyPrompt` | `Alt+Shift+C` | Copy the whole prompt |
| `app.stt.toggle` | `Alt+H` | Toggle speech-to-text recording |
| Action ID | Default | Meaning |
| --------------------------- | -------------------------------------- | --------------------------------------------- |
| `app.model.cycleForward` | `Ctrl+P` | Cycle role models forward |
| `app.model.cycleBackward` | `Shift+Ctrl+P` | Cycle role models in temporary mode |
| `app.model.selectTemporary` | `Alt+P` | Pick a model temporarily for this session |
| `app.model.select` | `Ctrl+L` | Open the model selector and set roles |
| `app.plan.toggle` | `Alt+Shift+P` | Toggle plan mode |
| `app.history.search` | `Ctrl+R` | Search prompt history |
| `app.tools.expand` | `Ctrl+O` | Toggle tool-output expansion |
| `app.thinking.toggle` | `Ctrl+T` | Toggle thinking-block visibility |
| `app.thinking.cycle` | `Shift+Tab` | Cycle thinking level |
| `app.editor.external` | `Ctrl+G` | Edit the draft in `$VISUAL` / `$EDITOR` |
| `app.message.followUp` | `Ctrl+Enter` | Queue a follow-up message |
| `app.message.dequeue` | `Alt+Up` | Dequeue a queued message back into the editor |
| `app.clipboard.copyLine` | `Alt+Shift+L` | Copy the current line |
| `app.clipboard.copyPrompt` | `Alt+Shift+C` | Copy the whole prompt |
| `app.clipboard.pasteImage` | `Ctrl+V` (`Alt+V` fallback on Windows) | Paste an image from the clipboard |
| `app.stt.toggle` | `Alt+H` | Toggle speech-to-text recording |
Older unqualified action names are migrated when `keybindings.json` is loaded, but new docs and new configs should use the namespaced action IDs above.
On Windows Terminal, `Ctrl+V` may be handled by the terminal paste command before `omp` sees it; use the `Alt+V` fallback when clipboard image paste appears to do nothing.
Older unqualified action names are migrated when `keybindings.yml` is loaded, but new docs and new configs should use the namespaced action IDs above. Existing `keybindings.json` files are still accepted and migrated to `keybindings.yml`; `keybindings.yaml` is also accepted.
+148
View File
@@ -0,0 +1,148 @@
# Embedded Local Tiny-Model Experiments
This document summarizes the experiments behind the optional **local** tiny-model paths for
session-title generation (`providers.tinyModel`), Mnemopi memory extraction/consolidation
(`providers.memoryModel`), and the `auto` thinking-level difficulty classifier
(`providers.autoThinkingModel`, which reuses the memory-model registry). It is a factual engineering
record for maintainers: what we measured, which recipes won, and which models we shipped. All three
settings default to `online`, so existing users incur no downloads or on-device inference cost unless
they opt in.
## Runtime / environment findings
- **Stack**: `@huggingface/transformers` (transformers.js) v4 running under Bun. In Bun the library
loads the **native `onnxruntime-node` backend** (not the WASM build).
- **Device policy**: local tiny models default to CPU-only inference and retry once on CPU if an
explicit accelerated provider cannot initialize.
- Pick a provider persistently with the `providers.tinyModelDevice` setting (`default` keeps CPU),
or per-run with the `PI_TINY_DEVICE` env var (which overrides the setting).
- Accepted values are `cpu`, `gpu`, `metal`/`webgpu`, `auto`, `cuda`, `dml`, `coreml`, `wasm`,
`webnn`, `webnn-gpu`, `webnn-cpu`, and `webnn-npu`.
- Direct `coreml` remains opt-in via `PI_TINY_DEVICE=coreml`; it is not part of the default because
cached decoder-LLM ONNX loads can fail during session initialization.
- WebGPU/Metal works for the single-process eval harness, but the production worker forces
Darwin `gpu`/`webgpu`/`auto` requests back to CPU because ONNX Runtime/Bun currently
hard-crashes on worker teardown after WebGPU inference.
- Use `providers.tinyModelDevice` or `PI_TINY_DEVICE` only when explicitly opting out of the CPU
default.
- **Quantization: q4 is the sweet spot** — smaller on disk, faster to load, and fast at inference.
q8/int8 loads slower _and_ infers slower on CPU. Every shipped model defaults to `q4`; override the
precision persistently with the `providers.tinyModelDtype` setting (`default` keeps `q4`, e.g. `fp16`
for higher fidelity), or per-run with `PI_TINY_DTYPE` (which overrides the setting). Accepts `auto`,
`fp32`, `fp16`, `q8`, `int8`, `uint8`, `q4`, `bnb4`, `q4f16`, `q2`, `q2f16`, `q1`, `q1f16`; an
unrecognized value fails loudly at worker startup.
- **Load-time correction (important).** An earlier belief that "q4 >=1B models take minutes to load"
was a **measurement artifact** caused by running ~5 multi-GB HuggingFace downloads in parallel
(I/O saturation). Clean, isolated **warm** loads are all sub-3s:
- TinyLlama-1.1B q4: ~0.5s
- Llama-3.2-1B q4: ~2.8s (`graphOpt=all`) / ~0.5s (`disabled`)
- LFM2-1.2B q4: ~0.36s
- Qwen2.5-1.5B q4: ~1.5s
- Qwen3-1.7B q4: ~1.6s
- gemma-3-1b q4: ~1.1s
- Conclusion: **1B–1.7B models are viable on CPU.**
- **`session_options.graphOptimizationLevel`** trades load vs inference speed: `disabled` = fastest
load, slightly slower inference; `all` = default.
- **First run** downloads weights from the HF Hub to a cache dir (q4 weights ~200MB–1.1GB depending
on model); subsequent **warm** loads are sub-second to ~3s. Inference is async and
background-friendly for memory tasks; titles are semi-interactive.
## Task 1: Session title generation (`providers.tinyModel`)
**Task**: turn the first user message into a 3–6 word title. Tiny models (sub-1B) suffice.
**Winning recipe**:
- Plain system prompt (no few-shot).
- **Prefill** the assistant turn with `<title>` and **stop at `</title>`**, then take the first line.
- Greedy decoding (`do_sample:false`), `enable_thinking:false` in the chat template.
**What we learned**:
- **Few-shot examples HURT sub-0.6B models** for titles; the tag-prefill rescues even 270M models.
- **Token biasing (`bad_words_ids`) is a confirmed no-op** here — the prefill already controls the
opener.
**Leaderboard** (tag trick, CPU, warm):
| Model | Verdict |
| ------------- | ----------------------------------- |
| LFM2-350M | Best speed/quality balance (~212MB) |
| Qwen3-0.6B | Most robust |
| gemma-3-270m | Smallest viable |
| Qwen2.5-0.5B | Acceptable |
| SmolLM2-135M | Too small |
| flan-t5-small | Rejected — just echoes the input |
**Shipped local options**: `lfm2-350m`, `qwen3-0.6b`, `gemma-270m`, `qwen2.5-0.5b`, `lfm2-700m`.
**Default**: `online` (pi/smol).
## Task 2: Mnemopi memory (`providers.memoryModel`)
Mnemopi runs two small-LLM tasks:
1. **Extraction** — pull durable, structured items from a single message.
2. **Consolidation** — summarize a list of memories into 1–3 faithful sentences.
These need **bigger models than titles: 1B–1.7B**. We tested LFM2-1.2B, Qwen2.5-1.5B, Qwen3-1.7B,
and gemma-3-1b (q4, CPU) via four parallel agents each running 27–31 experiments.
### Extraction findings
The stock 5-category JSON prompt fails on small models in two ways:
1. The all-empty example `{"facts":[],...}` gets **copied verbatim** → 0 facts extracted.
2. Capable models emit **JSON objects inside arrays**, which Mnemopi's `String(item)` coerces into
the literal string `[object Object]`.
The robust fix is a **one-item-per-line output format** (consumed by Mnemopi's parser line-fallback)
or a **flat JSON array of strings**. Every model also over-extracts pure small talk; an explicit
chit-chat → NONE example is the best mitigation.
### Technique polarity flips vs titles
- At 1B+, **few-shot is the dominant quality lever**: e.g. Qwen2.5-1.5B extraction F1 0.52 → 0.83
going 1 → 3 shots; gemma recall 0.65 → 0.92 with 2 shots.
- **Prefill HURTS extraction** — it forces output on small talk, producing false positives.
- **System-split** (instructions in the system role) helps models that have a system role.
- **Greedy >= temperature** for both tasks.
- **Token biasing** is again a no-op.
### Per-model verdicts (head-to-head, 16-fixture set)
- **Qwen3-1.7B** — most disciplined extraction: returns empty on small talk, no buried-fact leak,
preserves language, clean flat JSON. Weaknesses: coarse granularity, missed a multi-turn value
update.
- **Qwen2.5-1.5B** — best extraction granularity (atomic facts), caught the value update, zero
small-talk leakage. Weaknesses: weakest consolidation (run-on, no dedup) and one degenerate
buried-fact output.
- **gemma-3-1b** — best consolidation (dedup works, faithful, clean single-memory). Weaknesses: leaks
small talk and translated German.
- **LFM2-1.2B** — solid and fastest to load. Weaknesses: `Label: value` noise, small-talk + buried
leaks, a fluffy single-memory summary.
### Recommendation
Extraction favors **precision** (do not pollute long-term memory) → **Qwen3-1.7B is the best single
pick** (its consolidation is good enough). If running a second model for consolidation, **gemma-3-1b**
wins that task.
**Shipped local options**: `qwen3-1.7b` (recommended), `gemma-3-1b`, `qwen2.5-1.5b`, `lfm2-1.2b`.
**Default**: `online` (the configured smol model).
### Known Mnemopi parser bugs (surfaced by these experiments)
- `String(item)` produces `[object Object]` on object array items.
- The line-fallback drops items `<=10` chars, so a correct short fact like `Name: Can` is discarded.
## Integration notes
- `providers.tinyModel`, `providers.memoryModel`, and `providers.autoThinkingModel` default to
`online`, so existing users get **no downloads or on-device inference cost** unless they opt in.
- Local inference runs **in a worker** (off the main thread); models are cached on disk and
downloaded on first use.
- The memory local path applies the refined recipes (line-format + small-talk-guarded extraction
prompt, hardened consolidation prompt) via Mnemopi prompt overrides; the **online path is
unchanged**.
- `providers.autoThinkingModel` uses the same shipped local options as `providers.memoryModel`.
+75 -75
View File
@@ -21,22 +21,22 @@ No configuration is required for common setups. The built-in server list covers
OMP merges LSP config from multiple files, lowest to highest priority:
| Priority | Location |
|----------|----------|
| 5 (lowest) | `~/lsp.json`, `~/.lsp.json`, `~/lsp.yaml`, `~/.lsp.yaml` |
| 4 | Plugin LSP configs (marketplace / `--plugin-dir` roots) |
| 3 | `~/.omp/agent/lsp.json`, `~/.omp/agent/lsp.yaml`, `~/.claude/lsp.*` |
| 2 | `<project>/.omp/lsp.json`, `<project>/.omp/lsp.yaml`, `<project>/.claude/lsp.*` |
| 1 (highest) | `<project>/lsp.json`, `<project>/.lsp.json`, `<project>/lsp.yaml` |
| Priority | Location |
| ----------- | --------------------------------------------------------------------------------------------------------------------------- |
| 5 (lowest) | `~/lsp.json`, `~/.lsp.json`, `~/lsp.yaml`, `~/.lsp.yaml`, `~/lsp.yml`, `~/.lsp.yml` |
| 4 | Plugin LSP configs (marketplace / `--plugin-dir` roots) |
| 3 | User config dirs: `~/.omp/agent/lsp.*`, `~/.claude/lsp.*`, `~/.codex/lsp.*`, `~/.gemini/lsp.*` |
| 2 | Project config dirs: `<project>/.omp/lsp.*`, `<project>/.claude/lsp.*`, `<project>/.codex/lsp.*`, `<project>/.gemini/lsp.*` |
| 1 (highest) | Project root: `<project>/lsp.*` and `<project>/.lsp.*` |
Each location accepts both `.json` and `.yaml` / `.yml` variants, as well as hidden-file versions (`.lsp.json`, `.lsp.yaml`). Files are merged in order: higher-priority files override lower-priority fields for the same server. Servers not mentioned in any override file remain at their built-in defaults.
Each location accepts `.json`, `.yaml`, and `.yml` variants, including hidden-file versions (`.lsp.json`, `.lsp.yaml`, `.lsp.yml`). Files are merged in order: higher-priority files override lower-priority fields for the same server. Servers not mentioned in any override file remain at their built-in defaults.
**Recommended locations:**
- User-wide preferences → `~/.omp/agent/lsp.json`
- Project-specific overrides → `<project>/.omp/lsp.json`
> **Note:** The presence of any LSP config file disables auto-detection. When at least one file is found, OMP skips the binary-scan phase and loads all servers that have matching `rootMarkers`, an available binary, and are not explicitly `disabled`.
> **Note:** Auto-detection is skipped only when at least one config file contributes server overrides. A config file that only sets `idleTimeoutMs` still lets OMP auto-detect built-in servers. When server overrides exist, OMP merges them with defaults and then loads servers that have matching `rootMarkers`, an available binary, and are not explicitly `disabled`.
## File shape
@@ -67,18 +67,18 @@ Top-level keys:
## ServerConfig fields
| Field | Type | Required | Description |
|-------|------|----------|-------------|
| `command` | `string` | yes | Binary name (resolved via PATH/local bins) or absolute path |
| `args` | `string[]` | no | Arguments passed to the binary |
| `fileTypes` | `string[]` | yes | File extensions this server handles, e.g. `[".ts", ".tsx"]` |
| `rootMarkers` | `string[]` | yes | Files/dirs that indicate a project root; glob patterns (e.g. `*.cabal`) are supported |
| `initOptions` | `object` | no | Sent as `initializationOptions` during LSP handshake |
| `settings` | `object` | no | Workspace settings pushed via `workspace/didChangeConfiguration` |
| `disabled` | `boolean` | no | Set to `true` to disable this server entirely |
| `warmupTimeoutMs` | `number` | no | Startup timeout in ms for this server (overrides the global default) |
| `isLinter` | `boolean` | no | Mark server as linter/formatter only; excluded from type-intelligence operations (hover, go-to-definition, etc.) |
| `capabilities` | `object` | no | Opt-in server-specific features; see [Capabilities](#capabilities) |
| Field | Type | Required | Description |
| ----------------- | ---------- | -------- | ---------------------------------------------------------------------------------------------------------------- |
| `command` | `string` | yes | Binary name (resolved via PATH/local bins) or absolute path |
| `args` | `string[]` | no | Arguments passed to the binary |
| `fileTypes` | `string[]` | yes | File extensions this server handles, e.g. `[".ts", ".tsx"]` |
| `rootMarkers` | `string[]` | yes | Files/dirs that indicate a project root; glob patterns (e.g. `*.cabal`) are supported |
| `initOptions` | `object` | no | Sent as `initializationOptions` during LSP handshake |
| `settings` | `object` | no | Workspace settings pushed via `workspace/didChangeConfiguration` |
| `disabled` | `boolean` | no | Set to `true` to disable this server entirely |
| `warmupTimeoutMs` | `number` | no | Startup timeout in ms for this server (overrides the global default) |
| `isLinter` | `boolean` | no | Mark server as linter/formatter only; excluded from type-intelligence operations (hover, go-to-definition, etc.) |
| `capabilities` | `object` | no | Opt-in server-specific features; see [Capabilities](#capabilities) |
`resolvedCommand` is populated automatically at runtime — do not set it manually.
@@ -184,57 +184,57 @@ The user-level config in `~/.omp/agent/lsp.json` is unaffected; pylsp is only su
The following servers ship in `defaults.json` and are eligible for auto-detection:
| Server key | Language(s) | Binary |
|---|---|---|
| `rust-analyzer` | Rust | `rust-analyzer` |
| `clangd` | C, C++, ObjC | `clangd` |
| `zls` | Zig | `zls` |
| `gopls` | Go | `gopls` |
| `typescript-language-server` | TypeScript, JavaScript | `typescript-language-server` |
| `denols` | TypeScript, JavaScript (Deno) | `deno` |
| `biome` | TS/JS/JSON (linter) | `biome` |
| `eslint` | TS/JS/Vue/Svelte (linter) | `vscode-eslint-language-server` |
| `vscode-html-language-server` | HTML | `vscode-html-language-server` |
| `vscode-css-language-server` | CSS, SCSS, Less | `vscode-css-language-server` |
| `vscode-json-language-server` | JSON | `vscode-json-language-server` |
| `tailwindcss` | HTML, CSS, TS/JS | `tailwindcss-language-server` |
| `svelte` | Svelte | `svelteserver` |
| `vue-language-server` | Vue | `vue-language-server` |
| `astro` | Astro | `astro-ls` |
| `pyright` | Python | `pyright-langserver` |
| `basedpyright` | Python | `basedpyright-langserver` |
| `pylsp` | Python | `pylsp` |
| `ruff` | Python (linter) | `ruff` |
| `jdtls` | Java | `jdtls` |
| `kotlin-lsp` | Kotlin | `kotlin-lsp` |
| `metals` | Scala | `metals` |
| `hls` | Haskell | `haskell-language-server-wrapper` |
| `ocamllsp` | OCaml | `ocamllsp` |
| `elixirls` | Elixir | `elixir-ls` |
| `erlangls` | Erlang | `erlang_ls` |
| `gleam` | Gleam | `gleam` |
| `solargraph` | Ruby | `solargraph` |
| `ruby-lsp` | Ruby | `ruby-lsp` |
| `rubocop` | Ruby (linter) | `rubocop` |
| `bashls` | Bash, Zsh | `bash-language-server` |
| `lua-language-server` | Lua | `lua-language-server` |
| `intelephense` | PHP | `intelephense` |
| `phpactor` | PHP | `phpactor` |
| `omnisharp` | C# | `omnisharp` |
| `yamlls` | YAML | `yaml-language-server` |
| `terraformls` | Terraform | `terraform-ls` |
| `dockerls` | Dockerfile | `docker-langserver` |
| `helm-ls` | Helm | `helm_ls` |
| `nixd` | Nix | `nixd` |
| `nil` | Nix | `nil` |
| `ols` | Odin | `ols` |
| `dartls` | Dart | `dart` |
| `marksman` | Markdown | `marksman` |
| `texlab` | LaTeX | `texlab` |
| `graphql` | GraphQL | `graphql-lsp` |
| `prismals` | Prisma | `prisma-language-server` |
| `vimls` | Vim script | `vim-language-server` |
| `emmet-language-server` | HTML, CSS, JSX | `emmet-language-server` |
| `sourcekit-lsp` | Swift | `sourcekit-lsp` |
| `swiftlint` | Swift (linter) | `swiftlint` |
| `tlaplus` | TLA+ | `tlapm_lsp` |
| Server key | Language(s) | Binary |
| ----------------------------- | ----------------------------- | --------------------------------- |
| `rust-analyzer` | Rust | `rust-analyzer` |
| `clangd` | C, C++, ObjC | `clangd` |
| `zls` | Zig | `zls` |
| `gopls` | Go | `gopls` |
| `typescript-language-server` | TypeScript, JavaScript | `typescript-language-server` |
| `denols` | TypeScript, JavaScript (Deno) | `deno` |
| `biome` | TS/JS/JSON (linter) | `biome` |
| `eslint` | TS/JS/Vue/Svelte (linter) | `vscode-eslint-language-server` |
| `vscode-html-language-server` | HTML | `vscode-html-language-server` |
| `vscode-css-language-server` | CSS, SCSS, Less | `vscode-css-language-server` |
| `vscode-json-language-server` | JSON | `vscode-json-language-server` |
| `tailwindcss` | HTML, CSS, TS/JS | `tailwindcss-language-server` |
| `svelte` | Svelte | `svelteserver` |
| `vue-language-server` | Vue | `vue-language-server` |
| `astro` | Astro | `astro-ls` |
| `pyright` | Python | `pyright-langserver` |
| `basedpyright` | Python | `basedpyright-langserver` |
| `pylsp` | Python | `pylsp` |
| `ruff` | Python (linter) | `ruff` |
| `jdtls` | Java | `jdtls` |
| `kotlin-lsp` | Kotlin | `kotlin-lsp` |
| `metals` | Scala | `metals` |
| `hls` | Haskell | `haskell-language-server-wrapper` |
| `ocamllsp` | OCaml | `ocamllsp` |
| `elixirls` | Elixir | `elixir-ls` |
| `erlangls` | Erlang | `erlang_ls` |
| `gleam` | Gleam | `gleam` |
| `solargraph` | Ruby | `solargraph` |
| `ruby-lsp` | Ruby | `ruby-lsp` |
| `rubocop` | Ruby (linter) | `rubocop` |
| `bashls` | Bash, Zsh | `bash-language-server` |
| `lua-language-server` | Lua | `lua-language-server` |
| `intelephense` | PHP | `intelephense` |
| `phpactor` | PHP | `phpactor` |
| `omnisharp` | C# | `omnisharp` |
| `yamlls` | YAML | `yaml-language-server` |
| `terraformls` | Terraform | `terraform-ls` |
| `dockerls` | Dockerfile | `docker-langserver` |
| `helm-ls` | Helm | `helm_ls` |
| `nixd` | Nix | `nixd` |
| `nil` | Nix | `nil` |
| `ols` | Odin | `ols` |
| `dartls` | Dart | `dart` |
| `marksman` | Markdown | `marksman` |
| `texlab` | LaTeX | `texlab` |
| `graphql` | GraphQL | `graphql-lsp` |
| `prismals` | Prisma | `prisma-language-server` |
| `vimls` | Vim script | `vim-language-server` |
| `emmet-language-server` | HTML, CSS, JSX | `emmet-language-server` |
| `sourcekit-lsp` | Swift | `sourcekit-lsp` |
| `swiftlint` | Swift (linter) | `swiftlint` |
| `tlaplus` | TLA+ | `tlapm_lsp` |
+62 -45
View File
@@ -1,6 +1,6 @@
# Marketplace plugin system
The marketplace system lets you discover, install, and manage plugins from Git-hosted catalogs. It is compatible with the Claude Code plugin registry format.
The marketplace system lets you discover, install, and manage plugins from Git, local, or direct-catalog sources. It is compatible with the Claude Code plugin registry format.
## Quick start
@@ -9,20 +9,20 @@ The marketplace system lets you discover, install, and manage plugins from Git-h
/marketplace install wordpress.com@claude-plugins-official
```
Or just type `/marketplace` with no arguments to open the interactive plugin browser.
In the TUI, `/marketplace` with no arguments opens the interactive plugin browser. In non-TUI command handling, `/marketplace` lists configured marketplaces; use `/marketplace discover` to browse.
## Concepts
A **marketplace** is a Git repository (or local directory) containing a catalog file at `.omp-plugin/marketplace.json` (preferred) or `.claude-plugin/marketplace.json` (Claude Code-compatible fallback). The catalog lists available plugins with their sources, descriptions, and metadata.
A **plugin** is a directory containing skills, commands, hooks, MCP servers, or LSP servers. Plugins are identified by `name@marketplace` (e.g. `code-review@claude-plugins-official`).
A **plugin** is a directory containing Claude/OMP plugin content such as skills, commands, hooks, tools, MCP servers, LSP servers, rules, prompts, or extension modules. Plugins are identified by `name@marketplace` (e.g. `code-review@claude-plugins-official`).
**Scopes**: plugins can be installed at two scopes:
**Scopes**: marketplace plugins can be installed at two scopes:
- **user** (default) -- available in all projects, stored in `~/.omp/plugins/installed_plugins.json`
- **project** -- available only in the current project, stored in `.omp/plugins/installed_plugins.json`
- **project** -- available only in the active project, stored in the nearest project `.omp/plugins/installed_plugins.json`
Project-scoped installs shadow user-scoped installs of the same plugin.
Enabled project-scoped installs shadow enabled user-scoped installs of the same plugin. A disabled project install does not shadow the user install.
## Commands
@@ -43,13 +43,16 @@ Project-scoped installs shadow user-scoped installs of the same plugin.
### Plugin operations
| Command | Effect |
| ------------------------------------------------------------------------- | ---------------------------------- |
| `/marketplace discover [marketplace]` | Browse available plugins |
| `/marketplace install [--force] [--scope user\|project] name@marketplace` | Install a plugin |
| `/marketplace uninstall [--scope user\|project] name@marketplace` | Uninstall a plugin |
| `/marketplace installed` | List installed marketplace plugins |
| `/marketplace upgrade [--scope user\|project] [name@marketplace]` | Upgrade one or all plugins |
| Command | Effect |
| ------------------------------------------------------------------------- | -------------------------------------------------- |
| `/marketplace discover [marketplace]` | Browse available plugins |
| `/marketplace install [--force] [--scope user\|project] name@marketplace` | Install a plugin |
| `/marketplace uninstall [--scope user\|project] name@marketplace` | Uninstall a plugin; no args opens the TUI selector |
| `/marketplace installed` | List installed marketplace plugins |
| `/marketplace upgrade [--scope user\|project] [name@marketplace]` | Upgrade one or all plugins |
| `/plugins list` | List npm/link and marketplace plugins |
| `/plugins enable [--scope user\|project] name@marketplace` | Enable a marketplace plugin |
| `/plugins disable [--scope user\|project] name@marketplace` | Disable a marketplace plugin |
### CLI equivalents
@@ -64,20 +67,23 @@ omp plugin discover [marketplace]
omp plugin install [--force] [--scope user|project] name@marketplace
omp plugin uninstall [--scope user|project] name@marketplace
omp plugin upgrade [--scope user|project] [name@marketplace]
omp plugin enable [--scope user|project] name@marketplace
omp plugin disable [--scope user|project] name@marketplace
```
## Marketplace sources
When you run `/marketplace add <source>`, the system classifies the source:
| Source format | Type | Example |
| ------------------------------- | ------------------ | -------------------------------------- |
| `owner/repo` | GitHub shorthand | `anthropics/claude-plugins-official` |
| `https://...*.json` | Direct catalog URL | `https://example.com/marketplace.json` |
| `https://...*.git` or `git@...` | Git repository | `https://github.com/org/repo.git` |
| `./path` or `~/path` or `/path` | Local directory | `./my-marketplace` |
| Source format | Type | Example |
| ------------------------------- | -------------------------------------------------- | -------------------------------------- |
| `owner/repo` | GitHub shorthand | `anthropics/claude-plugins-official` |
| `https://...*.json` | Direct catalog URL | `https://example.com/marketplace.json` |
| `https://...` / `http://...` | Git repository unless the URL path ends in `.json` | `https://github.com/org/repo` |
| `git@...` / `ssh://...` | Git repository | `git@github.com:org/repo.git` |
| `./path` or `~/path` or `/path` | Local directory | `./my-marketplace` |
The system clones the repository (or reads the local directory), locates the catalog (`.omp-plugin/marketplace.json` if present, otherwise `.claude-plugin/marketplace.json`), validates it, and caches the catalog locally.
Git and local sources must contain a catalog at `.omp-plugin/marketplace.json` (preferred) or `.claude-plugin/marketplace.json` (Claude Code-compatible fallback). Direct catalog URLs cache only the JSON catalog; plugins in URL-sourced catalogs cannot use relative string sources like `"./plugins/foo"`.
## Catalog format (marketplace.json)
@@ -91,12 +97,16 @@ A marketplace catalog lives at `.omp-plugin/marketplace.json` in the repository
"name": "Your Name",
"email": "you@example.com"
},
"description": "A collection of plugins",
"metadata": {
"description": "A collection of plugins",
"version": "1.0.0",
"pluginRoot": "plugins"
},
"plugins": [
{
"name": "my-plugin",
"description": "What this plugin does",
"source": "./plugins/my-plugin",
"source": "./my-plugin",
"category": "development",
"homepage": "https://github.com/you/my-plugin"
}
@@ -112,33 +122,38 @@ A marketplace catalog lives at `.omp-plugin/marketplace.json` in the repository
| `owner.name` | Marketplace owner name |
| `plugins` | Array of plugin entries |
Top-level `metadata.description`, `metadata.version`, and `metadata.pluginRoot` are optional. When `metadata.pluginRoot` is set, it is prepended to relative plugin `source` paths.
### Plugin entry fields
| Field | Required | Description |
| ------------- | -------- | ---------------------------------------------------------------- |
| `name` | yes | Plugin name (same rules as marketplace name) |
| `source` | yes | Where to find the plugin (see below) |
| `description` | no | Short description |
| `version` | no | Version string |
| `author` | no | `{ name, email? }` |
| `homepage` | no | URL |
| `category` | no | Category string (e.g. `development`, `productivity`, `security`) |
| `tags` | no | Array of string tags |
| `strict` | no | Boolean |
| `commands` | no | Slash commands provided |
| `agents` | no | Agents provided |
| `hooks` | no | Hook definitions |
| `mcpServers` | no | MCP server definitions |
| `lspServers` | no | LSP server definitions |
| Field | Required | Description |
| ------------- | -------- | --------------------------------------------------------------------------------------- |
| `name` | yes | Plugin name (same rules as marketplace name) |
| `source` | yes | Where to find the plugin (see below) |
| `description` | no | Short description |
| `version` | no | Version string; install version falls back to plugin manifest, source SHA, then `0.0.0` |
| `author` | no | `{ name, email? }` |
| `homepage` | no | URL |
| `repository` | no | Repository URL/string |
| `license` | no | License string |
| `keywords` | no | Array of string keywords |
| `category` | no | Category string (e.g. `development`, `productivity`, `security`) |
| `tags` | no | Array of string tags |
| `strict` | no | Boolean |
| `commands` | no | Slash commands provided |
| `agents` | no | Agents provided |
| `hooks` | no | Hook definitions |
| `mcpServers` | no | MCP server definitions |
| `lspServers` | no | LSP server definitions or path; copied to `.lsp.json` on install |
### Plugin source formats
The `source` field supports several formats:
The `source` field supports these formats. String sources must start with `./` and are resolved inside the marketplace root, after optional `metadata.pluginRoot` is prepended:
**Relative path** (within the marketplace repo):
```json
"source": "./plugins/my-plugin"
"source": "./my-plugin"
```
**Git repository URL**:
@@ -174,7 +189,7 @@ The `source` field supports several formats:
}
```
**npm package**:
**npm package** (parsed but not installable yet):
```json
"source": {
@@ -184,20 +199,22 @@ The `source` field supports several formats:
}
```
Current installer behavior rejects npm marketplace sources with `npm plugin sources are not yet supported`; use relative, GitHub, URL, or git-subdir sources.
## On-disk layout
```
~/.omp/
marketplaces.json # Registry of added marketplaces
plugins/
installed_plugins.json # User-scoped installed plugins
installed_plugins.json # User-scoped marketplace plugins (version: 2)
cache/
marketplaces/ # Cached marketplace catalogs
plugins/ # Cached plugin directories
marketplaces/<name>/ # Cached marketplace clone/catalog
plugins/<marketplace>___<plugin>___<version>/ # Cached plugin directories
<project>/.omp/
plugins/
installed_plugins.json # Project-scoped installed plugins
installed_plugins.json # Project-scoped marketplace plugins (version: 2)
```
## Naming rules
+24 -24
View File
@@ -12,11 +12,13 @@ Source of truth in code:
## Preferred config locations
OMP can discover MCP servers from multiple tools (`.claude/`, `.cursor/`, `.vscode/`, `opencode.json`, and more), but for OMP-native configuration you should usually use one of these files:
OMP can discover MCP servers from multiple tools (`.claude/`, `.cursor/`, `.vscode/`, `opencode.json`, and more), but for OMP-native configuration you should usually use one of these primary files:
- Project: `.omp/mcp.json`
- User: `~/.omp/agent/mcp.json`
The native provider also reads `.omp/.mcp.json` and `~/.omp/agent/.mcp.json` for compatibility, but OMP writes to the primary `mcp.json` paths above.
OMP also accepts fallback standalone files in the project root:
- `mcp.json`
@@ -317,7 +319,27 @@ This matches GitHub's official local Docker image `ghcr.io/github/github-mcp-ser
This is the part that usually trips people up.
### In `.omp/mcp.json` and `~/.omp/agent/mcp.json`
### Discovery-time `${...}` expansion
OMP expands `${VAR}` and `${VAR:-default}` placeholders while discovering MCP configs from OMP-native files and standalone fallback files. Expansion applies recursively to string values in `command`, `args`, `env`, `cwd`, `url`, `headers`, `auth`, and `oauth`; unresolved placeholders remain literal strings.
Example:
```json
{
"mcpServers": {
"github": {
"type": "http",
"url": "https://api.githubcopilot.com/mcp/",
"headers": {
"Authorization": "Bearer ${GITHUB_TOKEN}"
}
}
}
}
```
### Pre-connect env/header resolution
Before OMP launches a stdio server or makes an HTTP/SSE request, it resolves stdio `env` values and HTTP/SSE `headers` values like this:
@@ -345,28 +367,6 @@ That means this is valid and convenient for local secrets:
- `"Authorization": "Bearer hardcoded-token"` → use the literal value
- `"Authorization": "!printf 'Bearer %s' \"$GITHUB_TOKEN\""` → build the header from a command
### In root `mcp.json` and `.mcp.json`
The standalone fallback loader also expands `${VAR}` and `${VAR:-default}` inside strings during discovery for `command`, `args`, `env`, `cwd`, `url`, `headers`, `auth`, and `oauth`.
Example:
```json
{
"mcpServers": {
"github": {
"type": "http",
"url": "https://api.githubcopilot.com/mcp/",
"headers": {
"Authorization": "Bearer ${GITHUB_TOKEN}"
}
}
}
}
```
If you want the least surprising OMP behavior, prefer `.omp/mcp.json` or `~/.omp/agent/mcp.json` and use explicit env/header values.
## `disabledServers`
`disabledServers` is read from the user config file (`~/.omp/agent/mcp.json`) when a server is discovered from any source and you want OMP to ignore it without editing that other tool's config.
+4 -3
View File
@@ -185,7 +185,7 @@ For `notify()`:
- timeout uses an internal `AbortController` (`config.timeout ?? 30000`)
- there is no external abort option on the transport interface
For HTTP OAuth configs managed by `MCPManager`, `request()` retries once on `HTTP 401`/`403` if token refresh returns replacement headers.
For HTTP OAuth configs managed by `MCPManager`, outbound requests and best-effort server-request responses retry once on `HTTP 401`/`403` if token refresh returns replacement headers.
## HTTP error propagation
@@ -212,14 +212,15 @@ Two SSE paths exist:
2. **Background SSE listener** (`startSSEListener()`)
- optional GET listener for server-initiated notifications and server-to-client requests
- `connectToServer()` starts it for HTTP/SSE transports after `initialize` and before `notifications/initialized`
- if GET returns `405`, another non-OK status, or no body, listener silently disables itself
- listener startup waits up to one second, or less for very small request timeouts; `timeout: 0` / `OMP_MCP_TIMEOUT_MS=0` disables that startup deadline
- if GET returns `405`, another non-OK status, no body, or times out, listener silently disables itself
## Malformed payload and disconnect handling
SSE JSON parsing errors bubble out of `readSseJson` and reject request/listener.
- Request SSE parse errors reject the active request.
- Background listener errors trigger `onError` (except AbortError).
- Background listener errors trigger `onError` (except AbortError), and an established listener ending while still connected triggers `onClose` so the manager can reconnect.
- Transport does not restart the listener itself; managed connections may reconnect through manager `onClose` handling.
## `json-rpc.ts` utility vs transport abstraction
+4 -4
View File
@@ -5,7 +5,7 @@ This document describes how MCP servers are discovered, connected, exposed as to
## Lifecycle at a glance
1. **SDK startup** calls `discoverAndLoadMCPTools()` (unless MCP is disabled).
2. **Discovery** (`loadAllMCPConfigs`) resolves MCP server configs from capability sources, filters disabled/project/Exa entries, and preserves source metadata.
2. **Discovery** (`loadAllMCPConfigs`) resolves MCP server configs from capability sources, filters disabled/project/Exa entries and browser MCP servers when the built-in browser tool is enabled, and preserves source metadata.
3. **Manager connect phase** (`MCPManager.connectServers`) starts per-server connect + `tools/list` in parallel.
4. **Fast startup gate** waits up to 250ms, then may return:
- fully loaded `MCPTool`s,
@@ -23,7 +23,7 @@ This document describes how MCP servers are discovered, connected, exposed as to
`createAgentSession()` in `src/sdk.ts` performs MCP startup when `enableMCP` is true (default):
- calls `discoverAndLoadMCPTools(cwd, { ... })`,
- passes `authStorage`, cache storage, and `mcp.enableProjectConfig` setting,
- passes `authStorage`, cache storage, `mcp.enableProjectConfig`, and browser-MCP filtering based on the `browser.enabled` setting,
- always sets `filterExa: true`,
- logs per-server load/connect errors,
- stores returned manager in `toolSession.mcpManager` and session result.
@@ -38,7 +38,7 @@ Filtering behavior:
- `enableProjectConfig: false` removes project-level entries (`_source.level === "project"`).
- `enabled: false` servers are skipped before connect attempts.
- Exa servers are filtered out by default and API keys are extracted for native Exa tool integration.
- Exa servers are filtered out by default and API keys are extracted for native Exa tool integration; browser automation MCP servers are filtered when `filterBrowser` is true.
Result includes both `configs` and `sources` (metadata used later for provider labeling).
@@ -180,7 +180,7 @@ Operationally:
- removes pending entries, source metadata, saved config, resource refresh/subscription state,
- detaches `onClose` so explicit close does not trigger reconnect,
- closes transport if connected,
- filters manager tool state for names beginning with `mcp__${name}_`.
- removes manager tool entries using the current raw-name prefix filter (`mcp__${name}_`); generated tool names are sanitized by `tool-bridge.ts`.
### Global teardown
+5 -6
View File
@@ -74,17 +74,16 @@ In practice MCP servers also come from higher-priority providers (for example na
Key behavior:
- transport inferred as `server.transport ?? (command ? "stdio" : url ? "http" : "stdio")`
- disabled servers (`enabled === false`) are dropped before connection
- disabled servers (`enabled === false`) and names in the user `disabledServers` list are dropped before connection
- optional fields are preserved when present
### Environment expansion during discovery
`mcp-json.ts` expands env placeholders in string fields with `expandEnvVarsDeep()`:
OMP-native MCP config (`.omp/mcp.json`, `~/.omp/agent/mcp.json`, plus their `.mcp.json` variants) expands `${VAR}` and `${VAR:-default}` placeholders recursively before converting to runtime config. It also accepts boolean/string forms for `enabled` (`true`, `false`, `1`, `0`) and numeric strings for `timeout`.
- supports `${VAR}` and `${VAR:-default}`
- unresolved values remain literal `${VAR}` strings
The standalone fallback provider in `src/discovery/mcp-json.ts` reads project-root `mcp.json` and `.mcp.json`, expands the same `${...}` placeholders, and type-checks `enabled`/`timeout` without coercing string values.
`mcp-json.ts` also performs runtime type checks for user JSON and logs warnings for invalid `enabled`/`timeout` values instead of hard-failing the whole file.
Invalid `enabled`/`timeout` values are ignored with warnings rather than failing the whole file.
## 3) Auth and runtime value resolution
@@ -138,7 +137,7 @@ This avoids many collisions, but not all. Different raw names can still sanitize
### Schema mapping
`tool-bridge.ts` passes each MCP `inputSchema` through `sanitizeSchemaForMCP()` before registering it as a `CustomTool` schema.
`tool-bridge.ts` passes each MCP `inputSchema` through `normalizeSchemaForMCP()` before registering it as a `CustomTool` schema.
### Execution mapping
+22 -20
View File
@@ -1,12 +1,12 @@
# Autonomous Memory
When enabled, the agent automatically extracts durable knowledge from past sessions and injects a compact summary into each new session. Over time it builds a project-scoped memory store — technical decisions, recurring workflows, pitfalls — that carries forward without manual effort.
When the local memory backend is enabled, the agent automatically extracts durable knowledge from past sessions and injects a compact summary into future sessions for the same project. Over time it builds a project-scoped memory store — technical decisions, recurring workflows, pitfalls — that carries forward without manual effort.
Disabled by default. Enable via `/settings` or `config.yml`:
Disabled by default. Enable the local summary pipeline via `/settings` or `config.yml`:
```yaml
memories:
enabled: true
memory:
backend: local
```
## Usage
@@ -31,17 +31,19 @@ The agent can read memory files directly using `memory://` URLs with the `read`
### `/memory` slash command
| Subcommand | Effect |
| --------------------- | ---------------------------------------------- |
| `view` | Show the current memory injection payload |
| `clear` / `reset` | Delete all memory data and generated artifacts |
| `enqueue` / `rebuild` | Force consolidation to run at next startup |
| Subcommand | Effect |
| --------------------- | --------------------------------------------------------- |
| `view` | Show the current backend injection payload |
| `stats` | Show backend-specific memory statistics, when supported |
| `diagnose` | Show backend-specific diagnostics, when supported |
| `clear` / `reset` | Delete active backend memory data/artifacts |
| `enqueue` / `rebuild` | Force consolidation/retention work for the active backend |
## How it works
Memories are built by a background pipeline that runs at startup or when manually triggered via slash command.
Local summary memories are built by a background pipeline that runs at startup or when manually triggered via slash command. The pipeline is skipped for subagents and for sessions that are not persisted to a session file.
**Phase 1 — per-session extraction:** For each past session that has changed since it was last processed, a model reads the session history and extracts durable signal: technical decisions, constraints, resolved failures, recurring workflows. Sessions that are too recent, too old, or currently active are skipped. Each extraction produces a raw memory block and a short synopsis for that session.
**Phase 1 — per-session extraction:** For each past session that has changed since it was last processed, a model reads the session history and extracts durable signal: technical decisions, constraints, resolved failures, recurring workflows. Sessions that are too recent, too old, currently active, or beyond the configured scan/age limits are skipped. Each extraction produces a raw memory block and a short synopsis for that session.
**Phase 2 — consolidation:** After extraction, a second model pass reads all per-session extractions and produces three outputs written to disk:
@@ -49,9 +51,9 @@ Memories are built by a background pipeline that runs at startup or when manuall
- `memory_summary.md` — the compact text injected at session start
- `skills/` — reusable procedural playbooks, each in its own subdirectory
Phase 2 uses a lease to prevent double-running when multiple processes start simultaneously. Stale skill directories from prior runs are pruned automatically.
Phase 2 uses a lease and heartbeat to prevent double-running when multiple processes start simultaneously. Stale skill directories from prior runs are pruned automatically.
All output is scanned for secrets before being written to disk.
Consolidated output is redacted for common secret/token patterns before `MEMORY.md`, `memory_summary.md`, or generated skills are written to disk.
### Extraction behavior
@@ -77,13 +79,13 @@ If the requested memory role is not configured, memory model resolution falls ba
## Configuration
| Setting | Default | Description |
| ------------------------------------- | ------- | --------------------------------------------------------- |
| `memories.enabled` | `false` | Master switch |
| `memories.maxRolloutAgeDays` | `30` | Sessions older than this are not processed |
| `memories.minRolloutIdleHours` | `12` | Sessions active more recently than this are skipped |
| `memories.maxRolloutsPerStartup` | `64` | Cap on sessions processed in a single startup |
| `memories.summaryInjectionTokenLimit` | `5000` | Max tokens of the summary injected into the system prompt |
| Setting | Default | Description |
| ------------------------------------- | ------- | ---------------------------------------------------------------------------------------------------------------------------------------- |
| `memory.backend` | `off` | Select `local` for this pipeline; legacy `memories.enabled: true` is migrated to `memory.backend: local` when no explicit backend is set |
| `memories.maxRolloutAgeDays` | `30` | Sessions older than this are not processed |
| `memories.minRolloutIdleHours` | `12` | Sessions active more recently than this are skipped |
| `memories.maxRolloutsPerStartup` | `64` | Cap on sessions processed in a single startup |
| `memories.summaryInjectionTokenLimit` | `5000` | Max tokens of the summary injected into the system prompt |
Additional tuning knobs (concurrency, lease durations, token budgets) are available in config for advanced use.
+157
View File
@@ -0,0 +1,157 @@
# Mnemopi memory backend
Oh My Pi can use `@oh-my-pi/pi-mnemopi` as a local long-term memory backend.
Set:
```yaml
memory:
backend: mnemopi
```
Example:
```yaml
memory:
backend: mnemopi
mnemopi:
scoping: per-project-tagged
```
With this backend enabled, the coding agent:
1. Opens one or more local Mnemopi SQLite databases according to the configured bank scoping.
2. Recalls relevant memories into a `<memories>` block for the first model turn of a session and refreshes the base prompt if recall happens from the `agent_start` listener.
3. Retains completed conversation turns into the retain bank after agent turns, no more often than `mnemopi.retainEveryNTurns`.
4. Adds recalled memory as extra compaction context when compaction asks the memory backend for `preCompactionContext`.
5. Uses the normal `/memory view`, `/memory stats`, `/memory diagnose`, `/memory clear`, and `/memory enqueue` commands through the shared memory backend interface.
Recalled memory is background context, not instructions. Current user messages and tool output take precedence when they conflict.
## Settings
| Setting | Default | Description |
| ------------------------------- | ---------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `memory.backend` | `off` | Set to `mnemopi` to enable this backend. |
| `mnemopi.dbPath` | agent memories dir | Optional SQLite database path. |
| `mnemopi.bank` | project directory name | Base bank name passed to `Mnemopi`; the coding-agent wrapper scopes from this base according to `mnemopi.scoping`. |
| `mnemopi.scoping` | `per-project` | Memory visibility mode: `global` = one shared bank, `per-project` = isolated project memory, `per-project-tagged` = project-local writes plus global recall visibility. |
| `mnemopi.autoRecall` | `true` | Recall memory on the first turn of a session. |
| `mnemopi.autoRetain` | `true` | Retain completed turns automatically. |
| `mnemopi.retainEveryNTurns` | `4` | Minimum user turns between automatic retain writes. |
| `mnemopi.recallLimit` | `8` | Maximum recalled memories in the prompt block. |
| `mnemopi.recallContextTurns` | `3` | Prior user-bounded turns included in recall queries. |
| `mnemopi.recallMaxQueryChars` | `4000` | Maximum composed recall query length. |
| `mnemopi.injectionTokenLimit` | `5000` | Approximate token budget for memory prompt injection. |
| `mnemopi.debug` | `false` | Enable debug logging for backend failures. |
| `mnemopi.noEmbeddings` | `false` | Pass `noEmbeddings` to `Mnemopi` and force FTS-only recall. |
| `mnemopi.embeddingModel` | env/default | Embedding model passed to `Mnemopi`. |
| `mnemopi.embeddingApiUrl` | env/default | OpenAI-compatible embedding endpoint passed to `Mnemopi`. |
| `mnemopi.embeddingApiKey` | env/default | Embedding API key passed to `Mnemopi`. |
| `mnemopi.llmMode` | `smol` | `smol` uses the configured pi-ai smol model, `remote` uses the settings below, and `none` disables LLM calls. |
| `mnemopi.llmBaseUrl` | env/default | OpenAI-compatible LLM endpoint for `llmMode: remote`. |
| `mnemopi.llmApiKey` | env/default | LLM API key for `llmMode: remote`. |
| `mnemopi.llmModel` | env/default | LLM model id for `llmMode: remote`. |
## Scoping
The coding-agent wrapper applies scoping on top of the underlying `Mnemopi` package:
- `global` uses one shared bank for recall and writes.
- `per-project` writes to and recalls from a bank derived from the current git repository root (or cwd) plus a stable hash.
- `per-project-tagged` writes to the project-local bank and recalls from both the project-local bank and the shared global bank, with duplicate recall results merged.
The combined project-plus-global behavior lives in the wrapper. The `@oh-my-pi/pi-mnemopi` package itself still exposes banks and constructor options directly, including `bank` for selecting a bank name. Project-local banks other than the shared bank are stored as sibling bank databases managed by Mnemopi's `BankManager`.
## LLM and embeddings
The backend passes these settings to the `Mnemopi` constructor; if a setting is omitted, Mnemopi falls back to its `MNEMOPI_*` environment defaults. The backend does not download or run a local GGUF LLM. LLM-dependent paths use a configured pi-ai model, a dynamic completion function, a remote OpenAI-compatible endpoint, or deterministic no-LLM fallbacks.
FTS-only:
```yaml
memory:
backend: mnemopi
mnemopi:
noEmbeddings: true
```
Equivalent constructor shape:
```ts
new Mnemopi({ noEmbeddings: true });
```
Remote embeddings:
```yaml
mnemopi:
embeddingModel: text-embedding-3-small
embeddingApiUrl: https://api.openai.com/v1
embeddingApiKey: ${OPENAI_API_KEY}
```
Equivalent constructor shape:
```ts
new Mnemopi({
embeddingModel: "text-embedding-3-small",
embeddingApiUrl: "https://api.openai.com/v1",
embeddingApiKey,
});
```
Remote LLM:
```yaml
mnemopi:
llmMode: remote
llmBaseUrl: https://api.openai.com/v1
llmApiKey: ${OPENAI_API_KEY}
llmModel: gpt-4.1-mini
```
Equivalent constructor shapes:
```ts
new Mnemopi({ llm: { baseUrl, apiKey, model } });
new Mnemopi({ llmBaseUrl: baseUrl, llmApiKey: apiKey, llmModel: model });
```
Dynamic function LLM for rotating OAuth tokens:
```ts
new Mnemopi({
llm: async (prompt, opts) => {
const token = await getFreshOauthToken();
return await completeWithPiAi(prompt, {
token,
maxTokens: opts?.maxTokens,
temperature: opts?.temperature,
});
},
});
```
pi-ai smol model LLM:
```yaml
mnemopi:
llmMode: smol
```
The coding agent resolves its configured smol role and passes a dynamic completion function so every Mnemopi LLM call can fetch the current provider credentials at call time:
```ts
new Mnemopi({
llm: async (prompt, opts) => completeSmolWithCurrentAuth(prompt, opts),
});
```
## Operational notes
- The default shared database lives under the agent memories directory in `mnemopi/mnemopi.db`; project-scoped banks use sibling database paths under that Mnemopi directory.
- `/memory clear` removes every scoped Mnemopi SQLite database and sidecar WAL/SHM files for the active configuration.
- `/memory enqueue` forces retention of the current session, flushes pending fact extractions, and runs Mnemopi sleep/consolidation.
- `/memory stats` and `/memory diagnose` render backend-specific bank statistics/diagnostics when the Mnemopi backend is active.
- Subagents do not own separate Mnemopi retain loops; they alias the parent state when a parent Mnemopi state exists, and otherwise remain inert.
+43 -10
View File
@@ -55,7 +55,7 @@ providers:
X-Team: platform
authHeader: true
auth: apiKey
disableStrictTools: false # set true for Anthropic-compatible endpoints that reject the strict field
disableStrictTools: false # set true for Anthropic-compatible endpoints that reject the strict field
discovery:
type: ollama
modelOverrides:
@@ -103,7 +103,8 @@ providers:
### Allowed auth/discovery values
- `auth`: `apiKey` (default), `none`, or `oauth`; for `models.yml` custom models, `oauth` is accepted by schema but does not waive the `apiKey` requirement
- `discovery.type`: `ollama`, `llama.cpp`, or `lm-studio`
- `discovery.type`: `ollama`, `llama.cpp`, `lm-studio`, `openai-models-list`, or `proxy`
- `transport`: `pi-native` only. When set, every model under that provider is sent to an `omp auth-gateway` compatible `baseUrl` via `POST /v1/pi/stream`; `apiKey` is the gateway bearer.
## Validation rules (current)
@@ -120,6 +121,7 @@ Required:
Must define at least one of:
- `baseUrl`
- `apiKey`
- `headers`
- `compat`
- `disableStrictTools`
@@ -229,7 +231,7 @@ Provider defaults vs per-model overrides:
- Provider `headers` are baseline.
- Model `headers` override provider header keys.
- `modelOverrides` can override model metadata (`name`, `reasoning`, `input`, `cost`, `contextWindow`, `maxTokens`, `headers`, `compat`, `contextPromotionTarget`).
- `modelOverrides` can override model metadata (`name`, `reasoning`, `thinking`, `input`, `cost`, `premiumMultiplier`, `contextWindow`, `maxTokens`, `headers`, `compat`, `contextPromotionTarget`).
- `compat` is deep-merged for nested routing blocks (`openRouterRouting`, `vercelGatewayRouting`, `extraBody`).
## Runtime discovery integration
@@ -288,6 +290,33 @@ providers:
type: llama.cpp
```
### Proxy discovery (`discovery.type: proxy`)
For Anthropic+OpenAI-compatible proxies (new-api / one-api / similar)
that expose both `/v1/messages` and `/v1/chat/completions` behind the same
host. Discovery hits `GET /v1/models` (10s timeout, OpenAI-style payload) and
derives each model's `api` from the entry's `supported_endpoint_types`:
- contains `"anthropic"` -> `api: anthropic-messages` (routes via `/v1/messages`)
- contains `"openai"` -> `api: openai-completions` (routes via `/v1/chat/completions`)
- otherwise -> falls back to provider-level `api` if set, else dropped
Provider-level `api` is **optional** with `discovery.type: proxy` because the
per-model wire is auto-detected. The Anthropic SDK strips a trailing `/v1`
from `baseUrl` before appending `/v1/messages`, so a single discovery `baseUrl`
(ending in `/v1`) round-trips correctly to both wires.
```yaml
providers:
newapi-reseller:
baseUrl: https://api.example.com/v1
apiKey: xxxx
authHeader: true # injects Authorization: Bearer for openai models
disableStrictTools: true # most anthropic-fronted proxies reject `strict`
discovery:
type: proxy
```
### Extension provider registration
Extensions can register providers at runtime (`pi.registerProvider(...)`), including:
@@ -463,21 +492,21 @@ Configure fallback directly in model metadata via `contextPromotionTarget`.
- `provider/model-id` (explicit)
- `model-id` (resolved within current provider)
Example (`models.yml`) for Spark -> non-Spark on the same provider:
Example (`models.yml`) for an explicit OpenAI fallback:
```yaml
providers:
openai-codex:
modelOverrides:
gpt-5.3-codex-spark:
contextPromotionTarget: openai-codex/gpt-5.3-codex
gpt-5.5:
contextPromotionTarget: openai-codex/gpt-5.4
```
The built-in model generator also assigns this automatically for `*-spark` models when a same-provider base model exists.
The built-in model policy currently links OpenAI `codex-spark` variants to `gpt-5.5`, and `gpt-5.5` to `gpt-5.4`, when that target exists on the same provider/API.
## Compatibility and routing fields
The `compat` block on a provider or model overrides the URL-based auto-detection in `packages/ai/src/providers/openai-completions-compat.ts`. It is validated by `OpenAICompatSchema` in `packages/coding-agent/src/config/model-registry.ts` and consumed by every `openai-completions` transport (`packages/ai/src/providers/openai-completions.ts`). The canonical type is `OpenAICompat` in `packages/ai/src/types.ts`.
The `compat` block on a provider or model overrides the URL-based auto-detection in `packages/ai/src/providers/openai-completions-compat.ts`. It is validated by `OpenAICompatSchema` in `packages/coding-agent/src/config/models-config-schema.ts` and consumed by every `openai-completions` transport (`packages/ai/src/providers/openai-completions.ts`). The canonical type is `OpenAICompat` in `packages/ai/src/types.ts`.
`models.yml` accepts the following keys (all optional; unset falls back to URL detection):
@@ -485,10 +514,12 @@ Request shaping:
- `supportsStore` — emit `store: false` on requests. Default: auto (off for non-standard endpoints).
- `supportsDeveloperRole` — use the `developer` system role for reasoning models instead of `system`. Default: auto.
- `supportsMultipleSystemMessages` — preserve separate leading system/developer messages instead of coalescing them. Default: auto (known OpenAI-compatible hosted APIs preserve; strict-template/local hosts coalesce).
- `supportsUsageInStreaming` — send `stream_options: { include_usage: true }` to receive token usage on streaming responses. Default: `true`.
- `maxTokensField` — `"max_completion_tokens"` or `"max_tokens"`. Default: auto.
- `supportsToolChoice` — emit the `tool_choice` parameter when the caller forces a specific tool. Default: `true`. Set `false` for endpoints that 400 on `tool_choice` (e.g. DeepSeek when reasoning is on).
- `disableReasoningOnForcedToolChoice` — drop `reasoning_effort` / OpenRouter `reasoning` whenever `tool_choice` forces a call. Default: auto (Kimi/Anthropic-fronted endpoints).
- `disableReasoningOnToolChoice` — drop reasoning fields whenever any `tool_choice` is sent. Default: auto (DeepSeek reasoning models).
- `extraBody` — extra top-level fields merged into every request body (gateway hints, controller selectors, etc.).
Reasoning / thinking:
@@ -498,6 +529,7 @@ Reasoning / thinking:
- `thinkingFormat` — request shape for thinking: `"openai"` (`reasoning_effort`), `"openrouter"` (`reasoning: { effort }`), `"zai"` (`thinking: { type: "enabled" }`), `"qwen"` (top-level `enable_thinking`), or `"qwen-chat-template"` (`chat_template_kwargs.enable_thinking`). Default: `"openai"`.
- `reasoningContentField` — assistant field carrying chain-of-thought: `"reasoning_content"`, `"reasoning"`, or `"reasoning_text"`. Default: auto.
- `requiresReasoningContentForToolCalls` — assistant tool-call turns must round-trip the reasoning field (DeepSeek-R1, Kimi, OpenRouter when reasoning is on). Default: `false`.
- `allowsSyntheticReasoningContentForToolCalls` — allow a placeholder reasoning field when a prior assistant tool-call turn lacks provider reasoning content. Default: `true`; set `false` for providers that validate the exact reasoning value.
- `requiresAssistantContentForToolCalls` — assistant tool-call turns must include non-empty text content (Kimi). Default: `false`.
Tool / message normalization:
@@ -518,7 +550,7 @@ Provider-level `compat` is the baseline; per-model `compat` is deep-merged on to
### Anthropic compatibility (`anthropic-messages`)
For `anthropic-messages` models the runtime uses a separate `AnthropicCompat` shape (`packages/ai/src/types.ts`). The `models.yml` schema currently exposes only the strict-tools opt-out as a top-level provider field (see below); the remaining Anthropic-side knobs (`disableAdaptiveThinking`, `supportsEagerToolInputStreaming`, `supportsLongCacheRetention`) are set by built-in catalog metadata and are not user-configurable from `models.yml`.
For `anthropic-messages` models the runtime uses a separate `AnthropicCompat` shape (`packages/ai/src/types.ts`). The `models.yml` schema currently exposes only the strict-tools opt-out as a top-level provider field (see below); the remaining Anthropic-side knobs (`disableAdaptiveThinking`, `supportsEagerToolInputStreaming`, `supportsLongCacheRetention`, `supportsMidConversationSystem`) are set by built-in catalog metadata and are not user-configurable from `models.yml`.
### Strict tool schemas (`disableStrictTools`)
@@ -555,6 +587,7 @@ plus the OpenAI strict-mode sanitize+enforce pipeline). See
edge cases (local `$ref` inlining, single-item `allOf` collapse,
`anyOf`-wrapper description hoist, enum/const primitive-type inference)
and the per-provider dispatcher mapping.
## Practical examples
### Local OpenAI-compatible endpoint (no auth)
@@ -579,7 +612,7 @@ providers:
apiKey: ANTHROPIC_PROXY_API_KEY
api: anthropic-messages
authHeader: true
disableStrictTools: true # if the proxy doesn't support strict tool schemas
disableStrictTools: true # if the proxy doesn't support strict tool schemas
models:
- id: claude-sonnet-4-20250514
name: Claude Sonnet 4 (Proxy)
+30 -16
View File
@@ -15,11 +15,12 @@ This document covers the runtime loader shipped by `@oh-my-pi/pi-natives`: how `
The loader is intentionally narrow:
- Build a platform/CPU-aware candidate list for addon filenames and directories.
- Treat an embedded-addon manifest as the authoritative compiled-binary signal when present.
- Optionally materialize an embedded addon into a versioned per-user cache directory.
- Attempt candidates in deterministic order and return the first addon that `require(...)` loads.
- Treat an embedded-addon manifest as a compiled-binary signal when present.
- Optionally materialize embedded addon archive contents into a versioned per-user cache directory.
- On Windows `node_modules` installs, stage addon files into the versioned cache to avoid locked-DLL update failures.
- Attempt candidates in deterministic order and return the first addon that `require(...)` loads and validates.
The current loader does **not** run a separate `validateNative(...)` export-presence gate. API shape is provided by the generated N-API binding file (`native/index.d.ts`) and the loaded addon itself. A stale binary therefore normally fails as a missing property or native load error rather than as a custom "missing exports" validation error.
For install and compiled-binary paths, the loader verifies a release sentinel export named from `package.json#version` (for example `__piNativesV15_7_2`). Workspace-dev loads skip this validation so a local checkout can rebuild after a pull. The loader does not validate the full export surface; stale same-version or incomplete binaries still surface as missing members or native errors at use sites.
## Runtime inputs and derived state
@@ -28,6 +29,7 @@ At module initialization, `native/index.js` computes:
- **Platform tag**: `${process.platform}-${process.arch}` (for example `darwin-arm64`).
- **Package version**: from `packages/natives/package.json`.
- **Core directories**:
- `leafPackageDir`: directory of the platform leaf package, resolved via `require.resolve("@oh-my-pi/pi-natives-<tag>/package.json")`; `null` when no leaf is installed (e.g. local dev).
- `nativeDir`: package-local `packages/natives/native`.
- `execDir`: directory containing `process.execPath`.
- `versionedDir`: `<getNativesDir()>/<packageVersion>`.
@@ -41,6 +43,7 @@ At module initialization, `native/index.js` computes:
- embedded-addon manifest is non-null,
- `PI_COMPILED` env var is set,
- `import.meta.url` contains Bun embedded markers (`$bunfs`, `~BUN`, `%7EBUN`).
- **Windows staging mode** (`shouldStageNodeModulesAddon`): true only on Windows, in non-compiled mode, when `nativeDir` is inside `node_modules`.
- **Variant override**: `PI_NATIVE_VARIANT` (`modern`/`baseline` only; invalid values ignored).
- **Selected variant**: explicit override, otherwise runtime AVX2 detection on x64 (`modern` if AVX2, else `baseline`).
@@ -92,10 +95,15 @@ The default unsuffixed fallback remains part of the x64 candidate list.
### Non-compiled runtime
For each filename, candidates are:
For each filename, candidates are, in order:
1. `<nativeDir>/<filename>`
2. `<execDir>/<filename>`
1. `<leafPackageDir>/<filename>` (omitted when `leafPackageDir` is `null`)
2. `<nativeDir>/<filename>`
3. `<execDir>/<filename>`
The leaf package dir comes first so the optional-dependency binary published with the release is preferred over any `.node` left in the core package's `native/` (e.g. a stale local-dev build).
On Windows installs where `nativeDir` is inside a `node_modules` segment (`shouldStageNodeModulesAddon`), `<versionedDir>/<filename>` staging candidates are prepended ahead of the leaf candidates so a locked `node_modules` binary can be sidestepped during `bun install -g` updates. The staged file is copied from `leafPackageDir ?? nativeDir` before probing.
### Compiled runtime
@@ -106,7 +114,7 @@ For each filename, candidates are:
3. `<nativeDir>/<filename>`
4. `<execDir>/<filename>`
At load time, an extracted embedded candidate, when produced, is prepended ahead of these de-duplicated candidates.
At load time, an extracted embedded candidate, or a staged Windows candidate when no embedded candidate exists, is prepended ahead of these de-duplicated candidates.
## Embedded addon extraction lifecycle
@@ -114,7 +122,8 @@ At load time, an extracted embedded candidate, when produced, is prepended ahead
- `platformTag`
- `version`
- `files[]` entries with `variant`, `filename`, and `filePath`
- `archive`: `{ format: "tar.gz", filename, filePath }`
- `files[]` entries with `variant`, `filename`, and `size`
Extraction (`maybeExtractEmbeddedAddon`) runs only when:
@@ -133,11 +142,12 @@ Variant file selection:
Materialization:
1. Ensure `<versionedDir>` exists.
2. Reuse `<versionedDir>/<selected filename>` if it already exists.
3. Otherwise read `selectedEmbeddedFile.filePath` and write the target path.
4. Return the target path as the first candidate.
2. Select `<versionedDir>/<selected filename>`.
3. If the current cached file exists and its size matches manifest metadata, reuse it.
4. Otherwise extract `embeddedAddon.archive.filePath` into `<versionedDir>` using the manifest `files[]` allowlist.
5. Verify the selected target by size and return it as the first candidate.
Directory creation or write failures are appended to the loader error list; probing continues through normal candidates.
Archive, directory, or write failures are appended to the loader error list; probing continues through normal candidates.
## Lifecycle and state transitions
@@ -146,11 +156,14 @@ Init
-> Load package metadata and embedded-addon manifest
-> Compute platform/version/variant/filenames/candidate paths
-> (compiled + embedded manifest matches?)
yes -> try extract to versionedDir (record errors, continue)
yes -> extract archive to versionedDir when needed (record errors, continue)
no -> skip extraction
-> (Windows non-compiled node_modules install and no embedded candidate?)
yes -> stage leaf/core addon to versionedDir (record errors, continue)
no -> skip staging
-> For each runtime candidate in order:
require(candidate)
-> success: return addon exports (READY)
-> sentinel validation passes or is workspace-dev: return addon exports (READY)
-> failure: record error, continue
-> none loaded:
if unsupported platform tag -> throw Unsupported platform
@@ -172,7 +185,7 @@ If all candidates fail and `platformTag` is not supported, the loader throws:
If the platform is supported but no candidate can be loaded, the final error includes:
- `Failed to load pi_natives native addon for <platformTag>` or `<platformTag> (<variant>)`
- every attempted path with the corresponding `require(...)` error
- every attempted path with the corresponding `require(...)` or sentinel-validation error
- mode-specific remediation hints
### Compiled-binary startup failures
@@ -182,6 +195,7 @@ Compiled mode diagnostics include:
- expected versioned cache target paths (`<versionedDir>/<filename>`),
- remediation to delete the versioned cache and rerun,
- direct release download `curl` commands for each expected filename.
- release sentinel mismatch details when a loadable `.node` belongs to another `@oh-my-pi/pi-natives` version.
### Non-compiled startup failures
+30 -21
View File
@@ -1,8 +1,8 @@
# Natives Architecture
`@oh-my-pi/pi-natives` is now a two-layer package around a loader:
`@oh-my-pi/pi-natives` is a two-layer package around an ESM loader:
1. **CommonJS loader/package entrypoint** resolves and loads the correct `.node` addon and patches generated enum objects onto the export object.
1. **ESM loader/package entrypoint** resolves and loads the correct `.node` addon with `createRequire`, validates the release sentinel outside workspace-dev loads, and re-exports generated classes/functions plus enum runtime objects as explicit named ESM exports.
2. **Rust N-API module layer** implements the exported functions/classes and emits the generated TypeScript declarations.
This document is the foundation for deeper module-level docs.
@@ -21,20 +21,20 @@ This document is the foundation for deeper module-level docs.
## Package entrypoint and public surface
`packages/natives/package.json` points directly at generated native bindings:
`packages/natives/package.json` points at generated native artifacts:
- `main`: `./native/index.js`
- `types`: `./native/index.d.ts`
- `exports["."].types`: `./native/index.d.ts`
- `exports["."].import`: `./native/index.js`
There is no current `packages/natives/src` TypeScript wrapper layer. Consumers import functions/classes/enums directly from `@oh-my-pi/pi-natives`; the type contract is the generated `native/index.d.ts` plus enum exports appended by `scripts/gen-enums.ts`.
There is no current `packages/natives/src` TypeScript wrapper layer. Consumers import functions/classes/enums directly from `@oh-my-pi/pi-natives`; the type contract is the generated `native/index.d.ts` plus the explicit named exports generated into `native/index.js` by `scripts/gen-enums.ts`.
Current capability groups in the generated API include:
- **Search/text/code primitives**: `grep`, `search`, `hasMatch`, `fuzzyFind`, `glob`, `astGrep`, `astEdit`, text width/slicing/wrapping/sanitization, syntax highlighting, token counting.
- **Execution/process/terminal primitives**: `executeShell`, `Shell`, `PtySession`, process-tree helpers, key parsing.
- **System/media/conversion primitives**: clipboard, image resize/encode/SIXEL, HTML-to-Markdown, macOS appearance/power helpers, work profiling, Windows ProjFS overlay helpers.
- **Search/text/code primitives**: `grep`, `search`, `hasMatch`, `fuzzyFind`, `glob`, `astGrep`, `astEdit`, `blockRangeAt`, `summarizeCode`, text width/slicing/wrapping/sanitization, syntax highlighting, token counting.
- **Execution/process/terminal primitives**: `executeShell`, `Shell`, `PtySession`, `Process`, key parsing, bash fixups.
- **System/media/isolation/conversion primitives**: clipboard, SIXEL encoding, HTML-to-Markdown, macOS appearance/power helpers, work profiling, workspace scanning, isolation backend helpers (`iso*`).
## Loader layer
@@ -72,7 +72,9 @@ For x64, variant selection uses:
### Binary distribution and extraction model
`packages/natives/package.json` publishes `native/`, which contains the loader, generated declarations, generated enum patch, embedded-addon manifest stub, and prebuilt `.node` artifacts.
The published `@oh-my-pi/pi-natives` package ships **only** the loader layer in `native/`: the ESM loader (`index.js`), generated declarations (`index.d.ts`), the `loader-state.js`/`.d.ts` helpers, and the embedded-addon manifest stub (`embedded-addon.js`). It carries no `.node` binaries.
Each platform's prebuilt `.node` is published as a separate optional-dependency leaf package — `@oh-my-pi/pi-natives-<platform>-<arch>`, one per supported tag — which the core lists in `optionalDependencies` at the lockstep version during publish. npm/bun install only the leaf whose `os`/`cpu` match the host. The working-tree package keeps built `.node` files under `native/` for local dev; the release-publish rewrite (`prepareNativeCorePackage` in `scripts/ci-release-publish.ts`) strips them from the core tarball, and the leaves are generated by `packages/natives/scripts/gen-npm-packages.ts` (`LEAF_TARGETS`). Adding a build target therefore requires a matching `LEAF_TARGETS` entry, or the binary never reaches npm users.
For compiled binaries, loader behavior is:
@@ -84,7 +86,9 @@ For compiled binaries, loader behavior is:
`getNativesDir()` uses `$XDG_DATA_HOME/omp/natives` when `$XDG_DATA_HOME/omp` exists; otherwise it uses `~/.omp/natives`.
If a populated embedded addon manifest is present, it is also treated as a compiled-binary signal. The loader can extract the matching embedded `.node` into the versioned cache directory before candidate probing.
If a populated embedded addon manifest is present, it is also treated as a compiled-binary signal. Current embedded manifests point at a gzip-compressed tar archive (`embedded-addons.<tag>.tar.gz`) that contains one or more matching `.node` files. The loader extracts the archive into the versioned cache directory, validates the selected file by size, and prepends that cache path before normal candidate probing.
For npm/bun installs (non-compiled), `loader-state.js` resolves the platform leaf directory via `require.resolve("@oh-my-pi/pi-natives-<tag>/package.json")` and probes its `.node` **before** the core package's `native/` directory and the executable directory. The optional-dependency binary is therefore preferred over any `.node` left in the core (e.g. a stale local-dev build). On Windows `node_modules` installs, the loader first stages the selected leaf/core addon into `<getNativesDir()>/<packageVersion>/...` and prepends that staged path so running processes do not lock the `node_modules` copy during global updates.
### Failure modes
@@ -92,9 +96,8 @@ Loader failures are explicit:
- **Unsupported platform tag**: after failed probing, throws with supported platform list.
- **No loadable candidate**: throws with all attempted paths and remediation hints.
- **Embedded extraction errors**: directory/write failures are recorded and included in final load diagnostics if no candidate loads.
The current loader does not perform a separate post-`require` export validation pass.
- **Embedded/staging errors**: directory/write/archive/staging failures are recorded and included in final load diagnostics if no candidate loads.
- **Release mismatch**: outside workspace-dev loads, a candidate that loads but lacks the version sentinel export for `package.json#version` is rejected with a reinstall hint.
## Rust N-API module layer
@@ -102,6 +105,7 @@ The current loader does not perform a separate post-`require` export validation
- `appearance`
- `ast`
- `block`
- `clipboard`
- `fd`
- `fs_cache`
@@ -110,19 +114,21 @@ The current loader does not perform a separate post-`require` export validation
- `grep`
- `highlight`
- `html`
- `image`
- `iso`
- `keys`
- `language`
- `language` (re-exported from `pi_ast`)
- `power`
- `prof`
- `projfs_overlay`
- `ps`
- `pty`
- `shell`
- `sixel`
- `summary`
- `task`
- `text`
- `tokens`
- `utils` (crate-private helpers)
- `workspace`
N-API exports are generated from Rust `#[napi]` functions/classes/objects/enums. Snake_case Rust names are exposed as camelCase JavaScript names unless explicitly configured by napi-rs.
@@ -131,8 +137,9 @@ N-API exports are generated from Rust `#[napi]` functions/classes/objects/enums.
- **Loader/package ownership (`packages/natives/native`, `packages/natives/scripts`)**
- runtime binary selection
- CPU variant selection and override handling
- compiled-binary embedded extraction
- generated TypeScript declarations and enum export patching
- compiled-binary embedded archive extraction
- Windows `node_modules` addon staging
- generated TypeScript declarations and explicit ESM export/enum patching
- **Rust ownership (`crates/pi-natives/src`)**
- algorithmic and system-level implementation
- platform-native behavior and performance-sensitive logic
@@ -145,16 +152,18 @@ N-API exports are generated from Rust `#[napi]` functions/classes/objects/enums.
1. Consumer imports from `@oh-my-pi/pi-natives`.
2. `native/index.js` computes platform/arch/variant and candidate paths.
3. Optional embedded binary extraction occurs for compiled distributions.
4. The first `require(candidate)` that succeeds becomes the exported addon object.
5. Generated enum objects are appended to `module.exports`.
3. Optional embedded archive extraction or Windows `node_modules` staging can prepend a versioned-cache candidate.
4. Each candidate is `require(...)`d; install/compiled loads must expose the package-version sentinel.
5. The loaded addon object is bound to explicit named ESM exports, including generated enum objects.
6. Caller invokes generated N-API functions/classes directly.
## Glossary
- **Native addon**: A `.node` binary loaded via Node-API (N-API).
- **Platform tag**: Runtime tuple `platform-arch` (for example `darwin-arm64`).
- **Platform leaf package**: Per-platform npm package `@oh-my-pi/pi-natives-<tag>` that carries one platform's prebuilt `.node`. The core depends on every leaf via `optionalDependencies`; the package manager installs only the host-matching one (`os`/`cpu`).
- **Variant**: x64 CPU-specific build flavor (`modern` AVX2, `baseline` fallback).
- **Generated binding declaration**: `native/index.d.ts` emitted by napi-rs during `build-native.ts`.
- **Version sentinel**: Rust export named from the package version (for example `__piNativesV15_7_2`) that lets the loader reject a `.node` from a different release.
- **Compiled binary mode**: Runtime mode where the CLI is bundled and native addons are resolved from embedded/cache paths before package-local paths.
- **Embedded addon**: Build artifact metadata and file references generated into `native/embedded-addon.js` so compiled binaries can extract matching `.node` payloads.
- **Embedded addon**: Build artifact metadata and archive reference generated into `native/embedded-addon.js` so compiled binaries can extract matching `.node` payloads.
+37 -36
View File
@@ -2,7 +2,7 @@
This document defines the JS/TS contract between `@oh-my-pi/pi-natives` callers and the loaded N-API addon.
Current package shape is direct-to-native: there is no `packages/natives/src/<module>` TypeScript wrapper layer. The public API is the generated `packages/natives/native/index.d.ts` declaration file, the CommonJS loader in `packages/natives/native/index.js`, and the Rust `#[napi]` exports in `crates/pi-natives/src`.
Current package shape is direct-to-native: there is no `packages/natives/src/<module>` TypeScript wrapper layer. The public API is the generated `packages/natives/native/index.d.ts` declaration file, the ESM loader/export wrapper in `packages/natives/native/index.js`, and the Rust `#[napi]` exports in `crates/pi-natives/src`.
## Implementation files
@@ -19,10 +19,10 @@ Current package shape is direct-to-native: there is no `packages/natives/src/<mo
The contract has three parts:
1. **Generated runtime loader** (`native/index.js`)
- computes candidates and `require(...)`s the `.node` addon;
- exports the loaded addon object directly;
- appends enum objects generated by `scripts/gen-enums.ts`.
1. **ESM runtime loader/export wrapper** (`native/index.js`)
- calls `loadNative()` from `loader-state.js`, which `require(...)`s the `.node` addon;
- binds generated classes/functions as explicit named ESM exports;
- emits enum runtime objects generated by `scripts/gen-enums.ts`.
2. **Generated TypeScript declarations** (`native/index.d.ts`)
- generated by napi-rs during `scripts/build-native.ts`;
- declares exported functions, classes, object interfaces, and native enums;
@@ -31,7 +31,7 @@ The contract has three parts:
- `#[napi]` functions/classes/objects/enums are the source of generated declarations and runtime symbols;
- snake_case Rust names become camelCase JavaScript names by napi-rs convention.
There is no current `NativeBindings` declaration-merging lifecycle and no `validateNative(...)` required-export list in the loader.
There is no current `NativeBindings` declaration-merging lifecycle and no full required-export list in the loader. Install/compiled loads do validate the package-version sentinel export; workspace-dev loads skip that check.
## Public export surface organization
@@ -54,35 +54,35 @@ Consumers in `packages/coding-agent` and `packages/tui` import directly from `@o
## JS API ↔ native export mapping (representative)
| Category | Public JS API | Rust source | Return style |
| ----------------- | ---------------------------------------------------------------------------------------------------------- | ----------------------------------------------------------- | ----------------------------- |
| Grep | `grep(options, onMatch?)` | `grep.rs` | `Promise<GrepResult>` |
| Grep | `search(content, options)` | `grep.rs` | `SearchResult` |
| Grep | `hasMatch(content, pattern, ignoreCase?, multiline?)` | `grep.rs` | `boolean` |
| Fuzzy path search | `fuzzyFind(options)` | `fd.rs` | `Promise<FuzzyFindResult>` |
| Glob | `glob(options, onMatch?)` | `glob.rs` | `Promise<GlobResult>` |
| Glob cache | `invalidateFsScanCache(path?)` | `fs_cache.rs` | `void` |
| AST search/edit | `astGrep(options)`, `astEdit(options)` | `ast.rs` | `Promise<...>` |
| Shell | `executeShell(options, onChunk?)` | `shell.rs` | `Promise<ShellExecuteResult>` |
| Shell | `new Shell(options?)`, `shell.run(...)`, `shell.abort()` | `shell.rs` | class / promises |
| PTY | `new PtySession()`, `start/write/resize/kill` | `pty.rs` | class / promises |
| Process | `killTree(pid, signal)`, `listDescendants(pid)` | `ps.rs` | sync |
| Keys | `parseKey`, `matchesKey`, Kitty/legacy helpers | `keys.rs` | sync |
| Text | `wrapTextWithAnsi`, `truncateToWidth`, `sliceWithWidth`, `extractSegments`, `visibleWidth` | `text.rs` | sync |
| Highlight | `highlightCode`, `supportsLanguage`, `getSupportedLanguages` | `highlight.rs` | sync |
| HTML | `htmlToMarkdown(html, options?)` | `html.rs` | `Promise<string>` |
| Image | `PhotonImage`, `encodeSixel` | `image.rs` | class / sync / promises |
| Clipboard | `copyToClipboard`, `readImageFromClipboard` | `clipboard.rs` | sync / promise |
| Tokens | `countTokens(input, encoding?)` | `tokens.rs` | sync |
| System | `detectMacOSAppearance`, `MacAppearanceObserver`, `MacOSPowerAssertion`, `getWorkProfile`, ProjFS helpers | `appearance.rs`, `power.rs`, `prof.rs`, `projfs_overlay.rs` | mixed |
| Category | Public JS API | Rust source | Return style |
| ----------------- | --------------------------------------------------------------------------------------------------------- | ------------------------------------------------ | -------------------------- |
| Grep | `grep(options, onMatch?)` | `grep.rs` | `Promise<GrepResult>` |
| Grep | `search(content, options)` | `grep.rs` | `SearchResult` |
| Grep | `hasMatch(content, pattern, ignoreCase?, multiline?)` | `grep.rs` | `boolean` |
| Fuzzy path search | `fuzzyFind(options)` | `fd.rs` | `Promise<FuzzyFindResult>` |
| Glob/workspace | `glob(options, onMatch?)`, `listWorkspace(options)` | `glob.rs`, `workspace.rs` | `Promise<...>` |
| Glob cache | `invalidateFsScanCache(path?)` | `fs_cache.rs` | `void` |
| AST/block/summary | `astGrep(options)`, `astEdit(options)`, `blockRangeAt(options)`, `summarizeCode(options)` | `ast.rs`, `block.rs`, `summary.rs` | mixed |
| Shell | `executeShell(options, onChunk?)` | `shell.rs` | `Promise<ShellRunResult>` |
| Shell | `new Shell(options?)`, `shell.run(...)`, `shell.abort()` | `shell.rs` | class / promises |
| PTY | `new PtySession()`, `start/write/resize/kill` | `pty.rs` | class / promises |
| Process | `Process.fromPid/fromPath`, `status/children/killTree/terminate/waitForExit` | `ps.rs` | class / mixed |
| Keys | `parseKey`, `matchesKey`, Kitty/legacy helpers | `keys.rs` | sync |
| Text | `wrapTextWithAnsi`, `truncateToWidth`, `sliceWithWidth`, `extractSegments`, `visibleWidth` | `text.rs` | sync |
| Highlight | `highlightCode`, `supportsLanguage`, `getSupportedLanguages` | `highlight.rs` | sync |
| HTML | `htmlToMarkdown(html, options?)` | `html.rs` | `Promise<string>` |
| SIXEL | `encodeSixel` | `sixel.rs` | sync |
| Clipboard | `copyToClipboard`, `readImageFromClipboard` | `clipboard.rs` | sync / promise |
| Tokens | `countTokens(input, encoding?)` | `tokens.rs` | sync |
| System/isolation | `detectMacOSAppearance`, `MacAppearanceObserver`, `MacOSPowerAssertion`, `getWorkProfile`, `iso*` helpers | `appearance.rs`, `power.rs`, `prof.rs`, `iso.rs` | mixed |
## Sync vs async contract differences
The contract preserves Rust/N-API call style:
- **Promise-returning exports** for worker-thread or async runtime work (`grep`, `glob`, `fuzzyFind`, `astGrep`, `astEdit`, `htmlToMarkdown`, shell/PTY runs, image parse/resize/encode, clipboard image read).
- **Synchronous exports** for deterministic in-memory transforms/parsers or direct system calls (`search`, `hasMatch`, highlighting, text utilities, token counting, process queries, `copyToClipboard`, `encodeSixel`).
- **Constructor exports** for stateful runtime objects (`Shell`, `PtySession`, `PhotonImage`, macOS observer/power handles).
- **Promise-returning exports** for worker-thread or async runtime work (`grep`, `glob`, `fuzzyFind`, `astGrep`, `astEdit`, `htmlToMarkdown`, shell/PTY runs, `isoStart`/`isoStop`/`isoDiff`, clipboard image read, workspace scan).
- **Synchronous exports** for deterministic in-memory transforms/parsers or direct system calls (`search`, `hasMatch`, highlighting, text utilities, token counting, process construction/status, `copyToClipboard`, `encodeSixel`, isolation probe/resolve helpers).
- **Constructor exports** for stateful runtime objects (`Shell`, `PtySession`, `Process`, macOS observer/power handles).
Changing sync ↔ async for an existing export is a breaking public API change because consumers call these exports directly.
@@ -94,29 +94,30 @@ Changing sync ↔ async for an existing export is a breaking public API change b
- `GrepResult`, `SearchResult`, `GlobResult`, `FuzzyFindResult`
- `ShellRunResult`, `ShellExecuteResult`, `PtyRunResult`, `MinimizerResult`
- `AstFindResult`, `AstReplaceResult`
- `System`/media payloads such as `ClipboardImage`, `WorkProfile`, `ParsedKittyResult`
- `AstFindResult`, `AstReplaceResult`, `BlockRange`, `SummaryResult`
- `System`/media/isolation payloads such as `ClipboardImage`, `WorkProfile`, `ParsedKittyResult`, `IsoResolveResult`
Runtime shape correctness is owned by napi-rs and the Rust implementation.
### Enum patterns
Native enums are represented in generated declarations and also appended to `module.exports` by `scripts/gen-enums.ts`, because the loader is hand-maintained CommonJS around the generated addon. Current enum objects include:
Native enums are represented in generated declarations and also emitted as runtime objects by `scripts/gen-enums.ts`, because napi-rs string enums are TS-only without explicit JS exports. Current enum objects include:
- `AstMatchStrictness`
- `Ellipsis`
- `Encoding`
- `FileType`
- `GrepOutputMode`
- `ImageFormat`
- `IsoBackendKind`
- `IsoChangeKind`
- `KeyEventType`
- `MacOSAppearance`
- `SamplingFilter`
- `ProcessStatus`
## Error behavior and caveats
- Addon load failure or unsupported platform throws during package import from `native/index.js`.
- The loader does not verify the full export set after `require(...)`; stale or mismatched binaries surface as native load errors or missing members at use sites.
- The loader rejects install/compiled candidates that lack the package-version sentinel export. It does not verify the full export set after `require(...)`; stale same-version or incomplete binaries surface as native load errors or missing members at use sites.
- N-API conversion validates basic argument conversion, but TS optional fields do not guarantee semantic validity for untyped callers.
- Numeric enum declarations do not prevent out-of-range numeric values from untyped callers unless the Rust function rejects them during conversion.
- Callback exports use napi-rs `ThreadsafeFunction` shape: `(error: Error | null, value) => void`. Native code generally emits successful values; hard failures reject/throw through the owning call.
+38 -35
View File
@@ -24,8 +24,8 @@ It follows the architecture terms from `docs/natives-architecture.md`:
`packages/natives/package.json` scripts:
- `bun scripts/build-native.ts` (`build`) → N-API build, addon install, generated declarations install, enum export patch.
- `bun scripts/embed-native.ts` (`embed:native`) → generate `native/embedded-addon.js` from built files.
- `bun scripts/build-native.ts` (`build`) → N-API build, addon install, generated declarations install, explicit ESM export and enum runtime patch.
- `bun scripts/embed-native.ts` (`embed:native`) → generate `native/embedded-addon.js` plus `native/embedded-addons.<tag>.tar.gz` from built files.
Root scripts include `build:native` as `bun --cwd=packages/natives run build`.
@@ -51,10 +51,10 @@ After napi-rs succeeds, `build-native.ts`:
1. resolves the built addon in the isolated output directory;
2. normalizes its name to `pi_natives.<platform>-<arch>(-variant).node` when needed;
3. installs the addon into `packages/natives/native/` with temp-file + rename semantics;
4. copies generated `index.js` and `index.d.ts` into `packages/natives/native/` when present;
5. runs `generateEnumExports()` to append enum runtime objects to `native/index.js`.
4. copies generated `index.d.ts` into `packages/natives/native/`;
5. runs `generateEnumExports()` to render explicit named ESM exports for classes/functions and runtime enum objects in the checked-in `native/index.js`.
Windows locked-DLL replacement failures are reported with an explicit close-running-processes hint.
Windows locked-DLL update failures are handled at runtime by staging install candidates into the versioned native cache; install/rename failures during local builds still include explicit file-operation diagnostics.
## Target/variant model and naming conventions
@@ -85,7 +85,7 @@ Runtime x64 candidate order also includes the unsuffixed default filename after
## Runtime flags
- `PI_NATIVE_VARIANT`: x64 runtime override; valid values are `modern` and `baseline`.
- `PI_COMPILED`: legacy compiled-mode signal. A populated embedded-addon manifest is also a compiled-mode signal and is the authoritative signal for Bun standalone builds that do not preserve `process.env.PI_COMPILED`.
- `PI_COMPILED`: legacy compiled-mode signal. A populated embedded-addon manifest is also a compiled-mode signal; compiled release builds additionally define `process.env.PI_COMPILED="true"` during `bun build --compile`.
## Build-time flags/options
@@ -115,11 +115,11 @@ Runtime x64 candidate order also includes the unsuffixed default filename after
4. **Compile**: run napi-rs against `crates/pi-natives` into an isolated output directory.
5. **Locate artifact**: accept the canonical filename or a single napi-rs-generated `pi_natives.<platform>-<arch>*.node` candidate.
6. **Install**: copy/rename addon into `packages/natives/native`.
7. **Install generated bindings**: copy `index.js`/`index.d.ts` if needed.
8. **Patch enums**: append generated enum runtime exports.
7. **Install generated declarations**: copy `index.d.ts`.
8. **Patch exports/enums**: regenerate explicit ESM exports and enum runtime objects.
9. **Cleanup**: remove the temporary build output directory.
Failure exits have explicit error text for invalid variants, failed napi build, missing/multiple output artifacts, generated binding install failure, and install/rename failure.
Failure exits have explicit error text for invalid variants, failed napi build, missing/multiple output artifacts, generated binding install failure, stripped CI ELF artifacts that still contain forbidden symbol/string-table sections, and install/rename failure.
### Embed lifecycle (`embed-native.ts`)
@@ -128,7 +128,7 @@ Failure exits have explicit error text for invalid variants, failed napi build,
- x64 looks for `modern` and `baseline` files;
- non-x64 looks for one default file.
3. **Validate availability**: at least one expected file must exist in `packages/natives/native`.
4. **Generate manifest** (`native/embedded-addon.js`) with Bun `file` imports and package version.
4. **Generate archive + manifest**: write `native/embedded-addons.<platform>-<arch>.tar.gz` containing all available target addon files and `native/embedded-addon.js` with package version, archive metadata, and file sizes.
5. **Runtime extraction ready** for compiled mode.
`--reset` writes the null manifest stub (`embeddedAddon = null`) without validating addon availability.
@@ -148,12 +148,13 @@ Typical local loop:
In compiled mode (`PI_COMPILED`, Bun embedded URL markers, or populated embedded manifest):
1. Loader computes versioned cache dir: `<getNativesDir()>/<packageVersion>`.
2. If embedded manifest matches current platform+version, loader may extract the selected embedded file into that versioned dir.
2. If embedded manifest matches current platform+version, loader extracts the selected file from `embedded-addons.<tag>.tar.gz` into that versioned dir when the cached file is absent or has the wrong size.
3. Runtime candidate order includes:
- extracted versioned cache path, if available,
- versioned cache dir,
- legacy compiled-binary dir (`%LOCALAPPDATA%/omp` on Windows, `~/.local/bin` elsewhere),
- package/executable directories.
4. First successfully loaded addon is returned.
4. First successfully loaded addon with the expected version sentinel is returned.
This is why packaging + runtime loader expectations must align: filenames, platform tags, CPU variants, and embedded manifest version must match what `native/index.js` probes.
@@ -161,13 +162,13 @@ This is why packaging + runtime loader expectations must align: filenames, platf
Generated declarations currently include exports from these Rust modules:
| Area | Representative JS exports | Rust source |
| ---------------------- | ------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------- |
| Search | `grep`, `search`, `hasMatch`, `fuzzyFind`, `glob`, `invalidateFsScanCache` | `grep.rs`, `fd.rs`, `glob.rs`, `fs_cache.rs` |
| AST | `astGrep`, `astEdit` | `ast.rs` |
| Text/highlight/tokens | `visibleWidth`, `truncateToWidth`, `highlightCode`, `countTokens` | `text.rs`, `highlight.rs`, `tokens.rs` |
| Shell/PTY/process/keys | `executeShell`, `Shell`, `PtySession`, `killTree`, `parseKey` | `shell.rs`, `pty.rs`, `ps.rs`, `keys.rs` |
| Media/system | `PhotonImage`, `encodeSixel`, clipboard, macOS appearance/power, `getWorkProfile`, ProjFS helpers | `image.rs`, `clipboard.rs`, `appearance.rs`, `power.rs`, `prof.rs`, `projfs_overlay.rs` |
| Area | Representative JS exports | Rust source |
| ---------------------- | ------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------- |
| Search/workspace | `grep`, `search`, `hasMatch`, `fuzzyFind`, `glob`, `listWorkspace`, `invalidateFsScanCache` | `grep.rs`, `fd.rs`, `glob.rs`, `workspace.rs`, `fs_cache.rs` |
| AST/block/summary | `astGrep`, `astEdit`, `blockRangeAt`, `summarizeCode` | `ast.rs`, `block.rs`, `summary.rs` |
| Text/highlight/tokens | `visibleWidth`, `truncateToWidth`, `highlightCode`, `countTokens` | `text.rs`, `highlight.rs`, `tokens.rs` |
| Shell/PTY/process/keys | `executeShell`, `Shell`, `PtySession`, `Process`, `parseKey`, `applyBashFixups` | `shell.rs`, `pty.rs`, `ps.rs`, `keys.rs` |
| Media/system/iso | `encodeSixel`, clipboard, macOS appearance/power, `getWorkProfile`, `isoBackend`, `isoStart`, `isoDiff` | `sixel.rs`, `clipboard.rs`, `appearance.rs`, `power.rs`, `prof.rs`, `iso.rs` |
## Failure behavior and diagnostics
@@ -186,18 +187,19 @@ Generated declarations currently include exports from these Rust modules:
- Unsupported platform tag: throws with supported platform list after probing fails.
- No candidate could load: throws with full candidate error list and mode-specific remediation hints.
- Embedded extraction problems: extraction mkdir/write errors are recorded and included in final diagnostics if load fails.
- Embedded extraction and Windows staging problems: archive/mkdir/write/copy errors are recorded and included in final diagnostics if load fails.
- Version mismatch: install/compiled loads that lack the package-version sentinel are rejected during candidate probing.
## Troubleshooting matrix
| Symptom | Likely cause | Verify | Fix |
| ---------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- | ----------------------------------------------------------------- | --------------------------------------------------------------------------------------------- |
| `Cannot find module` or dynamic library load error for every candidate | Missing release artifact, wrong platform tag, or stale compiled cache | Inspect loader error list and `packages/natives/native` filenames | Build correct target/variant; delete stale cache for the package version |
| Export is missing at runtime but present in TypeScript | Stale `.node` loaded, generated declarations newer than binary, or Rust export not compiled | Require the actual candidate and inspect `Object.keys(mod)` | Rebuild native package and remove stale candidate/cache paths |
| x64 machine loads baseline when modern expected | `PI_NATIVE_VARIANT=baseline`, no AVX2 detected, or modern file unavailable | Check env and filenames in `native/` | Build modern variant (`TARGET_VARIANT=modern ... build`) and ship it |
| Cross-build produces wrong-labeled binary | Mismatch between `CROSS_TARGET` and `TARGET_PLATFORM`/`TARGET_ARCH`, or missing x64 variant | Confirm env tuple and output filename | Re-run with consistent env values and explicit x64 `TARGET_VARIANT` |
| Compiled binary fails after upgrade | Stale extracted cache or embedded manifest version mismatch | Inspect `<getNativesDir()>/<version>` and loader error list | Delete versioned cache for the package version; regenerate embedded manifest during packaging |
| `embed:native` fails with `No native addons found` | Required platform artifact was not built before embedding | Check expected list in error text | Build at least one expected artifact for the target, then rerun `embed:native` |
| Symptom | Likely cause | Verify | Fix |
| ---------------------------------------------------------------------- | ------------------------------------------------------------------------------------------- | ----------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------- |
| `Cannot find module` or dynamic library load error for every candidate | Missing release artifact, wrong platform tag, or stale compiled cache | Inspect loader error list and `packages/natives/native` filenames | Build correct target/variant; delete stale cache for the package version |
| Export is missing at runtime but present in TypeScript | Stale `.node` loaded, generated declarations newer than binary, or Rust export not compiled | Require the actual candidate and inspect `Object.keys(mod)` | Rebuild native package and remove stale candidate/cache paths |
| x64 machine loads baseline when modern expected | `PI_NATIVE_VARIANT=baseline`, no AVX2 detected, or modern file unavailable | Check env and filenames in `native/` | Build modern variant (`TARGET_VARIANT=modern ... build`) and ship it |
| Cross-build produces wrong-labeled binary | Mismatch between `CROSS_TARGET` and `TARGET_PLATFORM`/`TARGET_ARCH`, or missing x64 variant | Confirm env tuple and output filename | Re-run with consistent env values and explicit x64 `TARGET_VARIANT` |
| Compiled binary fails after upgrade | Stale extracted cache, embedded archive mismatch, or embedded manifest version mismatch | Inspect `<getNativesDir()>/<version>` and loader error list | Delete versioned cache for the package version; regenerate embedded archive/manifest during packaging |
| `embed:native` fails with `No native addons found` | Required platform artifact was not built before embedding | Check expected list in error text | Build at least one expected artifact for the target, then rerun `embed:native` |
## Operational commands
@@ -211,6 +213,7 @@ TARGET_VARIANT=baseline bun --cwd=packages/natives run build
# Generate embedded addon manifest from built native files
bun --cwd=packages/natives run embed:native
# Output archive: packages/natives/native/embedded-addons.<platform>-<arch>.tar.gz
# Reset embedded manifest to null stub
bun --cwd=packages/natives run embed:native -- --reset
@@ -270,13 +273,13 @@ Workspaces that hardlinked a `.node` before GC retain access via the kernel inod
### Configuration (settings on `robomp.config.Settings`)
| Env var | Default | Effect |
| -------------------------------------------- | ------------------------ | ------------------------------------------------------------- |
| `ROBOMP_NATIVES_CACHE_ENABLED` | `true` | Master switch. When false the populate/capture hooks no-op and every workspace builds from scratch. |
| `ROBOMP_NATIVES_CACHE_ROOT` | `/data/cache/pi-natives` | Cache root directory. Must be `root:omp 02770` for cross-slot reads. |
| `ROBOMP_NATIVES_CACHE_MAX_ENTRIES_PER_REPO` | `8` | LRU entry-count cap, per repo slug. |
| `ROBOMP_NATIVES_CACHE_MAX_BYTES` | `4294967296` (4 GiB) | LRU byte cap, per repo slug. |
| `ROBOMP_NATIVES_CACHE_GC_INTERVAL_SECONDS` | `3600` | Period of the background GC loop in `WorkerPool`. |
| Env var | Default | Effect |
| ------------------------------------------- | ------------------------ | --------------------------------------------------------------------------------------------------- |
| `ROBOMP_NATIVES_CACHE_ENABLED` | `true` | Master switch. When false the populate/capture hooks no-op and every workspace builds from scratch. |
| `ROBOMP_NATIVES_CACHE_ROOT` | `/data/cache/pi-natives` | Cache root directory. Must be `root:omp 02770` for cross-slot reads. |
| `ROBOMP_NATIVES_CACHE_MAX_ENTRIES_PER_REPO` | `8` | LRU entry-count cap, per repo slug. |
| `ROBOMP_NATIVES_CACHE_MAX_BYTES` | `4294967296` (4 GiB) | LRU byte cap, per repo slug. |
| `ROBOMP_NATIVES_CACHE_GC_INTERVAL_SECONDS` | `3600` | Period of the background GC loop in `WorkerPool`. |
### Manual invalidation
+45 -70
View File
@@ -1,83 +1,64 @@
# Natives media + system utilities
This document covers the media/system/conversion exports in `@oh-my-pi/pi-natives`: image processing, HTML conversion, clipboard access, token counting, macOS appearance/power helpers, ProjFS helpers, and work profiling.
This document covers the media/system/conversion exports currently present in `@oh-my-pi/pi-natives`: terminal SIXEL image encoding, HTML conversion, clipboard access, token counting, macOS appearance/power helpers, and work profiling.
## Implementation files
- `crates/pi-natives/src/image.rs`
- `crates/pi-natives/src/sixel.rs`
- `crates/pi-natives/src/html.rs`
- `crates/pi-natives/src/clipboard.rs`
- `crates/pi-natives/src/tokens.rs`
- `crates/pi-natives/src/appearance.rs`
- `crates/pi-natives/src/power.rs`
- `crates/pi-natives/src/projfs_overlay.rs`
- `crates/pi-natives/src/prof.rs`
- `crates/pi-natives/src/task.rs`
- `packages/natives/native/index.d.ts`
> Note: there is no `crates/pi-natives/src/work.rs`; work profiling is implemented in `prof.rs` and fed by instrumentation in `task.rs`.
There is no native `PhotonImage` class, `image.rs`, or ProjFS overlay helper module in the current `pi-natives` addon. General-purpose image decode/resize/encode is expected to live outside this native surface; the native image export here is only terminal SIXEL encoding.
## JS API ↔ Rust export/module mapping
| JS export | Rust N-API export | Rust module |
| --------------------------------------------------- | ------------------------------ | ------------------- |
| `PhotonImage.parse(bytes)` | `PhotonImage::parse` | `image.rs` |
| `PhotonImage#resize(width, height, filter)` | `PhotonImage::resize` | `image.rs` |
| `PhotonImage#encode(format, quality)` | `PhotonImage::encode` | `image.rs` |
| `encodeSixel(bytes, targetWidthPx, targetHeightPx)` | `encode_sixel` | `image.rs` |
| `htmlToMarkdown(html, options?)` | `html_to_markdown` | `html.rs` |
| `copyToClipboard(text)` | `copy_to_clipboard` | `clipboard.rs` |
| `readImageFromClipboard()` | `read_image_from_clipboard` | `clipboard.rs` |
| `countTokens(input, encoding?)` | `count_tokens` | `tokens.rs` |
| `detectMacOSAppearance()` | `detect_mac_os_appearance` | `appearance.rs` |
| `MacAppearanceObserver.start(callback)` | `MacAppearanceObserver::start` | `appearance.rs` |
| `MacOSPowerAssertion.start(options?)` | `MacOSPowerAssertion::start` | `power.rs` |
| `projfsOverlayProbe/start/stop` | ProjFS exports | `projfs_overlay.rs` |
| `getWorkProfile(lastSeconds)` | `get_work_profile` | `prof.rs` |
| JS export | Rust N-API export | Rust module |
| ------------------------------------- | ------------------------------ | --------------- |
| `encodeSixel(bytes, width, height)` | `encode_sixel` | `sixel.rs` |
| `htmlToMarkdown(html, options?)` | `html_to_markdown` | `html.rs` |
| `copyToClipboard(text)` | `copy_to_clipboard` | `clipboard.rs` |
| `readImageFromClipboard()` | `read_image_from_clipboard` | `clipboard.rs` |
| `countTokens(input, encoding?)` | `count_tokens` | `tokens.rs` |
| `detectMacOSAppearance()` | `detect_mac_os_appearance` | `appearance.rs` |
| `MacAppearanceObserver.start(cb)` | `MacAppearanceObserver::start` | `appearance.rs` |
| `MacOSPowerAssertion.start(options?)` | `MacOSPowerAssertion::start` | `power.rs` |
| `getWorkProfile(lastSeconds)` | `get_work_profile` | `prof.rs` |
## Data format boundaries and conversions
### Image (`image`)
### SIXEL image encoding (`sixel`)
- **JS input boundary**: `Uint8Array` encoded image bytes for `PhotonImage.parse` and `encodeSixel`.
- **Rust decode boundary**: bytes are copied/read, format is guessed with `ImageReader::with_guessed_format()`, then decoded to `DynamicImage`.
- **In-memory state**: `PhotonImage` stores `Arc<DynamicImage>`.
- **Output boundary**:
- `PhotonImage#encode(format, quality)` returns a promise for encoded bytes (`Vec<u8>` in Rust; generated TS currently declares `Promise<Array<number>>`).
- `encodeSixel(...)` returns a SIXEL escape string synchronously.
- **JS input boundary**: `Uint8Array` containing encoded image bytes.
- **Rust decode boundary**: format is guessed with `ImageReader::with_guessed_format()`, then decoded to `DynamicImage`.
- **Resize boundary**: image is resized with `resize_exact(..., FilterType::Lanczos3)` only when source dimensions differ from `targetWidthPx`/`targetHeightPx`.
- **Output boundary**: `encodeSixel(...)` returns a SIXEL escape string synchronously.
Format IDs:
- `0`: PNG
- `1`: JPEG
- `2`: WebP
- `3`: GIF
Encoding behavior:
- JPEG uses the provided `quality` with `JpegEncoder::new_with_quality`.
- WebP uses the `webp` crate encoder with `quality` as `f32` in the same 0..=100 range.
- PNG/GIF ignore `quality`.
- Invalid dimensions for SIXEL (`0` width or height) fail with `Target SIXEL dimensions must be greater than zero`.
Supported decode formats are whatever the compiled `image` crate supports for `ImageReader` in this build (commonly PNG/JPEG/WebP/GIF). Invalid target dimensions (`0` width or height) fail with `Target SIXEL dimensions must be greater than zero`.
### HTML conversion (`html`)
- **JS input boundary**: HTML `string` + optional `{ cleanContent?: boolean; skipImages?: boolean }`.
- **Rust conversion boundary**: conversion is scheduled through `task::blocking("html_to_markdown", (), ...)`.
- **Rust conversion boundary**: conversion is scheduled through `task::blocking("html_to_markdown", (), ...)`; there is no timeout/abort option on this export.
- **Output boundary**: Markdown `string` promise.
Conversion behavior:
- `cleanContent` defaults to `false`.
- When `cleanContent=true`, preprocessing uses `PreprocessingPreset::Aggressive` and hard-removal flags for navigation/forms.
- `skipImages` defaults to `false`.
- When `cleanContent=true`, preprocessing is enabled with `PreprocessingPreset::Aggressive`, `remove_navigation=true`, and `remove_forms=true`.
- `skipImages` defaults to `false` and is passed to `html_to_markdown_rs::ConversionOptions`.
### Clipboard (`clipboard`)
- `copyToClipboard(text)` is a synchronous native call using `arboard::Clipboard::set_text`.
- `readImageFromClipboard()` runs in `task::blocking("clipboard.read_image", (), ...)`.
- Image read returns `null`/`undefined` when `arboard` reports `ContentNotAvailable`.
- Successful image read re-encodes clipboard RGBA data as PNG and returns `{ data: Uint8Array, mimeType: "image/png" }`.
- Successful image read converts clipboard RGBA data into PNG bytes and returns `{ data: Uint8Array, mimeType: "image/png" }`.
- Clipboard access or image encoding failures reject/throw as native errors.
There is no current `packages/natives` TS wrapper that emits OSC52, handles Termux, or suppresses native clipboard failures. Any best-effort clipboard policy must live in consumers.
@@ -85,25 +66,19 @@ There is no current `packages/natives` TS wrapper that emits OSC52, handles Term
### Tokens (`tokens`)
- `countTokens(input, encoding?)` accepts a single string or an array of strings.
- Arrays return one aggregate token count; encoding work is parallelized in Rust.
- Arrays return one aggregate token count; array elements are encoded in parallel via rayon.
- Default encoding is `O200kBase`; `Cl100kBase` is also exported.
- The implementation uses ordinary encoding, not special-token handling.
- The implementation uses `encode_ordinary`, not special-token handling.
- BPE tables are initialized once through `LazyLock` and reused.
### macOS appearance and power helpers
- `detectMacOSAppearance()` returns `"dark"`, `"light"`, or `null` on non-macOS.
- `MacAppearanceObserver.start(callback)` returns a handle with `stop()`; on macOS it uses distributed notifications plus a 2-second polling fallback, and on non-macOS it is a no-op observer.
- `MacOSPowerAssertion.start(options?)` returns a handle with `stop()`; on macOS it acquires an IOKit assertion, and on other platforms it is a no-op handle.
- `MacOSPowerAssertion.start(options?)` returns a handle with `stop()`; on macOS it acquires one or more IOKit assertions, and on other platforms it is a no-op handle.
- Power assertion options are `{ reason?, idle?, system?, user?, display? }`. If every boolean is unset or omitted, `idle` behavior is used by default.
### Windows ProjFS helpers
- `projfsOverlayProbe()` reports whether ProjFS APIs are available.
- `projfsOverlayStart(lowerRoot, projectionRoot)` starts an overlay.
- `projfsOverlayStop(projectionRoot)` stops an overlay session.
These helpers are platform-specific; availability must be checked before relying on overlay behavior.
### Work profiling (`work`)
### Work profiling (`prof`)
- **Collection boundary**: profiling samples are produced by `profile_region(tag)` guards in `task::blocking` and `task::future`.
- **Storage format**: fixed-size circular buffer (`MAX_SAMPLES = 10_000`) storing stack path, duration, and timestamp.
@@ -115,25 +90,25 @@ These helpers are platform-specific; availability must be checked before relying
## Lifecycle and state transitions
### Image lifecycle
### SIXEL lifecycle
1. `PhotonImage.parse(bytes)` schedules a blocking decode task (`image.decode`).
2. On success, a native `PhotonImage` handle exists in JS.
3. `resize(...)` creates a new native handle (`image.resize`); old and new handles can coexist.
4. `encode(...)` schedules `image.encode` and materializes bytes without mutating image dimensions.
5. `encodeSixel(...)` decodes, optionally resizes to exact target dimensions with Lanczos3, and returns SIXEL text synchronously.
1. `encodeSixel(bytes, targetWidthPx, targetHeightPx)` validates target dimensions.
2. Rust guesses and decodes the encoded image.
3. Image is resized exactly to the target dimensions when needed.
4. Pixels are converted to RGBA8 and encoded with `icy_sixel::sixel_encode`.
5. The SIXEL escape string is returned synchronously.
Failure transitions:
- Format detection/decode failure rejects parse promise or throws from SIXEL encoding.
- Encode failure rejects encode promise.
- Invalid SIXEL dimensions throw.
- Format detection/decode failure throws.
- Invalid target dimensions throw.
- SIXEL encoding failure throws with `Failed to encode SIXEL: ...`.
### HTML lifecycle
1. `htmlToMarkdown(html, options)` schedules a blocking conversion task.
2. Conversion runs with defaulted options (`cleanContent=false`, `skipImages=false`) unless specified.
3. Returns markdown string or rejects.
3. Returns markdown string or rejects with `Conversion error: ...`.
### Clipboard lifecycle
@@ -154,11 +129,11 @@ Failure transitions:
## Unsupported operations and error propagation
### Image
### SIXEL
- Unsupported decode input or corrupted bytes: strict failure.
- Invalid SIXEL target dimensions: strict failure.
- No JS fallback path in the natives package.
- Unsupported or corrupted image input is a strict failure.
- Invalid SIXEL target dimensions are a strict failure.
- No JS fallback path is exposed by the natives package.
### HTML
@@ -180,4 +155,4 @@ Failure transitions:
- Clipboard access depends on OS/session support exposed through `arboard`.
- macOS appearance and power helpers intentionally return no-op/null behavior on unsupported platforms.
- ProjFS helpers are Windows-specific and should be gated by `projfsOverlayProbe()`.
- ProjFS is not exposed by this media/system native utility surface. Isolation backend selection, including any ProjFS support, lives in the separate `iso` subsystem.
+19 -22
View File
@@ -12,7 +12,7 @@ This document describes how `crates/pi-natives` schedules native work and how ca
- `crates/pi-natives/src/shell.rs`
- `crates/pi-natives/src/pty.rs`
- `crates/pi-natives/src/html.rs`
- `crates/pi-natives/src/image.rs`
- `crates/pi-natives/src/sixel.rs`
- `crates/pi-natives/src/clipboard.rs`
- `crates/pi-natives/src/text.rs`
- `crates/pi-natives/src/ps.rs`
@@ -36,8 +36,8 @@ This document describes how `crates/pi-natives` schedules native work and how ca
3. `CancelToken` / `AbortToken` / `AbortReason`
- `CancelToken::new(timeout_ms, signal)` combines an optional deadline and optional JS `AbortSignal` converted from `Unknown`.
- `CancelToken::heartbeat()` is cooperative cancellation for blocking loops.
- `CancelToken::wait()` asynchronously waits for signal, timeout, or Ctrl-C.
- `CancelToken::emplace_abort_token()` creates an abortable flag when a later `Shell.abort()`/internal bridge needs one.
- `CancelToken::wait()` asynchronously waits for signal or timeout.
- `CancelToken::emplace_abort_token()` creates an abortable flag when `AbortSignal`, `Shell.abort()`, or an internal bridge needs one.
- `AbortToken::abort(reason)` lets external code request abort.
## `blocking` vs `future`: execution model and selection
@@ -48,8 +48,6 @@ Use when work is CPU-heavy or fundamentally synchronous/blocking:
- regex/file scanning (`grep`, `glob`, `fuzzyFind`)
- ast-grep search/edit worker work
- PTY loop internals through `tokio::task::spawn_blocking`
- image decode/resize/encode
- HTML conversion
- clipboard image read
@@ -65,7 +63,7 @@ Use when work must `await` async operations:
- shell session orchestration (`Shell.run`, `executeShell`)
- PTY outer promise (`PtySession.start`) before it enters `spawn_blocking`
- task racing (`tokio::select!`) between completion and cancellation
- async task orchestration that must bridge completion and cancellation
Behavior:
@@ -74,20 +72,20 @@ Behavior:
## JS API ↔ Rust export mapping (task/cancel relevant)
| JS-facing API | Rust export | Scheduler | Cancellation hookup |
| --------------------------------------- | ------------------------------------ | -------------------------------------------------------------- | ---------------------------------------------------------------------------------------- |
| `grep(options, onMatch?)` | `grep` | `task::blocking("grep", ct, ...)` | `CancelToken::new(options.timeoutMs, options.signal)` + heartbeat checks |
| `glob(options, onMatch?)` | `glob` | `task::blocking("glob", ct, ...)` | `CancelToken::new(...)` + heartbeat checks |
| `fuzzyFind(options)` | `fuzzy_find` | `task::blocking("fuzzy_find", ct, ...)` | `CancelToken::new(...)` + heartbeat checks |
| `astGrep(options)` / `astEdit(options)` | ast exports | blocking worker path | timeout/signal fields are accepted by options and checked cooperatively in worker loops |
| `Shell#run(options, onChunk?)` | `Shell::run` | `task::future(env, "shell.run", ...)` | `ct.wait()` raced against run task; bridges to Tokio cancellation token and `AbortToken` |
| `executeShell(options, onChunk?)` | `execute_shell` | `task::future(env, "shell.execute", ...)` | same cancel race and 2s graceful window |
| `PtySession#start(options, onChunk?)` | `PtySession::start` | `task::future(env, "pty.start", ...)` + inner `spawn_blocking` | `CancelToken` checked in sync PTY loop via `heartbeat()` |
| `htmlToMarkdown(html, options?)` | `html_to_markdown` | `task::blocking("html_to_markdown", (), ...)` | none (`()` token) |
| `PhotonImage.parse/encode/resize` | `PhotonImage::{parse,encode,resize}` | `task::blocking(...)` | none (`()` token) |
| `readImageFromClipboard()` | `read_image_from_clipboard` | `task::blocking("clipboard.read_image", (), ...)` | none (`()` token) |
| JS-facing API | Rust export | Scheduler | Cancellation hookup |
| --------------------------------------- | --------------------------- | -------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------ |
| `grep(options, onMatch?)` | `grep` | `task::blocking("grep", ct, ...)` | `CancelToken::new(options.timeoutMs, options.signal)` + heartbeat checks |
| `glob(options, onMatch?)` | `glob` | `task::blocking("glob", ct, ...)` | `CancelToken::new(...)` + heartbeat checks |
| `fuzzyFind(options)` | `fuzzy_find` | `task::blocking("fuzzy_find", ct, ...)` | `CancelToken::new(...)` + heartbeat checks |
| `astGrep(options)` / `astEdit(options)` | ast exports | blocking worker path | timeout/signal fields are accepted by options and checked cooperatively in worker loops |
| `Shell#run(options, onChunk?)` | `Shell::run` | `task::future(env, "shell.run", ...)` | JS `CancelToken` is converted into `pi_shell::cancel::CancelToken`; shell races it against command completion and descendant cleanup |
| `executeShell(options, onChunk?)` | `execute_shell` | `task::future(env, "shell.execute", ...)` | same cancel race and 2s graceful window |
| `PtySession#start(options, onChunk?)` | `PtySession::start` | `task::future(env, "pty.start", ...)` + inner `spawn_blocking` | `CancelToken` checked in sync PTY loop via `heartbeat()` |
| `htmlToMarkdown(html, options?)` | `html_to_markdown` | `task::blocking("html_to_markdown", (), ...)` | none (`()` token) |
| `encodeSixel(...)` | `encode_sixel` | synchronous native function | none |
| `readImageFromClipboard()` | `read_image_from_clipboard` | `task::blocking("clipboard.read_image", (), ...)` | none (`()` token) |
`text.rs`, `tokens.rs`, `keys.rs`, most `ps.rs` functions, and synchronous utility exports do not use `task::blocking`/`task::future` and therefore do not participate in this cancellation path.
`text.rs`, `tokens.rs`, `keys.rs`, most `ps.rs` functions, SIXEL encoding, and synchronous utility exports do not use `task::blocking`/`task::future` cancellation and therefore do not participate in this cancellation path.
## Cancellation lifecycle and state transitions
@@ -102,7 +100,6 @@ Created
Running
├─ heartbeat()/wait() sees signal -> AbortReason::Signal
├─ heartbeat()/wait() sees deadline -> AbortReason::Timeout
├─ wait() sees Ctrl-C -> AbortReason::User
└─ no abort -> continue
Aborted
@@ -118,8 +115,8 @@ Aborted
- **Mid-execution**:
- `blocking`: next `heartbeat()` returns `Err("Aborted: ...")`.
- `future`: `ct.wait()` branch wins `select!`, then code cancels subordinate async machinery.
- shell: cancellation triggers a Tokio cancellation token, waits up to 2 seconds, then aborts the task if needed.
- PTY: heartbeat failure or `kill()` terminates PTY child/process tree and drains output briefly.
- shell: cancellation triggers a Tokio cancellation token, sends descendant termination waves, waits up to 2 seconds for the command task, then aborts the task if needed.
- PTY: heartbeat failure or `kill()` terminates PTY child/process targets and drains output briefly.
## Heartbeat expectations for long-running loops
+69 -58
View File
@@ -1,11 +1,14 @@
# Natives Shell, PTY, Process, and Key Internals
This document covers the execution/process/terminal primitives in `@oh-my-pi/pi-natives`: `shell`, `pty`, `ps`, and `keys`, using the architecture terms from `docs/natives-architecture.md`.
This document covers execution/process/terminal primitives in `@oh-my-pi/pi-natives`: `shell`, `pty`, `ps`, and `keys`, using the architecture terms from `docs/natives-architecture.md`.
## Implementation files
- `crates/pi-natives/src/shell.rs`
- `crates/pi-natives/src/shell/windows.rs` (Windows-only PATH enrichment)
- `crates/pi-shell/src/shell.rs`
- `crates/pi-shell/src/fixup.rs`
- `crates/pi-shell/src/windows.rs` (Windows-only PATH enrichment)
- `crates/pi-shell/src/process.rs`
- `crates/pi-natives/src/pty.rs`
- `crates/pi-natives/src/ps.rs`
- `crates/pi-natives/src/keys.rs`
@@ -15,19 +18,24 @@ This document covers the execution/process/terminal primitives in `@oh-my-pi/pi-
## Layer ownership
- **Package entrypoint** (`packages/natives/native/index.js`): loads the `.node` addon and exports generated N-API bindings.
- **Rust N-API module layer** (`crates/pi-natives/src/*`): shell/PTY process execution, process-tree traversal/termination, and key-sequence parsing.
- **Rust N-API module layer** (`crates/pi-natives/src/*`): JS-facing shell/PTY/process/key exports and callback bridging.
- **Runtime core** (`crates/pi-shell/src/*`): brush shell execution, cancellation cleanup, minimizer integration, command fixups, and cross-platform process references.
- **Consumers** (`packages/coding-agent`, `packages/tui`): higher-level session policy, output artifact/minimizer handling, render policy, and UI key handling.
## Shell subsystem (`shell`)
### API model
Two execution modes are exposed:
Shell execution modes:
1. **One-shot** via `executeShell(options, onChunk?)`.
2. **Persistent session** via `new Shell(options?)` then `shell.run(...)` repeatedly.
Both stream output through a threadsafe callback and return `{ exitCode?, cancelled, timedOut, minimized? }`.
Both stream merged stdout/stderr text through a threadsafe callback and return `{ exitCode?, cancelled, timedOut, minimized? }`.
Related synchronous helper:
- `applyBashFixups(command)` strips safe trailing `| head`/`| tail` pipeline caps and redundant trailing `2>&1` according to `pi_shell::fixup` rules. It returns `{ command, stripped }` and does not execute anything.
`ShellOptions` supports `sessionEnv`, `snapshotPath`, and optional output `minimizer`. `ShellExecuteOptions` supports command-scoped `env`, session-level `sessionEnv`, `snapshotPath`, timeout/signal, and optional minimizer. `ShellRunOptions` supports command, cwd, command-scoped env, timeout, and signal.
@@ -35,19 +43,19 @@ Both stream output through a threadsafe callback and return `{ exitCode?, cancel
Rust creates `brush_core::Shell` with:
- non-interactive, non-login mode,
- `no_profile` and `no_rc`,
- `do_not_inherit_env: true`,
- inherited environment disabled (`do_not_inherit_env: true`), followed by explicit environment reconstruction from host env,
- profile and rc loading skipped,
- bash-mode builtins, with `exec` and `suspend` disabled,
- explicit environment reconstruction from host env,
- skip-list for shell-sensitive vars (`PS1`, `PWD`, `SHLVL`, bash function exports, etc.).
- native `sleep` and `timeout` builtins registered,
- skip-list for shell-sensitive vars (`PS1`, `PWD`, `SHLVL`, bash function exports, etc.),
- a non-exported `env="$env"` fallback so PowerShell-style `$env:NAME` survives brush parameter expansion unless the user shadows `env`.
Session env behavior:
- `ShellOptions.sessionEnv` / one-shot `sessionEnv` is applied at session creation.
- `ShellRunOptions.env` / one-shot `env` is command-scoped (`EnvironmentScope::Command`) and popped after the command.
- `PATH` is merged specially on Windows with case-insensitive dedupe.
- Windows-only path enrichment (`shell/windows.rs`) appends discovered Git-for-Windows paths when present and not already included.
- Windows-only path enrichment (`pi-shell/src/windows.rs`) appends discovered Git-for-Windows paths when present and not already included.
- `snapshotPath`, when present, is sourced during session creation with stdout/stderr/stdin wired to null files.
### Runtime lifecycle and state transitions
@@ -58,7 +66,7 @@ Persistent shell (`Shell.run`) uses this state machine:
- **Running**: first `run()` lazily creates a session, stores an abort token, executes command.
- **Completed + keepalive**: if execution control flow is normal, abort state is cleared and session is reused.
- **Completed + teardown**: if control flow is loop/script/shell-exit related, session is dropped.
- **Cancelled/Timed out**: run task is cancelled, grace wait is 2 seconds, task may be force-aborted, session is dropped if lock can be acquired.
- **Cancelled/Timed out**: Tokio cancellation token is triggered, descendants started after the baseline snapshot receive termination waves, a 2-second graceful wait is allowed, the task may be aborted, and the persistent session is dropped if the lock can be acquired.
- **Error**: session is dropped.
One-shot shell (`executeShell`) always creates and drops a fresh session per call.
@@ -67,14 +75,15 @@ One-shot shell (`executeShell`) always creates and drops a fresh session per cal
- Stdout/stderr are routed into a shared pipe and read concurrently.
- Reader decodes UTF-8 incrementally; invalid byte sequences emit `U+FFFD` replacement chunks.
- The command runs in a new process group policy.
- The command runs with `ProcessGroupPolicy::NewProcessGroup`.
- After the foreground command completes, the reader drains until EOF, 250ms of idle output, or 2s maximum; reader shutdown then gets a 250ms timeout.
- Optional minimizer configuration can capture and rewrite output. When minimization occurs, the result includes `minimized` with filter name, replacement text, original text, and byte counts.
- Consumers are responsible for persisting or displaying minimizer artifacts; the native result only carries the data.
### Cancellation, timeout, and abort
- `CancelToken` is constructed from `timeoutMs` and optional `AbortSignal`.
- On cancellation/timeout, shell cancellation token is triggered, then task gets a 2-second graceful window before forced abort.
- `CancelToken` is constructed from `timeoutMs` and optional `AbortSignal`, then converted into the shared `pi_shell::cancel::CancelToken`.
- On cancellation/timeout, shell cancellation token is triggered, descendant cleanup runs, then the task gets a 2-second graceful window before forced abort.
- Structured result flags are used:
- timeout -> `exitCode` omitted, `timedOut: true`.
- abort signal / `Shell.abort()` -> `exitCode` omitted, `cancelled: true`.
@@ -107,7 +116,7 @@ Common surfaced errors include:
- `resize(cols, rows)`
- `kill()`
`PtyStartOptions` supports `command`, optional `cwd`, optional `env`, `timeoutMs`, `signal`, `cols`, and `rows`.
`PtyStartOptions` supports `command`, optional `cwd`, optional `env`, `timeoutMs`, `signal`, `cols`, `rows`, and optional `shell`. The default shell is `sh`.
### Runtime lifecycle and state transitions
@@ -126,11 +135,15 @@ Concurrency guard:
### Spawn/attach/write/read/terminate patterns
- PTY opened via `portable_pty::native_pty_system().openpty(...)`.
- Command currently runs as `sh -lc <command>` with optional `cwd` and env overrides.
- Default size is `120x40`; dimensions are clamped (`cols 20..400`, `rows 5..200`).
- On Windows, `openpty()` is run on a helper thread with a 5s startup timeout; timeout rejects with `PTY creation timed out (5s). ConPTY may be unavailable on this system.`
- Command runs through the configured shell:
- `cmd.exe`/`cmd` gets `/c`,
- `powershell`/`pwsh` gets `-Command`,
- other shells get `-lc`.
- Default size is `120x40`; dimensions are clamped (`cols 20..400`, `rows 5..200`) on start and resize.
- `write()` sends raw bytes to PTY stdin.
- `resize()` sends a control message and clamps dimensions again.
- `kill()` sends a control message that marks the run cancelled and terminates the child/process tree.
- `kill()` sends a control message that marks the run cancelled and terminates PTY process targets.
Output path:
@@ -140,8 +153,9 @@ Output path:
Termination path:
- Unix: terminate process group when known, terminate child tree, call child kill, then repeat with SIGKILL.
- Non-Unix: terminate child tree, call child kill, then repeat with SIGKILL-equivalent process-tree helper.
- `terminate_pty_processes` targets the PTY process group when available and the child pid when available.
- It sends the platform `TERM_SIGNAL`, calls `child.kill()`, then sends the platform `KILL_SIGNAL`.
- On Windows, ConPTY input is closed before dropping the master; master drop is offloaded to a background thread and waited for up to 2s to avoid deadlock.
### Cancellation and timeout semantics
@@ -149,12 +163,14 @@ Termination path:
- Loop calls `ct.heartbeat()` periodically with a 16ms maximum wait cadence.
- Timeout classification is based on the heartbeat error string containing `Timeout`.
- Cancellation/kill starts a 300ms post-cancel drain window; normal child exit starts a 300ms post-exit drain window.
- Final reader drain is 50ms on non-Windows and 500ms on Windows.
### Failure behavior
Error surfaces include:
- PTY allocation/open failure,
- Windows PTY startup timeout,
- PTY spawn failure,
- writer/reader acquisition failure,
- child status/wait failures,
@@ -165,38 +181,27 @@ Control call failures when not running:
- `write/resize/kill` return `PTY session is not running`.
## Process-tree subsystem (`ps`)
## Process subsystem (`ps`)
### API model
- `killTree(pid, signal) -> number`
- `listDescendants(pid) -> number[]`
Current JS surface is the `Process` class:
### Platform-specific implementation
- `Process.fromPid(pid) -> Process | null`
- `Process.fromPath(path) -> Process[]`
- getters: `pid`, `ppid`
- methods: `args()`, `killTree(signal?)`, `terminate(options?)`, `waitForExit(options?)`, `groupId()`, `children()`, `status()`
- **Linux**: recursively reads `/proc/<pid>/task/<pid>/children`.
- **macOS**: uses `libproc` `proc_listchildpids`.
- **Windows**: snapshots process table with `CreateToolhelp32Snapshot`, builds parent->children map, terminates with `OpenProcess(PROCESS_TERMINATE)` + `TerminateProcess`.
`ProcessTerminateOptions` supports `{ group?, gracefulMs?, timeoutMs?, signal? }`. `ProcessWaitOptions` supports `{ timeoutMs?, signal? }`.
### Kill-tree behavior
### Behavior
- Descendants are collected recursively.
- Kill order is bottom-up (deepest descendants first).
- Root pid is killed last.
- Return value is count of successful terminations.
- `killTree(signal?)` sends the requested signal to the process and descendants, children first; on Windows the signal argument is ignored and processes are terminated via `TerminateProcess`.
- `terminate(options?)` is async. By default it uses a 1000ms graceful phase and a 5000ms post-hard-kill wait. Passing `gracefulMs < 0` skips the graceful phase.
- `waitForExit(options?)` resolves `true` when the process exits and `false` on timeout.
- `status()` returns `"running"` or `"exited"`.
Signal behavior:
- POSIX: provided `signal` is passed to `kill`.
- Windows: `signal` is ignored; termination is unconditional process terminate.
### Failure behavior
This module is intentionally non-throwing at API surface for ordinary process misses:
- missing/inaccessible process tree branches are skipped,
- per-pid kill failures are counted as unsuccessful,
- lookup miss typically yields `[]` from `listDescendants` and `0` from `killTree`.
The platform-specific implementation lives in `pi_shell::process`; `crates/pi-natives/src/ps.rs` is a N-API shim plus re-exports used by PTY termination.
## Key parsing subsystem (`keys`)
@@ -239,19 +244,25 @@ Layout behavior:
### Shell + PTY + Process
| JS API | Rust N-API export | Notes |
| --------------------------------- | -------------------------------------- | ----------------------------------------- |
| `executeShell(options, onChunk?)` | `executeShell` (`execute_shell`) | One-shot shell execution |
| `new Shell(options?)` | `Shell` class | Persistent shell session |
| `shell.run(options, onChunk?)` | `Shell::run` | Reuses session on keepalive control flow |
| `shell.abort()` | `Shell::abort` | Aborts active run for that shell instance |
| `new PtySession()` | `PtySession` class | Stateful PTY session |
| `pty.start(options, onChunk?)` | `PtySession::start` | Interactive PTY run |
| `pty.write(data)` | `PtySession::write` | Raw stdin passthrough |
| `pty.resize(cols, rows)` | `PtySession::resize` | Clamped terminal dimensions |
| `pty.kill()` | `PtySession::kill` | Force-kills active PTY child |
| `killTree(pid, signal)` | `killTree` (`kill_tree`) | Children-first process tree termination |
| `listDescendants(pid)` | `listDescendants` (`list_descendants`) | Recursive descendants listing |
| JS API | Rust N-API export | Notes |
| --------------------------------- | --------------------------------------- | ----------------------------------------- |
| `executeShell(options, onChunk?)` | `executeShell` (`execute_shell`) | One-shot shell execution |
| `new Shell(options?)` | `Shell` class | Persistent shell session |
| `shell.run(options, onChunk?)` | `Shell::run` | Reuses session on keepalive control flow |
| `shell.abort()` | `Shell::abort` | Aborts active run for that shell instance |
| `applyBashFixups(command)` | `applyBashFixups` (`apply_bash_fixups`) | Synchronous command rewrite helper |
| `new PtySession()` | `PtySession` class | Stateful PTY session |
| `pty.start(options, onChunk?)` | `PtySession::start` | Interactive PTY run |
| `pty.write(data)` | `PtySession::write` | Raw stdin passthrough |
| `pty.resize(cols, rows)` | `PtySession::resize` | Clamped terminal dimensions |
| `pty.kill()` | `PtySession::kill` | Terminates active PTY child/targets |
| `Process.fromPid(pid)` | `Process::from_pid` | Stable process reference lookup |
| `Process.fromPath(path)` | `Process::from_path` | Executable-path process lookup |
| `process.killTree(signal?)` | `Process::kill_tree` | Children-first process tree termination |
| `process.terminate(options?)` | `Process::terminate` | Graceful then hard process termination |
| `process.waitForExit(options?)` | `Process::wait_for_exit` | Async exit wait |
| `process.children()` | `Process::children` | Direct children as `Process[]` |
| `process.status()` | `Process::status` | `running` / `exited` |
### Keys
+10 -7
View File
@@ -77,9 +77,8 @@ Terminology follows `docs/natives-architecture.md`:
- Output modes:
- `content` -> one `GrepMatch` per hit.
- `count` and `filesWithMatches` map to count-style entries (`lineNumber=0`, `line=""`, `matchCount` set).
- Limits:
- Global `offset` and `maxCount` apply across files.
- Parallel path is used only when `maxCount` is unset and `offset == 0`; otherwise sequential path preserves deterministic global offset/limit semantics.
- `offset` and `maxCount` are applied during aggregation across sorted file results.
- Directory searches use parallel filesystem walking/searching, then aggregate per-file results to preserve global offset/limit semantics in the returned result and callback stream.
### Result shaping back to JS
@@ -158,11 +157,15 @@ These exports are direct native APIs used by tooling; they are not mediated by a
## 4) Shared scan/cache lifecycle (`fs_cache`)
`fs_cache` stores scan results as normalized relative entries (`path`, `fileType`, optional `mtime`) keyed by:
`fs_cache` stores scan results as normalized relative entries (`path`, `fileType`, optional `mtime` and regular-file `size`) keyed by:
- canonical search root,
- `include_hidden`,
- `use_gitignore`.
- `use_gitignore`,
- `skip_node_modules`,
- scan detail (`Minimal` vs `Full`).
`follow_links` affects a fresh scan but is not currently part of the cache key.
### Cache state transitions
@@ -193,7 +196,7 @@ These are pure, in-memory utilities.
- `text.rs` owns terminal-cell semantics:
- ANSI sequence parsing,
- grapheme-aware width and slicing,
- wrap/truncate/sanitize behavior,
- wrap/truncate/slice behavior,
- explicit tab-width parameter on width-sensitive APIs.
- `grep.rs` line truncation (`maxColumns`) is separate:
- simple character-boundary truncation of matched lines with `...`,
@@ -242,7 +245,7 @@ Text functions generally return deterministic transformed output; errors are lim
| Flow | Filesystem access | Shared cache | Notes |
| ---------------------------- | ----------------- | -------------------- | --------------------------------------------- |
| `search` / `hasMatch` | No | No | regex on provided bytes/string only |
| `text` module functions | No | No | ANSI/width/sanitization only |
| `text` module functions | No | No | ANSI/width utilities only |
| `highlight` module functions | No | No | syntax + ANSI coloring only |
| `countTokens` | No | No | tokenization only |
| `astGrep` / `astEdit` | Yes | No | syntax-aware file search/edit |
+12 -7
View File
@@ -65,10 +65,11 @@ Flow (`#handleRetryableError`):
6. Compute base delay: `retry.baseDelayMs * 2^(attempt-1)`.
7. For usage-limit errors, parse retry hints and call auth storage (`markUsageLimitReached(...)`); if credential switching succeeds, force delay to `0`, otherwise use a larger retry-after/backoff hint when present.
8. If no credential switch occurred, suppress the current model selector for cooldown, try configured retry model fallback chains, and force delay to `0` on model switch.
9. Emit `auto_retry_start`.
10. Remove the trailing assistant error message from agent runtime state (kept in persisted session history).
11. Sleep with abort support.
12. Schedule `agent.continue()` through the post-prompt task scheduler (`delayMs: 1`) for the same prompt generation.
9. If the final delay exceeds `retry.maxDelayMs` and no credential/model switch happened, emit final failure and do not sleep.
10. Emit `auto_retry_start`.
11. Remove the trailing assistant error message from agent runtime state (kept in persisted session history).
12. Sleep with abort support.
13. Schedule `agent.continue()` through the post-prompt task scheduler (`delayMs: 1`) for the same prompt generation.
### What resets retry counters
@@ -77,8 +78,9 @@ Flow (`#handleRetryableError`):
- first successful non-error, non-aborted assistant message after retries started (emits `auto_retry_end { success: true }`)
- retry cancellation during backoff sleep
- max retries exceeded path
- max delay exceeded path
`#retryPromise` resolves/clears when retry chain ends (success, cancellation, or max-exceeded), via `#resolveRetry()`.
`#retryPromise` resolves/clears when retry chain ends (success, cancellation, max-exceeded, or max-delay failure), via `#resolveRetry()`.
## Backoff and max-attempt semantics
@@ -87,6 +89,7 @@ Settings:
- `retry.enabled` (default `true`)
- `retry.maxRetries` (default `3`)
- `retry.baseDelayMs` (default `2000`)
- `retry.maxDelayMs` (default `300000`, 5 minutes; `<= 0` disables the fail-fast cap)
Attempt numbering:
@@ -100,7 +103,7 @@ Backoff sequence with default settings:
- attempt 2: 4000 ms
- attempt 3: 8000 ms
Delay override inputs can come from parsed retry headers (`retry-after-ms`, `retry-after`, `x-ratelimit-reset-ms`, `x-ratelimit-reset`) or usage-limit backoff. Credential/model fallback switches set delay to `0`; otherwise parsed hints can extend the exponential local delay.
Delay override inputs can come from parsed retry headers (`retry-after-ms`, `retry-after`, `x-ratelimit-reset-ms`, `x-ratelimit-reset`) or usage-limit backoff. Credential/model fallback switches set delay to `0`; otherwise parsed hints can extend the exponential local delay. If the computed delay is greater than `retry.maxDelayMs` and no switch succeeded, retry ends immediately with a final error instead of sleeping.
## Abort mechanics
@@ -149,6 +152,7 @@ Defined in settings schema under retry group:
- `retry.enabled`
- `retry.maxRetries`
- `retry.baseDelayMs`
- `retry.maxDelayMs`
- `retry.fallbackChains`
- `retry.fallbackRevertPolicy` (`"cooldown-expiry"` by default; `"never"` disables automatic restoration)
@@ -190,7 +194,7 @@ Propagation:
Final failure surfacing:
- On max-exceeded or cancellation, `auto_retry_end.success === false`
- On max-exceeded, max-delay failure, or cancellation, `auto_retry_end.success === false`
- TUI shows: `Retry failed after N attempts: <finalError>`
- Extensions/hooks receive `auto_retry_end` with same fields
- RPC consumers receive same event object on stdout stream
@@ -203,6 +207,7 @@ Retry stops and will not auto-continue when any of these occur:
- error is not retry-classified
- error is context overflow (delegated to compaction path)
- max retries exceeded
- provider-requested delay exceeds `retry.maxDelayMs` and no credential/model switch is available
- user cancels retry (`abort_retry` or `Esc` during retry loader)
- global abort (`abort`) cancels retry first
+94 -125
View File
@@ -1,77 +1,79 @@
# Notebook tool runtime internals
# Notebook file runtime internals
This document describes the current `notebook` tool implementation and its relationship to the kernel-backed Python runtime.
This document describes current `.ipynb` handling in `coding-agent` and its relationship to the kernel-backed Python runtime.
The critical distinction: **`notebook` is a JSON notebook editor, not a notebook executor**. It edits `.ipynb` cell sources directly; it does not start or talk to a Python kernel.
The critical distinction: **notebook support is file conversion/editing, not notebook execution**. `.ipynb` files are exposed as editable cell-marked text through `read` and the edit pipeline; no notebook-specific tool starts or talks to a Python kernel.
## Implementation files
- [`src/edit/notebook.ts`](../packages/coding-agent/src/edit/notebook.ts)
- [`src/edit/read-file.ts`](../packages/coding-agent/src/edit/read-file.ts)
- [`src/tools/read.ts`](../packages/coding-agent/src/tools/read.ts)
- [`src/tools/eval.ts`](../packages/coding-agent/src/tools/eval.ts)
- [`src/eval/py/executor.ts`](../packages/coding-agent/src/eval/py/executor.ts)
- [`src/eval/py/kernel.ts`](../packages/coding-agent/src/eval/py/kernel.ts)
- [`src/session/streaming-output.ts`](../packages/coding-agent/src/session/streaming-output.ts)
- [`src/tools/eval.ts`](../packages/coding-agent/src/tools/eval.ts)
## 1) Runtime boundary: editing vs executing
## `notebook` tool (`src/edit/notebook.ts`)
## `.ipynb` file conversion (`src/edit/notebook.ts`)
- Supports `action: edit | insert | delete` on a `.ipynb` file.
- Resolves path relative to session CWD (`resolveToCwd`).
- Loads notebook JSON, validates `cells` array, validates `cell_index` bounds.
- Applies source edits in-memory and writes full notebook JSON back with `JSON.stringify(notebook, null, 1)`.
- Returns textual summary + structured `details` (`action`, `cellIndex`, `cellType`, `totalCells`, `cellSource`).
- `read` treats `.ipynb` files as notebooks unless the selector is `:raw`.
- The default notebook view is editable text with markers:
- `# %% [code] cell:N`
- `# %% [markdown] cell:N`
- `# %% [raw] cell:N`
- Line selectors and multi-range selectors operate on that virtual text.
- Edit/write paths round-trip virtual text back to notebook JSON through `serializeEditedNotebookText(...)`.
- Existing notebook metadata is preserved when a marker references an existing `cell:N`; new cells get fresh empty metadata.
- Missing notebooks edited through this path start from an empty nbformat 4.5 notebook.
No kernel lifecycle exists in this tool:
No kernel lifecycle exists in this path:
- no gateway acquisition
- no kernel session ID
- no `execute_request`
- no stream chunks from kernel channels
- no rich display capture (`image/png`, JSON display, status MIME)
- no code execution
- no stream chunks from Python
- no rich display capture
- no output artifact pipeline from execution
## Notebook-like execution path (`src/tools/eval.ts` + `src/eval/py/*`)
## Kernel-backed execution path (`src/tools/eval.ts` + `src/eval/py/*`)
When the agent needs to run cell-style Python code (sequential cells, persistent state, rich displays), that goes through the **`eval` tool** with `language: "python"`, not `notebook`.
When the agent needs to run cell-style Python code (sequential cells, persistent state, rich displays), that goes through the **`eval` tool** with per-cell `language: "py"`, not through notebook file handling.
That path is where kernel modes, restart/cancel behavior, chunk streaming, and output artifact truncation live.
That path is where Python subprocess lifecycle, reset/cancel behavior, chunk streaming, rich displays, and output artifact truncation live.
## 2) Notebook cell handling semantics (`notebook` tool)
## 2) Notebook cell handling semantics
## Source normalization
`content` is split into `source: string[]` with newline preservation:
Notebook JSON `source` is converted to virtual text by joining source arrays. When virtual text is serialized back, cell source is split with newline preservation:
- each non-final line keeps trailing `\n`
- final line has no forced trailing newline
- each line ending in `\n` stays as a separate source entry with the newline
- a final non-newline-terminated line is stored without forcing a trailing newline
- empty content becomes an empty `source` array
This mirrors notebook JSON conventions and avoids accidental line concatenation on later edits.
## Action behavior
## Marker parsing and cell preservation
- `edit`
- replaces `cells[cell_index].source`
- preserves existing `cell_type`
- `insert`
- inserts at `[0..cellCount]`
- `cell_type` defaults to `code`
- code cells initialize `execution_count: null` and `outputs: []`
- markdown cells initialize only `metadata` + `source`
- `delete`
- removes `cells[cell_index]`
- returns removed `source` in details for renderer preview
- The first representation line must be a marker; text before the first marker, including a blank line, is rejected.
- Markers must match `# %% [code|markdown|raw]` with optional `cell:N`.
- If `cell:N` points at an unused existing cell, that cell is cloned, its `cell_type` and `source` are updated, and unrelated metadata is preserved.
- If no valid unused original index is present, a new cell is created.
- Code cells ensure `execution_count` exists and `outputs` exists.
- Markdown/raw cells remove `execution_count` and `outputs`.
## Error surfaces
Hard failures are thrown for:
- missing notebook file
- missing notebook on read
- invalid JSON
- missing/non-array `cells`
- out-of-range index (insert and non-insert have different valid ranges)
- missing `content` for `edit`/`insert`
- invalid cell objects or cell types
- invalid editable representation (for example, text before the first cell marker)
These become `Error:` tool responses upstream; renderer uses notebook path + formatted error text.
These surface through the caller (`read`, edit, or `write`) as normal tool errors.
## 3) Kernel session semantics (where they actually exist)
@@ -82,136 +84,103 @@ Kernel semantics are implemented in `executePython` / `PythonKernel` and apply t
`PythonKernelMode`:
- `session` (default)
- kernels cached in `kernelSessions` map
- max 4 sessions; oldest evicted on overflow
- idle/dead cleanup every 30s, timeout after 5 minutes
- per-session queue serializes execution (`session.queue`)
- kernels are cached by `(session id, cwd)`
- multiple owners can share a retained kernel for the same key
- execution is serialized by the tool's exclusive concurrency and backend execution path
- dead kernels are replaced before execution
- `per-call`
- creates kernel for request
- creates a subprocess for the request
- executes
- always shuts down kernel in `finally`
- always shuts down the subprocess in `finally`
## Reset behavior
`eval` passes `reset` only for the first cell in a multi-cell Python call; later cells always run with `reset: false`.
Each eval cell has its own optional `reset` flag. `reset: true` resets the selected Python session before that cell executes; it is not a top-level tool parameter.
## Kernel death / restart / retry
In session mode (`withKernelSession`):
In session mode:
- dead kernel detected by heartbeat (`kernel.isAlive()` check every 5s) or execute failure.
- pre-run dead state triggers `restartKernelSession`.
- execute-time crash path retries once: restart kernel, rerun handler.
- `restartCount > 1` in same session throws `Python kernel restarted too many times in this session`.
Startup retry behavior:
- shared gateway kernel creation retries once on `SharedGatewayCreateError` with HTTP 5xx.
Resource exhaustion recovery:
- detects `EMFILE`/`ENFILE`/"Too many open files" style failures
- clears tracked sessions
- calls `shutdownSharedGateway()`
- retries kernel session creation once
- if the retained subprocess is not alive before execution, it is replaced
- if execution fails because the subprocess died, the kernel is replaced and the code is retried once
- explicit `reset` is rejected while another reset for the same session key is already in progress
## 4) Environment/session variable injection
Kernel startup receives the optional session file path from executor:
Kernel startup and per-execution environment patching can receive:
- `PI_SESSION_FILE` (session state file path)
- `PI_SESSION_FILE`
- `PI_ARTIFACTS_DIR`
- `PI_TOOL_BRIDGE_URL`
- `PI_TOOL_BRIDGE_TOKEN`
- `PI_TOOL_BRIDGE_SESSION`
`PythonKernel.#initializeKernelEnvironment(...)` then runs init script inside kernel to:
- `os.chdir(cwd)`
- inject env entries into `os.environ`
- prepend cwd to `sys.path` if missing
Implication:
- prelude helpers that read session context rely on this env var in Python process state.
The runner initializes process state so code executes in the requested cwd, managed env entries are reflected in `os.environ`, and cwd is available on `sys.path`.
## 5) Streaming/chunk and display handling (kernel-backed path)
The kernel client processes Jupyter protocol messages per execution:
The Python backend uses an NDJSON subprocess runner. The host processes frames per execution:
- `stream` -> text chunk to `onChunk`
- `execute_result` / `display_data` ->
- display text chosen by MIME precedence: `text/markdown` > `text/plain` > converted `text/html`
- structured outputs captured separately:
- `application/json` -> `{ type: "json" }`
- `image/png` -> `{ type: "image" }`
- `application/x-omp-status` -> `{ type: "status" }` (no text emission)
- `error` -> traceback text pushed to chunk stream + structured error metadata
- `input_request` -> emits stdin warning text, sends empty `input_reply`, marks stdin requested
- completion waits for both `execute_reply` and kernel `status=idle`
- `stdout` / `stderr` -> text chunks to `onChunk`
- `display` / `result` -> MIME bundle rendering
- `error` -> traceback text and structured error metadata
- `done` -> final status, execution count, cancellation state
Display text MIME precedence:
1. `text/markdown`
2. `text/plain`
3. converted `text/html`
Structured outputs captured separately include:
- `application/json` -> JSON display output
- `image/png` / `image/jpeg` -> image output
- `application/x-omp-status` -> status event
Cancellation/timeout:
- abort signal triggers `interrupt()` (REST `/interrupt` + control-channel `interrupt_request`)
- result marks `cancelled=true`
- timeout path annotates output with `Command timed out after <n> seconds`
- abort/timeout sends `SIGINT` to the runner
- if the runner does not settle after the interrupt grace window, shutdown escalates and the kernel is recreated on the next call
- timeout output is annotated with a timeout message
## 6) Truncation and artifact behavior
`OutputSink` in `src/session/streaming-output.ts` is used by kernel execution paths (`executeWithKernel`):
`OutputSink` in `src/session/streaming-output.ts` is used by kernel execution paths:
- sanitizes every chunk (`sanitizeText`)
- sanitizes every chunk
- tracks total/output lines and bytes
- optional artifact spill file (`artifactPath`, `artifactId`)
- when in-memory buffer exceeds threshold (`DEFAULT_MAX_BYTES` unless overridden):
- marks truncated
- keeps tail bytes in memory (UTF-8 safe boundary)
- can spill full stream to artifact sink
`dump()` returns:
- visible output text (possibly tail-truncated)
- truncation flag + counts
- artifact ID (for `artifact://<id>` references)
- optionally spills full output to an artifact file
- keeps a UTF-8-safe in-memory tail buffer when output exceeds the configured threshold
`eval` converts this metadata into result truncation notices and TUI warnings.
`notebook` tool does **not** use `OutputSink`; it has no stream/artifact truncation pipeline because it does not execute code.
Notebook file conversion does **not** use `OutputSink`; it has no stream/artifact truncation pipeline because it does not execute code.
## 7) Renderer assumptions and formatting
## Notebook renderer (`notebookToolRenderer`)
## Read/edit notebook representation
- call view: status line with action + notebook path + cell/type metadata
- result view:
- success summary derived from `details`
- `cellSource` rendered via `renderCodeCell`
- markdown cells set language hint `markdown`; other cells have no explicit language override
- collapsed code preview limit is `PREVIEW_LIMITS.COLLAPSED_LINES * 2`
- supports expanded mode via shared render options
- uses render cache keyed by width + expanded state
Error rendering assumption:
- if first text content starts with `Error:`, renderer formats as notebook error block.
Notebook files are rendered to the model as text. The visible cell markers are part of the editable representation, not comments that are ignored during serialization.
## Python renderer (for actual execution output)
Kernel-backed execution rendering expects:
- per-cell status transitions (`pending/running/complete/error`)
- optional structured status event section
- per-cell status transitions (`pending` / `running` / `complete` / `error`)
- optional structured status events
- optional JSON output trees
- image outputs
- truncation warnings + optional `artifact://<id>` pointer
This renderer behavior is unrelated to `notebook` JSON editing results except that both reuse shared TUI primitives.
This renderer behavior is unrelated to notebook JSON editing except that both reuse shared TUI primitives.
## 8) Divergence from eval Python backend behavior
## 8) Practical workflow
If "plain Python execution" means the `eval` tool with `language: "python"`:
If a workflow needs both notebook mutation and execution:
- `eval` executes code in a kernel, persists state by mode, streams chunks, captures rich displays, handles interrupts/timeouts, and supports output truncation/artifacts.
- `notebook` performs deterministic notebook JSON mutations only; no execution, no kernel state, no chunk stream, no display outputs, no artifact pipeline.
If a workflow needs both:
1. edit notebook source with `notebook`
2. execute code cells via `eval` with `language: "python"` (manually passing code), not through `notebook`
1. read or edit the `.ipynb` file through the normal file tools
2. copy the desired cell source into `eval` cells with `language: "py"` to execute it
3. write resulting source changes back to the notebook if needed
Current implementation does not provide a single tool that both mutates `.ipynb` and executes notebook cells through kernel context.
+23 -9
View File
@@ -1,6 +1,6 @@
# Plugin manager and installer plumbing
This document describes how `omp plugin` operations mutate plugin state on disk and how installed plugins become runtime capabilities (tools and extensions today, hooks/commands path resolution available).
This document describes how `omp plugin` npm/link operations mutate plugin state on disk and how installed npm/link plugins become runtime capabilities (tools and extensions today, hooks/commands path resolution available). Marketplace installs use separate marketplace registries and cache plumbing; see `docs/marketplace.md`.
## Scope and architecture
@@ -9,14 +9,14 @@ There are two plugin-management implementations in the codebase:
1. **Active path used by CLI commands**: `PluginManager` (`src/extensibility/plugins/manager.ts`)
2. **Legacy helper module**: installer functions (`src/extensibility/plugins/installer.ts`)
`omp plugin ...` command execution goes through `PluginManager`.
`omp plugin` npm/link actions go through `PluginManager`; marketplace actions go through `MarketplaceManager`.
`installer.ts` still documents important safety checks and filesystem behavior, but it is not the path used by `src/commands/plugin.ts` + `src/cli/plugin-cli.ts`.
## Lifecycle: from CLI invocation to runtime availability
```text
omp plugin <action> ...
omp plugin <npm/link action> ...
-> src/commands/plugin.ts
-> runPluginCommand(...) in src/cli/plugin-cli.ts
-> PluginManager method (install/list/uninstall/link/...)
@@ -24,22 +24,28 @@ omp plugin <action> ...
-> runtime discovery: discoverAndLoadCustomTools(...) and discoverAndLoadExtensions(...)
-> getAllPluginToolPaths(cwd) / getAllPluginExtensionPaths(cwd)
-> custom tool loader imports tool modules; extension loader imports extension modules
omp plugin install name@marketplace / omp install name@marketplace
-> MarketplaceManager
-> mutate ~/.omp/marketplaces.json, ~/.omp/plugins/installed_plugins.json, cache dirs
-> installed marketplace plugin cache is surfaced as plugin roots/capabilities
```
### Command entrypoints
- `src/commands/plugin.ts` defines command/flags and forwards to `runPluginCommand`.
- `src/cli/plugin-cli.ts` maps subcommands to `PluginManager` methods:
- `src/cli/plugin-cli.ts` maps npm/link subcommands to `PluginManager` methods:
- `install`, `uninstall`, `list`, `link`, `doctor`, `features`, `config`, `enable`, `disable`
- No explicit `update` action exists; update is done by re-running `install` with a new package/version spec.
- `discover`, `upgrade`, and `marketplace ...` subcommands use `MarketplaceManager`.
- No explicit npm-plugin `update` action exists; update is done by re-running `install` with a new package/version spec.
## On-disk model
Global plugin state lives under `~/.omp/plugins`:
- `package.json` — dependency manifest used by `bun install`/`bun uninstall`
- `node_modules/` — installed plugin packages or symlinks
- `omp-plugins.lock.json` — runtime state:
- `package.json` — dependency manifest used by `bun install`/`bun uninstall` for npm-installed plugins
- `node_modules/` — installed npm plugin packages or symlinks
- `omp-plugins.lock.json` — runtime state for npm/link plugins:
- enabled/disabled per plugin
- selected feature set per plugin
- persisted plugin settings
@@ -50,6 +56,13 @@ Project-local overrides live at:
Overrides are read-only from manager/loader perspective (no write path here) and can disable plugins or override features/settings for this project.
Marketplace registries live separately:
- `~/.omp/marketplaces.json` — configured marketplace catalogs
- `~/.omp/plugins/installed_plugins.json` — user-scoped marketplace installs
- `<cwd>/.omp/plugins/installed_plugins.json` — project-scoped marketplace installs when available
- `~/.omp/plugins/cache/{marketplaces,plugins}/` — cached catalogs and plugin directories
## Plugin spec parsing and metadata interpretation
## Install spec grammar
@@ -171,10 +184,11 @@ For each enabled plugin:
Each resolver includes base entries plus feature entries:
- base entries are always included
- explicit feature list -> only selected features
- `enabledFeatures === null` -> enable features marked `default: true`
Missing files are silently skipped (`existsSync` guard).
Manifest entries may point to a file or to a directory containing `index.ts`, `index.js`, `index.mjs`, or `index.cjs`. Missing files are silently skipped (`existsSync` guard).
## Current runtime wiring differences
+15 -15
View File
@@ -18,12 +18,12 @@ Avoid ports that depend on JS-only state or dynamic imports. N-API exports shoul
`@oh-my-pi/pi-natives` no longer has a `packages/natives/src/<module>` TypeScript wrapper layer. The package root points at generated native artifacts:
- runtime entry: `packages/natives/native/index.js`
- runtime entry/export wrapper: `packages/natives/native/index.js`
- types entry: `packages/natives/native/index.d.ts`
- loader helpers: `packages/natives/native/loader-state.js`
- embedded manifest: `packages/natives/native/embedded-addon.js`
Consumers import directly from `@oh-my-pi/pi-natives`. The generated declarations are produced during `bun --cwd=packages/natives run build`.
Consumers import directly from `@oh-my-pi/pi-natives`. The generated declarations and explicit ESM exports are produced during `bun --cwd=packages/natives run build`.
## Anatomy of a native export
@@ -38,9 +38,9 @@ Consumers import directly from `@oh-my-pi/pi-natives`. The generated declaration
**Package/build side:**
- `packages/natives/scripts/build-native.ts` runs napi-rs, installs the `.node` artifact, copies generated `index.js`/`index.d.ts`, and appends enum runtime exports.
- `packages/natives/native/index.js` is the loader that chooses a candidate `.node` file and returns the loaded addon.
- `packages/natives/package.json` exposes only the package root (`@oh-my-pi/pi-natives`).
- `packages/natives/scripts/build-native.ts` runs napi-rs, installs the `.node` artifact, copies generated `index.d.ts`, and regenerates explicit ESM class/function exports plus enum runtime exports in the checked-in `native/index.js`.
- `packages/natives/native/index.js` is the ESM entrypoint that calls the loader, exposes named exports, and rejects install/compiled `.node` files that do not expose the package-version sentinel.
- `packages/natives/package.json` exposes only the package root (`@oh-my-pi/pi-natives`) as the import surface. At publish time the binaries are split out: the core ships the loader only (no `.node`), and each platform's `.node` is published as an optional-dependency leaf package `@oh-my-pi/pi-natives-<tag>` (`scripts/ci-release-publish.ts` + `packages/natives/scripts/gen-npm-packages.ts`). This is transparent to importers — you still `import` from `@oh-my-pi/pi-natives`.
**Consumer side:**
@@ -62,7 +62,7 @@ Consumers import directly from `@oh-my-pi/pi-natives`. The generated declaration
- Run `bun --cwd=packages/natives run build`.
- Confirm the generated `packages/natives/native/index.d.ts` includes the new export with the intended JS name/signature.
- Confirm `packages/natives/native/index.js` still has generated enum exports appended when enum changes are involved.
- Confirm `packages/natives/native/index.js` has generated explicit ESM exports for the new class/function and enum objects when enum changes are involved.
3. **Update consumers**
@@ -94,7 +94,7 @@ The loader probes platform-tagged artifacts in deterministic order. For x64, sel
Non-x64 uses `pi_natives.<tag>.node`.
Compiled binaries also probe `<getNativesDir()>/<version>/...` and a legacy user-data directory before package/executable locations. If any earlier candidate is stale, a new export may appear missing.
Compiled binaries also probe `<getNativesDir()>/<version>/...` and a legacy user-data directory before package/executable locations. Windows `node_modules` installs stage leaf/core addons into the same versioned directory before probing. If any earlier candidate is stale, a new export may appear missing unless the version sentinel rejects it first.
**Fix:** remove stale candidate/cache files and rebuild.
@@ -105,16 +105,16 @@ rm packages/natives/native/pi_natives.<platform>-<arch>-baseline.node
bun --cwd=packages/natives run build
```
For compiled binaries, delete the versioned addon cache shown in the loader error (normally under `~/.omp/natives/<version>` unless `$XDG_DATA_HOME/omp` is used).
For compiled binaries or Windows staging, delete the versioned addon cache shown in the loader error (normally under `~/.omp/natives/<version>` unless `$XDG_DATA_HOME/omp` is used).
### 2) Generated types do not match loaded binary
This can happen when `native/index.d.ts` was regenerated but the `.node` file being loaded is stale or from a different platform/variant.
This can happen when `native/index.d.ts` was regenerated but the `.node` file being loaded is stale, same-version incomplete, or from a different platform/variant. Different-version install/compiled binaries should be rejected by the version sentinel during loading.
Verify the loaded export set from the actual candidate path:
Verify the loaded export set from the actual candidate path reported by the loader:
```bash
bun -e 'const tag = `${process.platform}-${process.arch}`; const mod = require(`./packages/natives/native/pi_natives.${tag}.node`); console.log(Object.keys(mod).sort())'
bun -e 'import { createRequire } from "node:module"; const require = createRequire(import.meta.url); const mod = require(process.argv[2]); console.log(Object.keys(mod).sort())' -- /path/from/loader/error/pi_natives.<tag>[-variant].node
```
Fix the build/candidate mismatch. Do not paper over it with optional consumer checks if the export is required.
@@ -123,9 +123,9 @@ Fix the build/candidate mismatch. Do not paper over it with optional consumer ch
Keep N-API signatures simple and owned. Avoid borrowed references like `&str` in public exports. If you need structured data, use `#[napi(object)]` structs. If you need callbacks, use napi-rs `ThreadsafeFunction` and keep callback error/value behavior explicit.
### 4) Enum runtime exports
### 4) Enum runtime exports and ESM named exports
napi-rs declarations alone are not enough for JS callers that use enum objects at runtime. `scripts/gen-enums.ts` appends enum objects to `native/index.js`. If you add or change a native enum, verify both `native/index.d.ts` and the generated enum export block in `native/index.js`.
napi-rs declarations alone are not enough for JS callers that import named symbols or use enum objects at runtime. `scripts/gen-enums.ts` reads `native/index.d.ts`, writes explicit `export const ... = nativeBindings...` entries for public classes/functions, and emits enum objects in `native/index.js`. If you add or change a native export, verify both `native/index.d.ts` and the generated export block in `native/index.js`.
### 5) Benchmarking mistakes
@@ -161,8 +161,8 @@ bench("feature/native", () => {
## Verification checklist
- Generated `native/index.d.ts` includes the new export and intended TS signature.
- The loaded `.node` file's `Object.keys(require(candidate))` includes the new export.
- Runtime enum objects are present when the change adds/changes enums.
- `native/index.js` includes the generated named export; enum objects are present when the change adds/changes enums.
- The loaded `.node` file's `Object.keys(require(candidate))` includes the new export and the package-version sentinel.
- Bench numbers are recorded in the PR/notes.
- Call sites are updated only if native is faster/equal and behavior-compatible.
- Obsolete JS code is removed when the native implementation becomes canonical.
+14 -12
View File
@@ -5,7 +5,7 @@ This document explains how token/tool streaming is normalized in `@oh-my-pi/pi-a
## End-to-end flow
1. `streamSimple()` (`packages/ai/src/stream.ts`) maps generic options and dispatches to a provider stream function.
2. Provider stream functions translate provider-native stream events into the unified `AssistantMessageEvent` sequence. Current built-ins include Anthropic, OpenAI Responses/Completions/Codex/Azure Responses, Google Gemini/Gemini CLI/Vertex, Bedrock Converse, Ollama, Cursor, plus GitLab Duo/Kimi wrappers and extension-registered custom APIs.
2. Provider stream functions translate provider-native stream events into the unified `AssistantMessageEvent` sequence. Current built-ins include Anthropic, OpenAI Responses/Completions/Codex/Azure Responses, Google Gemini/Gemini CLI/Vertex, Bedrock Converse, Ollama, Cursor, pi-native gateway transport, plus GitLab Duo/Kimi/Synthetic wrappers and extension-registered custom APIs.
3. Each provider pushes events into `AssistantMessageEventStream` (`packages/ai/src/utils/event-stream.ts`), which throttles delta events and exposes:
- async iteration for incremental updates
- `result()` for final `AssistantMessage`
@@ -72,16 +72,17 @@ Sources: `packages/ai/src/providers/openai-responses.ts`, `openai-codex-response
Normalization points:
- `response.output_item.added` starts reasoning/text/function-call blocks
- reasoning summary events (`response.reasoning_summary_text.delta`) become `thinking_delta`
- `response.output_item.added` starts reasoning/text/function-call/custom-tool blocks
- reasoning summary events (`response.reasoning_summary_text.delta`) and raw reasoning events (`response.reasoning_text.delta`) become `thinking_delta`
- output/refusal deltas become `text_delta`
- `response.function_call_arguments.delta` becomes `toolcall_delta`
- `response.function_call_arguments.delta` and `response.custom_tool_call_input.delta` become `toolcall_delta`
- `response.output_item.done` emits `thinking_end` / `text_end` / `toolcall_end`
- `response.completed` maps status to stop reason and usage
- `response.completed` maps status to stop reason and usage; `response.failed` / SDK `error` events throw into the wrapper's terminal `error` path
Tool-call argument streaming:
- same `partialJson` accumulation pattern as Anthropic
- same `partialJson` accumulation pattern as Anthropic for function-call JSON arguments
- custom tools stream raw string input and expose final arguments as `{ input: <raw> }`
- providers that send only `response.function_call_arguments.done` still populate final args
- tool call IDs are normalized as `"<call_id>|<item_id>"`
@@ -139,13 +140,14 @@ If provider stream throws or signals failure, each provider wrapper catches and
## Malformed chunk / SSE parse failure behavior
For these provider paths, chunk/SSE framing is handled by vendor SDK streams (Anthropic SDK, OpenAI SDK, Google SDK). This code does not implement a custom SSE decoder here.
Most provider paths delegate chunk/SSE framing to vendor SDK streams (Anthropic SDK, OpenAI SDK, Google SDK). The Codex SSE fallback uses `readSseJson()` directly, and websocket Codex frames are normalized through the same event handler.
Observed behavior in current implementation:
- malformed chunk/SSE parsing at SDK level surfaces as an exception or stream `error` event
- provider wrapper converts that into unified terminal `error` event
- no provider-specific resume/retry inside the stream function itself
- malformed SDK stream parsing surfaces as an exception or stream `error` event
- malformed Codex SSE JSON/framing throws from the local SSE reader
- provider wrapper converts failures into unified terminal `error` events
- no provider-specific resume/retry inside the stream function itself, except Codex websocket-to-SSE transport fallback before replay-unsafe output is emitted
- higher-level retries are handled in `AgentSession` auto-retry logic (message-level retry, not stream-chunk replay)
## Cancellation boundaries
@@ -211,9 +213,9 @@ Provider-specific (not fully abstracted):
- [`../../ai/src/utils/event-stream.ts`](../packages/ai/src/utils/event-stream.ts) — generic stream queue + assistant delta throttling.
- [`../../ai/src/utils/json-parse.ts`](../packages/ai/src/utils/json-parse.ts) — partial JSON parsing for streamed tool arguments.
- [`../../ai/src/providers/anthropic.ts`](../packages/ai/src/providers/anthropic.ts) — Anthropic event translation and tool JSON delta accumulation.
- [`../../ai/src/providers/openai-responses.ts`](../packages/ai/src/providers/openai-responses.ts), [`openai-codex-responses.ts`](../packages/ai/src/providers/openai-codex-responses.ts), [`azure-openai-responses.ts`](../packages/ai/src/providers/azure-openai-responses.ts) — Responses-family event translation and status mapping.
- [`../../ai/src/providers/openai-responses.ts`](../packages/ai/src/providers/openai-responses.ts), [`openai-responses-shared.ts`](../packages/ai/src/providers/openai-responses-shared.ts), [`openai-codex-responses.ts`](../packages/ai/src/providers/openai-codex-responses.ts), [`azure-openai-responses.ts`](../packages/ai/src/providers/azure-openai-responses.ts) — Responses-family event translation and status mapping.
- [`../../ai/src/providers/google.ts`](../packages/ai/src/providers/google.ts), [`google-gemini-cli.ts`](../packages/ai/src/providers/google-gemini-cli.ts), [`google-vertex.ts`](../packages/ai/src/providers/google-vertex.ts) — Gemini stream chunk-to-block translation variants.
- [`../../ai/src/providers/google-shared.ts`](../packages/ai/src/providers/google-shared.ts) — Gemini finish-reason mapping and shared conversion rules.
- [`../../ai/src/providers/amazon-bedrock.ts`](../packages/ai/src/providers/amazon-bedrock.ts), [`openai-completions.ts`](../packages/ai/src/providers/openai-completions.ts), [`ollama.ts`](../packages/ai/src/providers/ollama.ts), [`cursor.ts`](../packages/ai/src/providers/cursor.ts) — additional built-in stream adapters using the same event contract.
- [`../../ai/src/providers/amazon-bedrock.ts`](../packages/ai/src/providers/amazon-bedrock.ts), [`openai-completions.ts`](../packages/ai/src/providers/openai-completions.ts), [`ollama.ts`](../packages/ai/src/providers/ollama.ts), [`cursor.ts`](../packages/ai/src/providers/cursor.ts), [`pi-native-client.ts`](../packages/ai/src/providers/pi-native-client.ts) — additional built-in stream adapters using the same event contract.
- [`../../agent/src/agent-loop.ts`](../packages/agent/src/agent-loop.ts) — provider stream consumption and `message_update` bridging.
- [`../src/session/agent-session.ts`](../packages/coding-agent/src/session/agent-session.ts) — session-level handling of streaming updates, abort, retry, and persistence.
+51 -47
View File
@@ -10,21 +10,26 @@ It covers tool behavior, runner lifecycle, environment handling, execution seman
- Subprocess kernel client: `src/eval/py/kernel.ts`
- Python wrapper / NDJSON server: `src/eval/py/runner.py`
- Prelude helpers loaded into every kernel: `src/eval/py/prelude.py`
- Host-side subagent helper bridge: `src/eval/agent-bridge.ts`
- MIME bundle renderer (text + structured outputs): `src/eval/py/display.ts`
- Interactive-mode renderer for user-triggered Python runs: `src/modes/components/eval-execution.ts`
- Runtime/env filtering and Python resolution: `src/eval/py/runtime.ts`
## What eval's Python backend is
The `eval` tool executes one or more Python cells inside a long-lived `python3` subprocess that speaks NDJSON over stdin/stdout. No Jupyter, no kernel gateway, no extra pip dependencies — a vanilla Python 3.8+ interpreter is enough. Rich `display()` output (PIL, pandas, plotly, matplotlib figures) keeps working because the wrapper reimplements the MIME-bundle dispatch that IPython previously provided.
The `eval` tool executes one or more Python cells inside a retained `python` subprocess that speaks NDJSON over stdin/stdout. No Jupyter gateway and no extra pip dependencies are required — a vanilla Python 3.8+ interpreter is enough. Rich `display()` output (PIL, pandas, plotly, matplotlib figures) keeps working because the wrapper implements MIME-bundle dispatch.
Tool params:
```ts
{
cells: Array<{ code: string; title?: string }>;
timeout?: number; // seconds, clamped to 1..600, default 30
reset?: boolean; // reset selected runtime before the first cell only
cells: Array<{
language: "py" | "js";
code: string;
title?: string;
timeout?: number; // seconds, clamped to 1..600, default 30. Inactivity budget — see "Cell timeout".
reset?: boolean; // reset this cell's selected runtime before execution
}>;
}
```
@@ -32,7 +37,7 @@ The tool is `concurrency = "exclusive"` for a session, so calls do not overlap.
## Kernel lifecycle
Each kernel is a single Python subprocess: `python -u <runner.py>`. The runner is bundled with the host binary (Bun text import), written to `~/.omp/python-env`-adjacent tmp cache once per script-hash, and reused by every subsequent spawn.
Each Python kernel is a single subprocess: `<resolved-python> -u <runner.py>`. The runner is bundled with the host binary (Bun text import), written to an `omp-python-runner` cache under the OS temp directory once per script hash, and reused by subsequent spawns.
Kernel startup sequence:
@@ -76,25 +81,25 @@ Status events the prelude emits (e.g. `_emit_status("find", count=…)`) ship in
The runner's source transformer rewrites IPython-style magics to plain Python calls before parsing. Supported set:
| Magic | Effect |
| --- | --- |
| `%pip <args>` | `python -m pip <args>` with live streaming output. Newly installed packages are evicted from `sys.modules` so the next `import` picks up the fresh install. |
| `%cd <path>` | `os.chdir(path)` (with `~` expansion); emits status event. |
| `%pwd` | Returns `os.getcwd()`. |
| `%ls [path]` | Returns `sorted(os.listdir(path))`. |
| `%env [KEY[=VAL]]` | List, read, or set env vars (matches prelude `env()` semantics). |
| `%set_env KEY VALUE` | Set `os.environ[KEY]`. |
| `%time <expr>` / `%timeit <expr>` | Time the expression; emits status event with elapsed ms. |
| `%who` / `%whos` | List user-namespace names. |
| `%reset` | Clear user globals and re-inject prelude. |
| `%load <path>` | Read a file into a fresh cell and execute. |
| `%run <path>` | `runpy.run_path` and merge globals back. |
| `%%bash` / `%%sh` | Run the cell body via `bash`/`sh`. |
| `%%capture [name]` | Run body with stdout/stderr captured into `name`. |
| `%%timeit` | Time the cell body. |
| `%%writefile <path>` | Write body to file. |
| `!cmd` / `var = !cmd` | Run command via subprocess shell; returns an SList-style result with `.n` / `.s` helpers. |
| `var = %name args` | Assignment forms work for line magics and `!cmd`. |
| Magic | Effect |
| --------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------- |
| `%pip <args>` | `python -m pip <args>` with live streaming output. Newly installed packages are evicted from `sys.modules` so the next `import` picks up the fresh install. |
| `%cd <path>` | `os.chdir(path)` (with `~` expansion); emits status event. |
| `%pwd` | Returns `os.getcwd()`. |
| `%ls [path]` | Returns `sorted(os.listdir(path))`. |
| `%env [KEY[=VAL]]` | List, read, or set env vars (matches prelude `env()` semantics). |
| `%set_env KEY VALUE` | Set `os.environ[KEY]`. |
| `%time <expr>` / `%timeit <expr>` | Time the expression; emits status event with elapsed ms. |
| `%who` / `%whos` | List user-namespace names. |
| `%reset` | Clear user globals and re-inject prelude. |
| `%load <path>` | Read a file into a fresh cell and execute. |
| `%run <path>` | `runpy.run_path` and merge globals back. |
| `%%bash` / `%%sh` | Run the cell body via `bash`/`sh`. |
| `%%capture [name]` | Run body with stdout/stderr captured into `name`. |
| `%%timeit` | Time the cell body. |
| `%%writefile <path>` | Write body to file. |
| `!cmd` / `var = !cmd` | Run command via subprocess shell; returns an SList-style result with `.n` / `.s` helpers. |
| `var = %name args` | Assignment forms work for line magics and `!cmd`. |
Unknown magic names raise `NameError: UsageError: ...` inside the cell.
@@ -103,12 +108,11 @@ Unknown magic names raise `NameError: UsageError: ...` inside the cell.
`python.kernelMode` controls retained kernel reuse:
- `session` (default)
- Reuses kernel sessions keyed by session file plus cwd when a session file exists; otherwise by cwd.
- Execution is serialized per session via a queue.
- Idle sessions are evicted after 5 minutes.
- At most 4 sessions; oldest is evicted on overflow.
- Heartbeat checks detect dead kernels.
- Auto-restart allowed once; repeated crash ⇒ hard failure.
- Reuses kernel sessions keyed by namespaced eval session id plus cwd.
- Multiple owners can share the same retained kernel for that key.
- Calls through the tool are exclusive, so tool invocations do not overlap.
- A dead retained subprocess is replaced before execution.
- If the subprocess dies during execution, it is replaced and the cell is retried once.
- `per-call`
- Spawns a fresh subprocess for each request.
- Shuts the subprocess down after the request.
@@ -116,7 +120,7 @@ Unknown magic names raise `NameError: UsageError: ...` inside the cell.
### Multi-cell behavior in a single tool call
Cells run sequentially in the same kernel instance for that tool call.
Python cells run sequentially in the same selected Python kernel instance for that tool call.
If an intermediate cell fails:
@@ -124,7 +128,7 @@ If an intermediate cell fails:
- Tool returns a targeted error indicating which cell failed.
- Later cells are not executed.
`reset=true` only applies to the first cell execution in that call.
`reset=true` is per cell and resets that language runtime before the cell executes.
## Environment filtering and runtime resolution
@@ -146,25 +150,25 @@ The runner additionally receives `PYTHONUNBUFFERED=1` and `PYTHONIOENCODING=utf-
## Tool availability and mode selection
`eval.py` / `eval.js` (both default `true`) plus optional `PI_PY` override controls eval backend exposure:
`eval.py` / `eval.js` (both default `true`) plus optional boolean env flags `PI_PY` / `PI_JS` control eval backend exposure:
- Python backend only (`eval.py=true`, `eval.js=false`)
- JavaScript backend only (`eval.py=false`, `eval.js=true`)
- both backends
- Python backend only (`eval.py=true`, `eval.js=false`, or `PI_PY=1 PI_JS=0`)
- JavaScript backend only (`eval.py=false`, `eval.js=true`, or `PI_PY=0 PI_JS=1`)
- both backends (`eval.py=true`, `eval.js=true`, or `PI_PY=1 PI_JS=1`)
`PI_PY` accepted values:
`PI_PY` and `PI_JS` use normal boolean flag parsing. If either env var is set, the env pair overrides the per-key settings; an unset member of the pair defaults to enabled.
- `0` / `bash` → JavaScript backend only
- `1` / `py` → Python backend only
- `mix` / `both` → both backends
If Python preflight fails and `eval.js` is enabled, `eval` remains available for `js` cells; `py` cells fail with a Python-backend availability error.
If Python preflight fails and `eval.js` is enabled, `eval` remains available and dispatches to JavaScript unless `language: "python"` is explicitly requested.
Python prelude helpers include `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)`. It synchronously calls the host bridge, runs one subagent through the task executor, and returns the final text. When `schema` is supplied, the helper parses the subagent's JSON output and returns the object.
## Execution flow and cancellation/timeout
### Tool-level timeout
### Cell timeout
`eval` timeout is in seconds, default 30, clamped to `1..600`. The tool combines caller abort signal and timeout signal with `AbortSignal.any(...)`.
Each eval cell `timeout` is in seconds, defaults to 30, and is clamped to `1..600`. It is a **wall-clock budget on the cell's own work** that the watchdog (`IdleTimeout`, `src/eval/idle-timeout.ts`) enforces, **but it is paused while a host-side `agent()`/`parallel()`/`llm()` bridge call is in flight**: those calls pump a heartbeat (`withBridgeHeartbeat`, `src/eval/heartbeat.ts`) that re-arms the watchdog, so a long fanout or a slow completion runs to completion instead of being killed mid-stream.
The heartbeat is the **sole** signal that extends the budget. Everything else the cell does — compute, `stdout`/`stderr`, `log()`/`phase()`, and ordinary (non-agent) tool calls — counts against `timeout`, so a cell that is not delegating to an agent/llm is bounded by a plain wall-clock timeout. The tool combines the caller abort signal, the session abort signal, and the watchdog's signal with `AbortSignal.any(...)`; no wall-clock deadline is passed to the backend, so neither runtime arms a competing fixed timer.
### Kernel execution cancellation
@@ -172,7 +176,7 @@ On abort/timeout:
- The host sends `kill("SIGINT")` to the runner subprocess.
- The runner's exec-time signal handler raises `KeyboardInterrupt` inside the user code.
- Result includes `cancelled=true`; timeout path annotates output as `Command timed out after <n> seconds`.
- Result includes `cancelled=true`; the timeout path annotates output as `Command timed out after <n> seconds`.
- Between requests the runner installs `SIG_IGN` for SIGINT so a stray cancel does not tear down the kernel.
If a second cancel is required (runner stuck in C code), the host escalates to `SIGTERM` and the session restarts on the next call.
@@ -217,7 +221,7 @@ Output is streamed through `OutputSink` and may be persisted to artifact storage
- Tool renderer (`eval.ts`):
- shows code-cell blocks with per-cell status
- collapsed preview defaults to 10 lines
- supports expanded mode for full output and richer status detail
- supports expanded mode for all output retained in the tool result
- Interactive renderer (`eval-execution.ts`):
- used for user-triggered Python execution in TUI
- collapsed preview defaults to 20 lines
@@ -226,7 +230,7 @@ Output is streamed through `OutputSink` and may be persisted to artifact storage
## Operational troubleshooting
- **Python backend not available** — Check `eval.py`, `PI_PY`, and that `python`/`python3` is on PATH. If preflight fails and `eval.js` is enabled, omit `language` or pass `language: "js"` to use JavaScript.
- **Python backend not available** — Check `eval.py`, `PI_PY`, and that `python`/`python3` is on PATH. If preflight fails and `eval.js` is enabled, use a `js` cell.
- **No Python on PATH** — Install a system Python 3.8+ or place a venv at `~/.omp/python-env`. `omp setup python --check` reports the resolved interpreter.
- **Execution hangs then times out** — Increase tool `timeout` (max 600s) if workload is legitimate. For stuck native code, cancellation triggers `SIGINT` first then escalates; the session restarts on the next request.
- **stdin/input prompts in Python code** — `input()` is not supported; pass data programmatically.
@@ -234,7 +238,7 @@ Output is streamed through `OutputSink` and may be persisted to artifact storage
## Relevant environment variables
- `PI_PY` — tool exposure override
- `PI_PY` / `PI_JS` — eval backend exposure overrides
- `PI_PYTHON_SKIP_CHECK=1` — bypass Python preflight/warm checks
- `PI_PYTHON_INTEGRATION=1` — enable gated integration tests that spawn a real Python
- `PI_PYTHON_IPC_TRACE=1` — log NDJSON frames exchanged with the runner subprocess
+8 -5
View File
@@ -14,8 +14,9 @@ This document explains how preview/apply workflows are modeled in coding-agent a
`resolve` is a hidden tool that finalizes a pending preview action.
- `action: "apply"` executes the queued action's `apply(reason)` callback and returns that result with resolve metadata.
- `action: "discard"` invokes `reject(reason)` if provided; otherwise returns `Discarded: <label>. Reason: <reason>`.
- `action: "apply"` executes the queued action's `apply(reason, extra)` callback and returns that result with resolve metadata.
- `action: "discard"` invokes `reject(reason, extra)` if provided; otherwise returns `Discarded: <label>. Reason: <reason>`.
- `extra` is optional free-form metadata. Queue handlers receive it; producers decide whether it has meaning.
If no pending action exists, `resolve` fails with:
@@ -32,7 +33,9 @@ Runtime behavior:
- if the model rejects the forced tool choice, the queue directive is requeued,
- `resolve` does not maintain a separate pending-action stack.
Multiple pending previews therefore follow the active tool-choice queue ordering, not an independent pending-action store.
`resolve` also checks a standing resolve handler after the queue invoker; this is used by long-lived approval flows that are not ordinary preview tool calls.
Multiple pending previews therefore follow the active tool-choice queue ordering, not an independent pending-action store. If an apply callback throws, the queued helper re-pushes the same resolve directive and reminder so the preview can still be discarded or retried.
## Built-in producer example (`ast_edit`)
@@ -40,9 +43,9 @@ Multiple pending previews therefore follow the active tool-choice queue ordering
- label (human-readable summary)
- `sourceToolName` (`ast_edit`)
- `apply(reason: string)` callback that reruns AST edit with `dryRun: false`
- `apply(reason: string, extra?: Record<string, unknown>)` callback that reruns AST edit with `dryRun: false`
`resolve(action="apply", reason="...")` passes `reason` into this callback.
`resolve(action="apply", reason="...")` passes `reason` into this callback. `ast_edit` currently ignores `extra`.
## Custom tools: `pushPendingAction`
+16 -4
View File
@@ -126,11 +126,17 @@ Important edge behavior from runtime:
- `{ id?, type: "get_branch_messages" }`
- `{ id?, type: "get_last_assistant_text" }`
- `{ id?, type: "set_session_name", name: string }`
- `{ id?, type: "handoff", customInstructions?: string }`
### Messages
- `{ id?, type: "get_messages" }`
### Login
- `{ id?, type: "get_login_providers" }`
- `{ id?, type: "login", providerId: string }`
## Response Schema
All command results use `RpcResponse`:
@@ -170,14 +176,19 @@ Data payloads are command-specific and defined in `rpc-types.ts`.
]
}
],
"systemPrompt": "...",
"systemPrompt": ["..."],
"dumpTools": [
{
"name": "read",
"description": "Read files and URLs",
"parameters": {}
}
]
],
"contextUsage": {
"tokens": 1100,
"contextWindow": 200000,
"percent": 0.55
}
}
```
@@ -364,6 +375,7 @@ Extensions in RPC mode use request/response UI frames.
- `select`, `confirm`, `input`, `editor`, `cancel`
- `notify`, `setStatus`, `setWidget`, `setTitle`, `set_editor_text`
- `open_url` (emitted by RPC login flows)
Runtime note:
@@ -622,8 +634,8 @@ Current helper characteristics:
- Spawns `bun <cliPath> --mode rpc`
- Correlates responses by generated `req_<n>` ids
- Dispatches only recognized `AgentEvent` types to listeners
- Dispatches recognized core `AgentEvent` types to listeners
- Supports host-owned custom tools via `setCustomTools()` and automatic handling of `host_tool_call` / `host_tool_cancel`
- Does **not** expose helper methods for every protocol command (for example, `set_interrupt_mode` and `set_session_name` are in protocol types but not wrapped as dedicated methods)
- Wraps common protocol commands including OAuth `getLoginProviders()` / `login(...)`; use raw protocol frames for any surface not wrapped by the helper.
Use raw protocol frames if you need complete surface coverage.
+50 -36
View File
@@ -9,18 +9,21 @@ It reflects the current implementation, including partial semantics and metadata
## Implementation files
- [`../src/capability/rule.ts`](../packages/coding-agent/src/capability/rule.ts)
- [`../src/capability/index.ts`](../packages/coding-agent/src/capability/index.ts)
- [`../src/discovery/index.ts`](../packages/coding-agent/src/discovery/index.ts)
- [`../src/discovery/helpers.ts`](../packages/coding-agent/src/discovery/helpers.ts)
- [`../src/discovery/builtin.ts`](../packages/coding-agent/src/discovery/builtin.ts)
- [`../src/discovery/cursor.ts`](../packages/coding-agent/src/discovery/cursor.ts)
- [`../src/discovery/windsurf.ts`](../packages/coding-agent/src/discovery/windsurf.ts)
- [`../src/discovery/cline.ts`](../packages/coding-agent/src/discovery/cline.ts)
- [`../src/sdk.ts`](../packages/coding-agent/src/sdk.ts)
- [`../src/system-prompt.ts`](../packages/coding-agent/src/system-prompt.ts)
- [`../src/internal-urls/rule-protocol.ts`](../packages/coding-agent/src/internal-urls/rule-protocol.ts)
- [`../src/utils/frontmatter.ts`](../packages/utils/src/frontmatter.ts)
- [`packages/coding-agent/src/capability/rule.ts`](../packages/coding-agent/src/capability/rule.ts)
- [`packages/coding-agent/src/capability/rule-buckets.ts`](../packages/coding-agent/src/capability/rule-buckets.ts)
- [`packages/coding-agent/src/capability/index.ts`](../packages/coding-agent/src/capability/index.ts)
- [`packages/coding-agent/src/discovery/index.ts`](../packages/coding-agent/src/discovery/index.ts)
- [`packages/coding-agent/src/discovery/helpers.ts`](../packages/coding-agent/src/discovery/helpers.ts)
- [`packages/coding-agent/src/discovery/builtin.ts`](../packages/coding-agent/src/discovery/builtin.ts)
- [`packages/coding-agent/src/discovery/builtin-defaults.ts`](../packages/coding-agent/src/discovery/builtin-defaults.ts)
- [`packages/coding-agent/src/discovery/agents.ts`](../packages/coding-agent/src/discovery/agents.ts)
- [`packages/coding-agent/src/discovery/cursor.ts`](../packages/coding-agent/src/discovery/cursor.ts)
- [`packages/coding-agent/src/discovery/windsurf.ts`](../packages/coding-agent/src/discovery/windsurf.ts)
- [`packages/coding-agent/src/discovery/cline.ts`](../packages/coding-agent/src/discovery/cline.ts)
- [`packages/coding-agent/src/sdk.ts`](../packages/coding-agent/src/sdk.ts)
- [`packages/coding-agent/src/system-prompt.ts`](../packages/coding-agent/src/system-prompt.ts)
- [`packages/coding-agent/src/internal-urls/rule-protocol.ts`](../packages/coding-agent/src/internal-urls/rule-protocol.ts)
- [`packages/utils/src/frontmatter.ts`](../packages/utils/src/frontmatter.ts)
## 1. Canonical rule shape
@@ -50,16 +53,20 @@ Consequence: precedence and deduplication are **name-based only**. Two different
`src/discovery/index.ts` auto-registers providers. For `rules`, current providers are:
- `native` (priority `100`)
- `agents` (priority `70`)
- `cursor` (priority `50`)
- `windsurf` (priority `50`)
- `cline` (priority `40`)
- `builtin-defaults` (priority `1`)
### Native provider (`builtin.ts`)
Loads `.omp` rules from:
- project: `<cwd>/.omp/rules/*.{md,mdc}`
- project: `<cwd>/.omp/rules/*.{md,mdc}` when the cwd `.omp` directory exists
- user: `~/.omp/agent/rules/*.{md,mdc}`
- sticky user rule: `~/.omp/agent/RULES.md`
- sticky project rule: nearest ancestor `.omp/RULES.md` while walking from cwd toward the repository root
Normalization:
@@ -67,9 +74,19 @@ Normalization:
- frontmatter parsed via `parseFrontmatter`
- `content` = body (frontmatter stripped)
- `globs`, `alwaysApply`, `description`, `condition`/legacy `ttsr_trigger`, `scope`, and `interruptMode` are parsed by `buildRuleFromMarkdown`
- top-level `RULES.md` is synthesized as rule name `RULES` and forced to `alwaysApply: true`
Important caveat: `condition` values that look like file globs are converted into `tool:edit(...)` / `tool:write(...)` scope shorthands with catch-all condition `.*`.
### Agents provider (`agents.ts`)
Loads from both `.agent` and `.agents` directories:
- project: walk upward from `cwd` to repo root, loading `<ancestor>/.agent/rules/*.{md,mdc}` and `<ancestor>/.agents/rules/*.{md,mdc}`
- user: `~/.agent/rules/*.{md,mdc}` and `~/.agents/rules/*.{md,mdc}`
Normalization uses the shared `buildRuleFromMarkdown` path: filename-derived name, stripped frontmatter body, and parsed `globs`, `alwaysApply`, `description`, `condition`/legacy `ttsr_trigger`, `scope`, and `interruptMode`.
### Cursor provider (`cursor.ts`)
Loads from:
@@ -141,9 +158,11 @@ Ambiguity consequences:
Effective rule provider order is currently:
1. `native` (100)
2. `cursor` (50)
3. `windsurf` (50)
4. `cline` (40)
2. `agents` (70)
3. `cursor` (50)
4. `windsurf` (50)
5. `cline` (40)
6. `builtin-defaults` (1)
### Intra-provider ordering caveat
@@ -151,35 +170,29 @@ Within a provider, item order comes from `loadFilesFromDir` glob result ordering
Notable source-order differences:
- `native` appends project then user config dirs.
- `native` appends project `.omp/rules`, user `~/.omp/agent/rules`, user `RULES.md`, then nearest project `RULES.md`.
- `agents` appends project-walk `.agent`/`.agents` rule dirs before user home dirs.
- `cursor` appends user then project results.
- `windsurf` appends user `global_rules` first, then project rules.
- `cline` loads only nearest `.clinerules` source.
- `builtin-defaults` uses the embedded rule source order.
## 5. Split into Rulebook, Always-Apply, and TTSR buckets
After rule discovery in `createAgentSession` (`sdk.ts`):
After rule discovery in `createAgentSession` (`sdk.ts`), `bucketRules(...)` applies session-level filtering and bucket assignment:
1. All discovered rules are scanned.
2. Rules with `condition` entries are registered into `TtsrManager`; legacy `ttsr_trigger` / `ttsrTrigger` are accepted during rule parsing as condition fallbacks.
3. A separate `rulebookRules` list is built with this predicate:
```ts
!isTtsrRule && rule.alwaysApply !== true && !!rule.description;
```
4. An `alwaysApplyRules` list is built:
```ts
!isTtsrRule && rule.alwaysApply === true;
```
1. Drop rules listed in `ttsr.disabledRules`.
2. Drop rules from the `builtin-defaults` provider when `ttsr.builtinRules === false`.
3. Register rules with non-empty `condition` into `TtsrManager`; if registration succeeds, the rule is TTSR-only.
4. Put remaining `alwaysApply === true` rules into `alwaysApplyRules`.
5. Put remaining rules with `description` into `rulebookRules`.
### Bucket behavior
- **TTSR bucket**: any rule with a non-empty parsed `condition` that `TtsrManager.addRule(...)` accepts. Takes priority over other buckets.
- **TTSR bucket**: any enabled rule with a non-empty parsed `condition` that `TtsrManager.addRule(...)` accepts. Takes priority over other buckets.
- **Always-apply bucket**: `alwaysApply === true`, not TTSR. Full content injected into system prompt. Resolvable via `rule://`.
- **Rulebook bucket**: must have description, must not be TTSR, must not be `alwaysApply`. Listed in system prompt by name+description; content read on demand via `rule://`.
- A rule with both `condition` and `alwaysApply` goes to TTSR only (TTSR takes priority).
- A rule with both `condition` and `alwaysApply` goes to TTSR only if TTSR registration accepts it; otherwise it can fall through to always-apply.
- A rule with both `alwaysApply` and `description` goes to always-apply only (not rulebook).
## 6. How metadata affects runtime surfaces
@@ -195,7 +208,8 @@ After rule discovery in `createAgentSession` (`sdk.ts`):
- Carried through on `Rule`.
- Rendered as `<glob>...</glob>` entries in the system prompt rules block.
- Exposed in rules UI state (`extensions` mode list).
- **Not enforced for automatic matching in this pipeline.** There is no runtime glob matcher selecting rules by current file/tool target.
- Used by TTSR as a global path gate: if a TTSR rule has globs, the match context must include at least one matching file path.
- Not used to automatically select rulebook rules for `rule://`; rulebook matching remains advisory prompt behavior.
### `alwaysApply`
@@ -244,7 +258,7 @@ Implications:
## 9. Known partial / non-enforced semantics
1. Provider descriptions mention legacy files (`.cursorrules`, `.windsurfrules`), but current loader code paths do not actually read those files.
2. `globs` metadata is surfaced to prompt/UI but not enforced by rule selection logic.
1. The rule providers currently loaded for `rules` are `native`, `agents`, `cursor`, `windsurf`, `cline`, and embedded `builtin-defaults`; provider files for other tools may parse other config formats but do not register rule loaders.
2. `globs` metadata is surfaced to prompt/UI and is used as a global path gate for TTSR matching, but it is not used to automatically select rulebook rules for `rule://`.
3. Rule selection for `rule://` includes rulebook and always-apply rules, but not TTSR-only rules.
4. Discovery warnings (`loadCapability("rules").warnings`) are produced but `createAgentSession` does not currently surface/log them in this path.
+10 -8
View File
@@ -65,7 +65,7 @@ If omitted, it resolves:
- `sessionManager`: `SessionManager.create(cwd)` (file-backed)
- skills/context files/prompt templates/slash commands/extensions/custom TS commands
- built-in tools via `createTools(...)`
- MCP tools (enabled by default)
- MCP tools (enabled by default; Exa MCP servers are folded into native Exa integration, and browser automation MCP servers are filtered when the built-in browser tool is enabled)
- LSP integration (enabled by default)
- `eventBus`: new `EventBus()` unless supplied
@@ -171,10 +171,12 @@ If restore fails, `modelFallbackMessage` explains fallback.
`AuthStorage.getApiKey(...)` resolves in this order:
1. runtime override (`setRuntimeApiKey`)
2. stored credentials in `agent.db`
3. provider environment variables
4. custom-provider resolver fallback (if configured)
1. runtime override (`setRuntimeApiKey`, used by CLI `--api-key`)
2. config-sourced API key override (`models.yml` provider `apiKey`)
3. stored API-key credential in `agent.db` / broker-backed storage
4. stored OAuth credential, including refresh when needed
5. provider environment variables
6. custom-provider resolver fallback
## Event subscription model
@@ -239,7 +241,7 @@ Related APIs:
```ts
const { session } = await createAgentSession({
toolNames: ["read", "grep", "find", "write"],
toolNames: ["read", "search", "find", "write"],
requireYieldTool: true,
});
```
@@ -318,7 +320,7 @@ Use `setToolUIContext(...)` only if your embedder provides UI capabilities that
- `options.hasUI === true` (interactive TUI), **and**
- the `lsp.diagnosticsOnWrite` setting is enabled.
Print / script / RPC / ACP invocations (`hasUI=false`) skip the warmup entirely: they don't render the warmup status indicator and typically finish before the language servers would stabilize, so warming them just spends CPU parsing big `initialize` responses concurrently with the LLM stream consumer and jitters perceived latency. Tools that actually need an LSP server still spin one up on demand through `getOrCreateClient()` — only the *startup* warmup is skipped. The returned `lspServers` field in `CreateAgentSessionResult` is therefore `undefined` (not an empty array) whenever the warmup branch was bypassed.
Print / script / RPC / ACP invocations (`hasUI=false`) skip the warmup entirely: they don't render the warmup status indicator and typically finish before the language servers would stabilize, so warming them just spends CPU parsing big `initialize` responses concurrently with the LLM stream consumer and jitters perceived latency. Tools that actually need an LSP server still spin one up on demand through `getOrCreateClient()` — only the _startup_ warmup is skipped. The returned `lspServers` field in `CreateAgentSessionResult` is therefore `undefined` (not an empty array) whenever the warmup branch was bypassed.
## Minimal controlled embed example
@@ -345,7 +347,7 @@ const { session } = await createAgentSession({
modelRegistry,
settings,
sessionManager: SessionManager.inMemory(),
toolNames: ["read", "grep", "find", "edit", "write"],
toolNames: ["read", "search", "find", "edit", "write"],
enableMCP: false,
enableLsp: true,
});
+6 -6
View File
@@ -1,6 +1,6 @@
# Secret Obfuscation
Prevents sensitive values (API keys, tokens, passwords) from being sent to LLM providers. When enabled, secrets are replaced with deterministic placeholders before leaving the process, and restored in tool call arguments returned by the model.
Prevents sensitive values (API keys, tokens, passwords) from being sent to LLM providers. When enabled, secrets are replaced before outbound text content leaves the process. Reversible obfuscation placeholders are restored when session context is rebuilt for display or resume.
## Enabling
@@ -19,14 +19,14 @@ secrets:
2. Outbound text messages to the LLM have secret values replaced with deterministic placeholders like `#AB12#`.
3. Session context/tool arguments returned from the model are deep-walked and obfuscation placeholders are restored to original values before display or execution.
3. Session context is deep-walked and obfuscation placeholders are restored when building display/resume context. Replace-mode substitutions are one-way and are not restored.
Two modes control what happens to each secret:
| Mode | Behavior | Reversible |
| --------------------- | ------------------------------------------------------- | ----------------------------------------------- |
| `obfuscate` (default) | Replaced with deterministic placeholder `#[A-Z0-9]{4}#` | Yes (deobfuscated in tool args/session context) |
| `replace` | Replaced with deterministic same-length string | No (one-way) |
| Mode | Behavior | Reversible |
| --------------------- | ------------------------------------------------------- | -------------------------------------------- |
| `obfuscate` (default) | Replaced with deterministic placeholder `#[A-Z0-9]{4}#` | Yes (deobfuscated in display/resume context) |
| `replace` | Replaced with deterministic same-length string | No (one-way) |
## secrets.yml
@@ -13,18 +13,18 @@ This document describes operator-visible behavior for session export/share/fork/
## Operation matrix
| Operation | Entry path | Session mutation | Session file creation/switch | Output artifact |
| --------------------------------------- | ------------------------- | ------------------------------------- | ---------------------------------------------------------------------------------- | -------------------------------------------------------------------------------- | ---- |
| `/dump` | Interactive slash command | No | No | Clipboard text |
| `/export [path]` | Interactive slash command | No | No | HTML file |
| `--export <session.jsonl> [outputPath]` | CLI startup fast-path | No runtime session mutation | No active session; reads target file | HTML file |
| `/share` | Interactive slash command | No | No | Temp HTML + share URL/gist |
| `/fork` | Interactive slash command | Yes (active session identity changes) | Creates new session file and switches current session to it (persistent mode only) | Copies artifact directory to new session namespace when present |
| `--fork <id | path>` | CLI startup | Yes after session creation | Creates a new session fork from the selected source into current cwd/session dir | None |
| `/resume` | Interactive slash command | Yes (active in-memory state replaced) | Switches to selected existing session file | None |
| `--resume` | CLI startup (picker) | Yes after session creation | Opens selected existing session file | None |
| `--resume <id | path>` | CLI startup | Yes after session creation | Opens existing session; cross-project case can fork into current project | None |
| `--continue` | CLI startup | Yes after session creation | Opens terminal breadcrumb or most-recent session; creates new one if none exists | None |
| Operation | Entry path | Session mutation | Session file creation/switch | Output artifact |
| --------------------------------------- | ------------------------- | ------------------------------------- | ---------------------------------------------------------------------------------- | --------------------------------------------------------------- |
| `/dump` | Interactive slash command | No | No | Clipboard text |
| `/export [path]` | Interactive slash command | No | No | HTML file |
| `--export <session.jsonl> [outputPath]` | CLI startup fast-path | No runtime session mutation | No active session; reads target file | HTML file |
| `/share` | Interactive slash command | No | No | Temp HTML + share URL/gist |
| `/fork` | Interactive slash command | Yes (active session identity changes) | Creates new session file and switches current session to it (persistent mode only) | Copies artifact directory to new session namespace when present |
| `--fork <id\|path>` | CLI startup | Yes after session creation | Creates a new session fork from the selected source into current cwd/session dir | None |
| `/resume` | Interactive slash command | Yes (active in-memory state replaced) | Switches to selected existing session file | None |
| `--resume` | CLI startup picker | Yes after session creation | Opens selected existing session file | None |
| `--resume <id\|path>` | CLI startup | Yes after session creation | Opens existing session; global cross-project match can fork into current project | None |
| `--continue` | CLI startup | Yes after session creation | Opens terminal breadcrumb or most-recent session; creates new one if none exists | None |
## Export and dump
@@ -177,7 +177,7 @@ Startup `--fork` is resolved before normal session creation:
1. `--fork` is rejected with `--no-session`.
2. Path-like values (`/`, `\`, or `.jsonl`) call `SessionManager.forkFrom(path, cwd, sessionDir)`.
3. Other values resolve like resumable session ids via current scope and then global search when allowed.
3. Other values resolve via `resolveResumableSession(...)`: local sessions first, then global search when `sessionDir` is not forced. Matching accepts lowercased session id prefixes, full JSONL filename prefixes, and timestamp-stripped filename id suffixes.
4. The forked file is created in the current cwd/session-dir scope and becomes the active session manager for startup.
## Resume and continue
@@ -207,9 +207,10 @@ Notes:
`createSessionManager()` resolution order:
1. If value looks like path (`/`, `\`, or `.jsonl`), open directly.
2. Else treat as id prefix:
- search current scope (`SessionManager.list(cwd, sessionDir)`)
- if not found and no explicit `sessionDir`, search global (`SessionManager.listAll()`)
2. Else `resolveResumableSession(...)` searches:
- current scope (`SessionManager.list(cwd, sessionDir)`)
- global sessions (`SessionManager.listAll()`) only when no explicit `sessionDir` was provided
3. Matching accepts case-insensitive session id prefixes, full JSONL filename prefixes, and the id suffix after the timestamp in `<timestamp>_<sessionId>.jsonl`.
Cross-project id match behavior:
@@ -235,15 +236,20 @@ This is startup-only behavior; there is no interactive `/continue` slash command
1. Emit `session_before_switch` with `reason: "resume"` and `targetSessionFile` (cancellable).
2. Disconnect agent event subscription and abort in-flight work.
3. Clear queued steering/follow-up/next-turn messages.
4. Flush current session manager writes.
5. `sessionManager.setSessionFile(sessionPath)` and update `agent.sessionId`.
6. Build session context from loaded entries.
7. Emit `session_switch` with `reason: "resume"`.
8. Replace agent messages from context.
9. Restore model (if available in current registry).
10. Restore or initialize thinking level.
11. Reconnect agent event subscription.
3. Flush current session manager writes.
4. Capture rollback state for the current session, agent messages, queued steering/follow-up/next-turn messages, model/thinking/service-tier, MCP selections, tools, and system prompt.
5. Clear queued steering/follow-up/next-turn messages.
6. `sessionManager.setSessionFile(sessionPath)` and update `agent.sessionId`.
7. Build session context from loaded entries.
8. Restore MCP selections/tools/system prompt for the target session.
9. Emit `session_switch` with `reason: "resume"`.
10. Replace agent messages from context and sync todos.
11. Close provider sessions when switching files, or when same-file reload changed replay messages.
12. Restore model (if available in current registry).
13. Restore or initialize thinking level and service tier.
14. Reconnect agent event subscription.
If any step after the capture fails, `switchSession()` restores the captured state and reconnects the previous agent subscription before rethrowing.
No new session file is created by `switchSession()` itself.
+16 -15
View File
@@ -31,15 +31,15 @@ It focuses on current implementation behavior, including fallback paths and cave
There are two different listing pipelines:
1. `getRecentSessions(sessionDir, limit)` (welcome/summary view)
- Reads only a 4KB prefix (`readTextPrefix(..., 4096)`) from each file.
- Reads only a 4KB prefix (`readTextPrefix(..., 4096)` or equivalent direct read for file storage) from each file.
- Parses header + earliest user text preview.
- Returns lightweight `RecentSessionInfo` with lazy `name` and `timeAgo` getters.
- Sorts by file `mtime` descending.
2. `SessionManager.list(...)` / `SessionManager.listAll()` (resume pickers and ID matching)
- Reads full session files.
- Reads the same 4KB prefix per file, not the full JSONL file.
- Builds `SessionInfo` objects (`id`, `cwd`, `title`, `messageCount`, `firstMessage`, `allMessagesText`, timestamps).
- Drops sessions with zero `message` entries.
- Uses prefix parsing plus marker counting; later messages beyond the prefix may not be present in `allMessagesText`.
- Sorts by `modified` descending.
### Metadata fallback behavior
@@ -52,8 +52,8 @@ For recent summaries (`RecentSessionInfo`):
For `SessionInfo` list entries:
- `title` is `header.title` or latest compaction `shortSummary`
- `firstMessage` is first user message text or `"(no messages)"`
- `title` is `header.title` or the last compaction `shortSummary` seen in the 4KB prefix
- `firstMessage` is first user message text discoverable from the prefix or `"(no messages)"`
## `--continue` resolution and terminal breadcrumb preference
@@ -80,10 +80,10 @@ Breadcrumb writes are best-effort and non-fatal.
1. Path-like value (contains `/`, `\\`, or ends with `.jsonl`)
- direct `SessionManager.open(sessionArg, parsed.sessionDir)`
2. ID prefix value
- find match in `SessionManager.list(cwd, sessionDir)` by `id.startsWith(sessionArg)`
- if no local match and `sessionDir` is not forced, try `SessionManager.listAll()`
- first match is used (no ambiguity prompt)
2. Resume key value
- `resolveResumableSession(...)` searches local sessions first, then all sessions when `sessionDir` is not forced
- matching is case-insensitive and accepts `id` prefix, full JSONL filename prefix, or the session-id suffix after the timestamp
- first match in modified-descending order is used (no ambiguity prompt)
Cross-project match behavior:
@@ -134,17 +134,18 @@ Flow:
- arrow/page navigation
- Enter to select
- Delete to delete after confirmation
- Esc to cancel
- Ctrl+C to exit
- fuzzy search across session id/title/cwd/first message/all messages/path
Empty-list render behavior:
- renders a message instead of crashing
- Enter on empty does nothing (no callback)
- renders `No sessions in current folder. Press Tab to view all.`
- Enter/Delete on empty do nothing (no callback)
- Esc/Ctrl+C still work
Caveat: UI text says `Press Tab to view all`, but this component currently has no Tab handler and current wiring only lists current-scope sessions.
Caveat: the empty-state UI mentions Tab, but this component currently has no Tab handler and current wiring only lists current-scope sessions.
## Runtime switch execution (`AgentSession.switchSession`)
@@ -237,6 +238,6 @@ Switch/open can still throw on true I/O failures (permission errors, rewrite fai
### ID prefix matching caveats
- ID matching uses `startsWith` and takes first match in sorted list.
- No ambiguity UI if multiple sessions share prefix.
- `SessionManager.list(...)` excludes sessions with zero messages, so those sessions are not resumable via ID match/list picker.
- Matching uses `startsWith` on the lowercased session id, lowercased JSONL filename, and lowercased id suffix after the filename timestamp.
- First match in modified-descending order wins; there is no ambiguity UI if multiple sessions share a prefix.
- Prefix-listing metadata is intentionally lightweight, so search text may not include messages outside the first 4KB of the session file.
+2 -1
View File
@@ -134,6 +134,7 @@ Tree selector behavior (`tree-selector.ts`):
- Flattens tree for navigation, keeps active-path highlighting, and prioritizes displaying the active branch first.
- Supports filter modes: `default`, `no-tools`, `user-only`, `labeled-only`, `all`.
- `default` suppresses `label`, `custom`, `model_change`, and `thinking_level_change`; it is not a complete "hide all internal entries" filter.
- Supports free-text search over rendered semantic content.
- `Shift+L` opens inline label editing and writes via `appendLabelChange`.
@@ -197,7 +198,7 @@ Naming source:
- trims whitespace
- capitalizes the first character
- returns `""` for whitespace-only / separator-only input
- The humanized name is applied with `sessionManager.setSessionName(name, "auto")`. Because `setSessionName` is a no-op when `titleSource === "user"`, the seeded name never overrides a name the user already chose (e.g. on the `preserveContext` path where the session continues with prior naming).
- The humanized name is applied only when the current session has no name (`!sessionManager.getSessionName()`). It then calls `sessionManager.setSessionName(name, "auto")`, which also refuses to overwrite user-named sessions.
- On successful apply, the terminal title (`setSessionTerminalTitle`) and the editor border color are refreshed to reflect the new name.
Examples (from `humanizePlanTitle`):
+4 -2
View File
@@ -96,6 +96,7 @@ All non-header entries include:
- `message`
- `thinking_level_change`
- `model_change`
- `service_tier_change`
- `compaction`
- `branch_summary`
@@ -460,12 +461,13 @@ Implementations:
Defined in `session-manager.ts`:
- `getRecentSessions(sessionDir, limit)` -> lightweight metadata for UI/session picker
- `getRecentSessions(sessionDir, limit)` -> lightweight metadata for UI/session picker, capped by `limit`
- `findMostRecentSession(sessionDir)` -> newest by mtime
- `list(cwd, sessionDir?)` -> sessions in one project scope
- `listAll()` -> sessions across all project scopes under `~/.omp/agent/sessions`
- `resolveResumableSession(sessionArg, cwd, sessionDir?)` -> local then global resume/fork target lookup
Metadata extraction reads only a prefix (`readTextPrefix(..., 4096)`) where possible.
Metadata extraction for `list`/`listAll` and `getRecentSessions` reads only a prefix (`readTextPrefix(..., 4096)` or an equivalent direct 4KB read for file storage). Resume matching is case-insensitive and accepts session id prefixes, full filename prefixes, or the id suffix after the timestamp in `<timestamp>_<sessionId>.jsonl`.
## Related but Distinct: Prompt History Storage
+11 -6
View File
@@ -55,6 +55,7 @@ Supported frontmatter fields on the skill type:
- `description?: string`
- `globs?: string[]`
- `alwaysApply?: boolean`
- `hide?: boolean`
- additional keys are preserved as unknown metadata
Current runtime behavior:
@@ -81,13 +82,15 @@ Provider ordering is priority-first (higher wins), then registration order for t
Current registered skill providers:
1. `native` (priority 100) — `.omp` user/project skills via `src/discovery/builtin.ts`
2. `claude` (priority 80)
3. priority 70 group (in registration order):
2. `omp-plugins` (priority 90) — `skills/` bundled next to extension packages loaded through `extensions:` or `--extension`/`-e`
3. `claude` (priority 80)
4. priority 70 group (in registration order):
- `claude-plugins`
- `agents`
- `codex`
4. `opencode` (priority 55)
Dedup key is skill name. First item with a given name wins.
5. `opencode` (priority 55)
Dedup key is skill name. First item with a given name wins.
### Source toggles and filtering
@@ -122,10 +125,12 @@ Filter order is:
System prompt construction (`src/system-prompt.ts`) uses discovered skills as follows:
- if `read` tool is available:
- include discovered skills list in prompt
- include discovered skills list in prompt, excluding skills with `hide: true`
- otherwise:
- omit discovered list
`hide: true` does not disable the skill. Hidden skills are still loaded and remain reachable through `skill://<name>` and `/skill:<name>` when skill commands are enabled.
Task tool subagents receive the session's discovered/provided skills list via normal session creation; there is no per-task skill pinning override.
### Interactive `/skill:<name>` commands
@@ -142,7 +147,7 @@ If `skills.enableSkillCommands` is true, interactive mode registers one slash co
- **Ctrl+Enter** (`app.message.followUp`) → invokes the skill on the `followUp` queue while streaming, or as a normal idle prompt when the agent is not streaming
- appends metadata (`Skill: <path>`, optional `User: <args>`)
There is no flag, mode-selector, or frontmatter knob to override this — the keybinding *is* the choice, identical to how free text is routed during streaming (`input-controller.ts:243-249` for Enter, `input-controller.ts:462-500` for Ctrl+Enter; both dispatch through `#invokeSkillCommand`).
There is no flag, mode-selector, or frontmatter knob to override this — the keybinding _is_ the choice, identical to how free text is routed during streaming (`input-controller.ts:243-249` for Enter, `input-controller.ts:462-500` for Ctrl+Enter; both dispatch through `#invokeSkillCommand`).
## `skill://` URL behavior
+16 -17
View File
@@ -73,24 +73,28 @@ export default function myExtension(pi: ExtensionAPI) {
}
```
## Discovery path
## Discovery paths
omp discovers extension modules in this order:
omp loads extension modules from these sources:
1. **Project-scoped auto-discovery** — `<cwd>/.omp/extensions/`
2. **User-scoped auto-discovery** — `~/.omp/agent/extensions/`
3. **Marketplace-installed plugins** — `~/.omp/plugins/node_modules/` (extensions shipped inside installed plugin packages)
4. **CLI flag** — `omp --extension ./my-ext.ts` (also `-e`; `--hook` is treated as an alias)
5. **Settings `extensions` array** — paths listed in `~/.omp/agent/config.yml` or `<cwd>/.omp/settings.json`
1. Native `.omp` locations discovered through the capability system:
- `<cwd>/.omp/extensions/`
- `~/.omp/agent/extensions/`
- legacy extension paths listed in `.omp/settings.json#extensions` or `~/.omp/agent/settings.json#extensions`
2. Marketplace-installed plugins from the OMP and Claude plugin registries.
3. Explicit configured paths passed by the CLI (`omp --extension ./my-ext.ts`, also `-e`; `--hook` is treated as an alias) and by the `extensions:` setting in config.
Within each scope, de-duplication is by resolved absolute path — first seen wins.
The runtime de-duplicates by resolved absolute path — first seen wins.
When a path points to a directory, omp resolves the entry point in this order:
1. `package.json` with `omp.extensions` (or legacy `pi.extensions`) field
2. `index.ts`
3. `index.js`
4. One-level scan for `*.ts` / `*.js` files and subdir `index.*` / `package.json` manifests
When scanning an `extensions/` directory, omp also loads direct `*.ts`/`*.js` files and one-level subdirectories that have `index.ts`, `index.js`, or a manifest.
Extension packages can also bundle sibling capability directories. When a package is loaded through `extensions:` or `--extension`/`-e`, the `omp-plugins` provider discovers its `skills/`, `hooks/pre|post/`, `tools/`, `commands/`, `rules/`, `prompts/`, and `.mcp.json`.
## package.json manifest
@@ -215,20 +219,15 @@ Extensions are a strict superset of hooks. New authoring should use `ExtensionAP
## Debugging
Start omp with `--log-level debug` to see extension load messages:
Start omp with `--log-level debug` to see extension load diagnostics:
```
omp --log-level debug
```
Watch for lines like:
Failed extension loads are logged with their path and error. Loaded extensions may also emit their own debug logs via `pi.logger`.
```
[extension-loader] loading /home/you/.omp/agent/extensions/my-ext.ts
[extension-loader] loaded: my-ext (1 tool, 1 command, 2 handlers)
```
To temporarily disable a specific extension by name without removing the file:
To temporarily disable a specific extension module by name without removing the file:
```yaml
# ~/.omp/agent/config.yml
+2 -2
View File
@@ -123,8 +123,8 @@ Contract:
- Handlers run in registration order. For `HookAPI`, each handler receives the original tool result event, and the last returned override wins.
- `content` replaces the full content array for the LLM.
- `details` replaces the structured details object.
- `isError` overrides the error flag (typed, but note: `HookToolWrapper` behavior for error path rethrows regardless).
- On a tool failure, `tool_result` is still emitted with `isError: true`; the original error is rethrown after handlers complete.
- `isError` exists on the shared result type, but `HookToolWrapper` does not propagate it into a successful tool result; on a tool failure, the original error is rethrown after handlers complete.
- On a tool failure, `tool_result` is still emitted with `isError: true`.
## Context modification contract
+11 -3
View File
@@ -52,9 +52,11 @@ The catalog file lives at either `.omp-plugin/marketplace.json` or `.claude-plug
| `owner` | yes | Object with at minimum `owner.name` (string) |
| `owner.name` | yes | Marketplace owner name |
| `owner.email` | no | Owner contact email |
| `description` | no | Short description of the marketplace |
| `plugins` | yes | Array of plugin entries (see below) |
| `metadata.description` | no | Short description of the marketplace |
| `metadata.version` | no | Catalog metadata version string |
| `metadata.pluginRoot` | no | String prepended to all relative plugin source paths |
| extra top-level fields | no | Preserved by the parser but not used by marketplace install/runtime logic |
### Plugin entry fields
@@ -67,7 +69,11 @@ The catalog file lives at either `.omp-plugin/marketplace.json` or `.claude-plug
| `author` | no | `{ name, email? }` |
| `homepage` | no | URL |
| `category` | no | e.g. `development`, `productivity`, `security` |
| `tags` | no | Array of string tags |
| `tags` / `keywords` | no | Arrays of string tags/keywords |
| `repository` | no | Repository URL |
| `license` | no | License string |
| `strict` | no | Boolean plugin metadata flag |
| `commands`, `agents`, `hooks`, `mcpServers`, `lspServers` | no | Capability metadata used by plugin tooling and selectors |
### Full catalog example
@@ -79,7 +85,9 @@ The catalog file lives at either `.omp-plugin/marketplace.json` or `.claude-plug
"name": "Acme Corp",
"email": "plugins@acme.example"
},
"description": "Official Acme plugins for oh-my-pi",
"metadata": {
"description": "Official Acme plugins for oh-my-pi"
},
"plugins": [
{
"name": "acme-linter",
@@ -28,7 +28,7 @@ omp --extension ./hello-extension
## Usage
After loading, type `/hello` in the omp prompt to trigger the notification.
After loading, type `/hello` or `/hello Ada` in the omp prompt. The command sends a visible greeting custom message into the conversation and shows a "Message sent!" notification.
## What it demonstrates
@@ -21,6 +21,7 @@ omp plugin install my-plugin@example-marketplace
- Minimum required `marketplace.json` fields: `name`, `owner.name`, `plugins`
- Relative path plugin source using `./` prefix (`"source": "./my-plugin"`)
- Plugin bundled inside the same directory tree as the marketplace catalog
- Extra catalog metadata: the example includes a top-level `description`; current marketplace parsing preserves extra top-level fields, while runtime behavior uses required fields and plugin entries.
## Structure
+3 -3
View File
@@ -1,12 +1,12 @@
# safety-hook
An `oh-my-pi` extension that demonstrates `tool_call` blocking. It intercepts every `bash` tool call and returns `{ block: true, reason: "..." }` if the command matches `rm -rf /`, preventing the LLM from executing the command.
An `oh-my-pi` extension that demonstrates `tool_call` blocking. It intercepts `bash` tool calls and returns `{ block: true, reason: "..." }` when the command contains `rm -rf /` with normal whitespace, preventing the tool from executing.
## What it demonstrates
- `pi.on("tool_call", ...)` — pre-execution interception
- `return { block: true, reason: "..." }` — blocking contract
- Exact-pattern guard on bash input
- Regex guard on bash input (`/\brm\s+-rf\s+\//`)
## Install
@@ -30,7 +30,7 @@ LLM calls bash tool
▼
tool_call handlers run
│
├─ command matches /rm\s+-rf\s+\// ?
├─ command matches /\brm\s+-rf\s+\// ?
│ yes → { block: true, reason: "..." } ← execution stops, reason sent to LLM
│ no → undefined ← execution continues normally
▼
+169
View File
@@ -0,0 +1,169 @@
# System Prompt Customization
How the coding-agent assembles the system prompt sent to the model, and what users can control via `SYSTEM.md`, `APPEND_SYSTEM.md`, and the matching CLI flags.
Primary implementation:
- `packages/coding-agent/src/system-prompt.ts` (`buildSystemPrompt`, `loadSystemPromptFiles`)
- `packages/coding-agent/src/main.ts` (`discoverSystemPromptFile`, `discoverAppendSystemPromptFile`)
- `packages/coding-agent/src/prompts/system/system-prompt.md` (default stable instruction template)
- `packages/coding-agent/src/prompts/system/custom-system-prompt.md` (internal custom-prompt template; not the normal CLI `SYSTEM.md` path)
- `packages/coding-agent/src/prompts/system/project-prompt.md` (project/environment footer)
---
## 1) Inputs
Four user-controllable inputs feed prompt assembly. All four resolve a value as either a literal string or, if the argument looks like a file path, the contents of that file (`resolvePromptInput`).
| Input | Source | Effect |
|---|---|---|
| `--system-prompt <text-or-file>` | CLI flag | Replaces block 0: the default stable instructions. Highest precedence. |
| `SYSTEM.md` | `<cwd>/.omp/SYSTEM.md`, then `~/.omp/agent/SYSTEM.md` (and equivalent paths under `.claude`, `.codex`, `.gemini`) | Same effect as `--system-prompt`; used when the flag is absent. |
| `--append-system-prompt <text-or-file>` | CLI flag | Adds a prompt block. Without a custom system prompt it goes after all default blocks; with one it goes after the custom block and before the preserved project/environment footer. |
| `APPEND_SYSTEM.md` | Same discovery as `SYSTEM.md` | Same effect as `--append-system-prompt`; used when the flag is absent. |
Discovery for `SYSTEM.md` / `APPEND_SYSTEM.md` uses `findConfigFile` (`packages/coding-agent/src/config.ts`): the first existing file across the ordered bases (`.omp`, `.claude`, `.codex`, `.gemini` — project-level at `<cwd>` first, then user-level at `~`) wins. **No ancestor walk-up.** Running `omp` from `<repo>/subdir` does not pick up `<repo>/.omp/SYSTEM.md`; the file must live directly under the cwd's config base or in the user-level location. See [`docs/config-usage.md`](./config-usage.md) for the full discovery contract.
Precedence (highest first):
1. `--system-prompt`
2. project `SYSTEM.md`
3. user `SYSTEM.md`
For append, the same precedence applies between `--append-system-prompt`, project `APPEND_SYSTEM.md`, and user `APPEND_SYSTEM.md`.
---
## 2) Replace vs. append
Normal CLI startup builds the default provider-facing prompt blocks first, then applies CLI / discovered file overrides in `packages/coding-agent/src/main.ts`:
```ts
if (resolvedSystemPrompt && resolvedAppendPrompt) {
options.systemPrompt = defaultPrompt => [resolvedSystemPrompt, resolvedAppendPrompt, ...defaultPrompt.slice(1)];
} else if (resolvedSystemPrompt) {
options.systemPrompt = defaultPrompt => [resolvedSystemPrompt, ...defaultPrompt.slice(1)];
} else if (resolvedAppendPrompt) {
options.systemPrompt = defaultPrompt => [...defaultPrompt, resolvedAppendPrompt];
}
```
The default blocks come from `buildSystemPrompt`:
- block 0: `system-prompt.md` — the stable default instructions (staff-engineer preamble, tool inventory, exploration rules, workflow rules, etc.);
- block 1, when non-empty: `project-prompt.md` — dynamic project/environment context (workstation info, context files, dir-context list, workspace tree, current date/cwd, and other project footer content).
Consequences for normal CLI use:
- Providing `--system-prompt` or `SYSTEM.md` replaces only block 0. The stable default instructions are removed, but the dynamic project/environment footer from `project-prompt.md` remains as `defaultPrompt.slice(1)`.
- Providing `--append-system-prompt` or `APPEND_SYSTEM.md` without a custom system prompt appends a new block after all default blocks.
- Providing both a custom system prompt and an append prompt produces: custom system prompt block, append prompt block, then the preserved dynamic project/environment footer.
If you want to keep both default blocks and add to them, use `--append-system-prompt` / `APPEND_SYSTEM.md` without `--system-prompt` / `SYSTEM.md`. If you want to replace the stable default instructions while keeping the dynamic footer, use `--system-prompt` / `SYSTEM.md`.
---
## 3) Templating contract
**Contents of `SYSTEM.md`, `APPEND_SYSTEM.md`, `--system-prompt`, and `--append-system-prompt` are treated as plain text.** They are resolved before prompt-block replacement and are not rendered as Handlebars templates.
The built-in prompt templates are Handlebars (`packages/utils/src/prompt.ts`), but user-provided strings are not compiled with that renderer. The secondary capability path can insert `systemPromptCustomization` into a Handlebars parent template, but a `{{value}}` reference in Handlebars still does not recursively render its substituted contents — the value is emitted as a string. Concretely:
```handlebars
{{! parent template — handled by Handlebars }}
{{#if systemPromptCustomization}}
{{systemPromptCustomization}}
{{/if}}
```
If `SYSTEM.md` contains:
```handlebars
Working in {{cwd}} on {{date}}.
{{#if hasMemoryRoot}}Memory enabled.{{/if}}
```
the rendered output contains those characters verbatim — `{{cwd}}`, `{{#if hasMemoryRoot}}`, etc. are NOT substituted. They will be shown to the model as literal Handlebars syntax.
This is by design. The internal template variables (`cwd`, `date`, `environment`, `workspaceTree`, `skills`, `rules`, `toolRefs`, `hasMemoryRoot`, `hasObsidian`, `mcpDiscoveryServerSummaries`, ...) are not a supported public surface — they change between releases as the prompt is rewritten, and they would couple user configs to internals. Treat them as private.
If a future release exposes a templating surface for `SYSTEM.md`, it will be opt-in (e.g. via a settings flag or a different filename) and documented here.
---
## 4) Recommended patterns
### "Tweak the default" — keep default, add a few rules
Use `APPEND_SYSTEM.md` (or `--append-system-prompt`) without `SYSTEM.md`. The default stable instructions and the dynamic project/environment footer stay intact; your text is appended as an additional block.
```text
# ~/.omp/agent/APPEND_SYSTEM.md
Prefer Bun APIs over Node APIs in this project.
When you change a public function, run `bun check` before yielding.
```
### "Replace the stable default instructions" — bring your own base prompt
Use `SYSTEM.md` (or `--system-prompt`). You replace the stable default instructions in block 0, but normal CLI startup still preserves the dynamic project/environment footer block (`project-prompt.md`): workstation info, context files, dir-context list, workspace tree, current date, cwd, and related project context.
```text
# ~/.omp/agent/SYSTEM.md
You are a code reviewer. Read diffs, surface issues, never edit files.
- Cite paths with backticks.
- Prefer concrete fixes over abstract advice.
```
If you do this and want default tool guidance, exploration rules, or workflow rules, copy what you need from `packages/coding-agent/src/prompts/system/system-prompt.md` and maintain it yourself — there is currently no way to inherit selected sections from that stable default instruction block.
### "Customize while keeping generated skills/rules/tool guidance"
Use `APPEND_SYSTEM.md`, not `SYSTEM.md`. Skills, rulebook summaries, always-apply rules, the tool inventory, and the built-in guidance that tells the model when to read `skill://<name>` are part of block 0 (`system-prompt.md`). Because `SYSTEM.md` replaces block 0, those generated lists are not available to the model in a custom system prompt.
The dynamic project/environment footer that remains after `SYSTEM.md` is only block 1 (`project-prompt.md`): workstation info, AGENTS.md context files, dir-context list, workspace tree, current date, cwd, and related project context. It does not include discovered skills.
There is currently no supported CLI mode for "replace the stable default instructions but keep the generated skills/rules/tool guidance." If you need automatic skills loading, keep the default block and add your customization via `APPEND_SYSTEM.md`. If you fully replace with `SYSTEM.md`, you must hard-code any skill names/instructions you want the model to know about, and those will not track discovery automatically.
### "Replace everything, including project context" — SDK-only
The normal CLI file/flag path intentionally preserves `defaultPrompt.slice(1)`. Code using `CreateAgentSessionOptions.systemPrompt` directly can return a full replacement array and omit the project footer, but that is not what `.omp/SYSTEM.md`, `~/.omp/agent/SYSTEM.md`, or `--system-prompt` do.
### "Replace, but keep one section of the default instructions" — not directly supported
There is no built-in way to inherit specific sections from `system-prompt.md` while replacing the rest. The supported CLI modes are: append to the default prompt, or replace block 0 and keep the dynamic footer.
---
## 5) Deduplication
The CLI path avoids double-injecting discovered `SYSTEM.md` by replacing block 0 after the default prompt blocks are rendered. Any `systemPromptCustomization` from the secondary capability path would have been rendered into block 0, and that block is discarded when `main.ts` applies `[resolvedSystemPrompt, ...defaultPrompt.slice(1)]`.
Inside `buildSystemPrompt` itself, secondary customization and always-apply rules are still deduplicated:
- `dedupePromptSource` drops a `systemPromptCustomization` block when it already appears in an internally supplied `customPrompt` or append prompt.
- `dedupeAlwaysApplyRules` omits always-apply rules whose body appears verbatim in any of `{customPrompt, appendPrompt, systemPromptCustomization}`.
---
## 6) Discovery paths
Only one path actually drives the customization a CLI user sees: the primary CLI path. The capability layer exists but its `SYSTEM.md` output never reaches the rendered prompt under normal CLI startup.
- The primary CLI path (`discoverSystemPromptFile` / `discoverAppendSystemPromptFile` in `main.ts`, which feeds `resolvedSystemPrompt` / `resolvedAppendPrompt`) calls `findConfigFile`. `findConfigFile` checks only `<cwd>/.omp`, `<cwd>/.claude`, `<cwd>/.codex`, `<cwd>/.gemini`, and the user-level equivalents — it does **not** walk up ancestors. Files in `<ancestor>/.omp/SYSTEM.md` are ignored when `omp` is started from a subdirectory.
- The secondary capability path (`loadSystemPromptFiles` → builtin discovery) does walk up via `findNearestProjectConfigDir` and requires the project `.omp/` directory to be non-empty. Its result is rendered into the template variable `systemPromptCustomization`. Under normal CLI startup the default template (`system-prompt.md`) never references that variable, so ancestor-walk capability content has no user-visible effect.
Net effect for CLI users: put `SYSTEM.md` / `APPEND_SYSTEM.md` directly under `<cwd>/.omp` (or another supported config base under cwd) or in the user-level location (`~/.omp/agent/SYSTEM.md` etc.). Ancestor paths are not searched.
---
## 7) Quick reference
| Goal | Use |
|---|---|
| Add an instruction on top of the full default prompt | `APPEND_SYSTEM.md` or `--append-system-prompt` |
| Replace the stable default instructions but keep project/environment context | `SYSTEM.md` or `--system-prompt` |
| Preserve generated skills/rules/tool guidance while customizing | `APPEND_SYSTEM.md`; `SYSTEM.md` replaces that generated block |
| Use `{{cwd}}` / `{{date}}` / other internals in my file | Not supported. Files are inserted verbatim. |
| Inherit specific sections from `system-prompt.md` | Not supported; use append, or copy what you need into `SYSTEM.md`. |
| Override at a per-repo level | Project `.omp/SYSTEM.md` under the cwd you launch `omp` from |
| Override globally | `~/.omp/agent/SYSTEM.md` or `~/.omp/agent/APPEND_SYSTEM.md` |
+6 -7
View File
@@ -195,13 +195,12 @@ In `runSubprocess` (`src/task/executor.ts`):
So deeper levels cannot spawn further tasks even if the agent definition includes `spawns`.
## Plan mode caveat (current implementation)
## Plan mode behavior
`TaskTool.execute` computes an `effectiveAgent` for plan mode (prepends plan-mode prompt, forces read-only tool subset, clears spawns), but `runSubprocess` is called with `agent` rather than `effectiveAgent`.
When parent plan mode is enabled, `TaskTool.execute` builds an `effectiveAgent` before launching subprocesses:
Current effect:
- prepends the plan-mode subagent system prompt
- restricts tools to `read`, `search`, `find`, `lsp`, and `web_search`
- clears child spawns
- model override / thinking level / output schema are derived from `effectiveAgent`
- system prompt and tool/spawn restrictions from `effectiveAgent` are not passed through in this call path
This is an implementation caveat worth knowing when reading plan-mode behavior expectations.
The same `effectiveAgent` is used for subprocess launch, model/thinking overrides, and output-schema selection.
+1
View File
@@ -85,6 +85,7 @@ If omitted, export code derives defaults from resolved theme colors.
- `symbols.preset` sets a theme-level default symbol set.
- `symbols.overrides` can override individual `SymbolKey` values.
- `symbols.spinnerFrames` overrides the loading spinner frames. Accepts either a flat `string[]` (applied to both spinner types) or an object `{ "status"?: string[], "activity"?: string[] }` to override each type independently. Any type not specified falls back to the symbol preset's default frames. `status` drives the ~12.5fps spinner used by loaders and tool-execution indicators; `activity` drives the ~60fps spinner used by markdown progress bars and similar high-frequency UI.
Runtime precedence:
+5 -5
View File
@@ -1,6 +1,6 @@
# ask
> Prompts the interactive user for one or more choices or free-form answers.
> Prompts the interactive user for one or more option-picker or free-form answers.
## Source
- Entry: `packages/coding-agent/src/tools/ask.ts`
@@ -22,7 +22,7 @@
| --- | --- | --- | --- |
| `id` | `string` | Yes | Stable identifier used in multi-question results. |
| `question` | `string` | Yes | Prompt text shown to the user. |
| `options` | `{ label: string }[]` | Yes | Explicit options. The UI always appends `Other (type your own)`; callers must not include it. |
| `options` | `{ label: string }[]` | Yes | Option labels for the picker. The schema does not require a minimum length; the UI always appends `Other (type your own)`, and callers must not include it. |
| `multi` | `boolean` | No | Enables multi-select mode. Default: `false`. |
| `recommended` | `number` | No | Zero-based recommended option index. In single-select mode the label gets ` (Recommended)` appended in the UI. |
@@ -39,7 +39,7 @@
## Flow
1. `AskTool.createIf()` only registers the tool when `session.hasUI` is true; headless sessions never get it.
2. `execute()` requires `context.ui`; if missing it aborts the context and throws `ToolAbortError("Ask tool requires interactive mode")`.
3. It reads `ask.timeout` from settings, converts seconds to milliseconds, and disables timeout entirely while plan mode is enabled (`packages/coding-agent/src/tools/ask.ts`).
3. It reads `ask.timeout` from settings, converts seconds to milliseconds (`0` disables timeout), and disables timeout entirely while plan mode is enabled (`packages/coding-agent/src/tools/ask.ts`).
4. If `ask.notify` is not `off`, it sends a terminal notification: `Waiting for input`.
5. For each question, `askSingleQuestion()` drives either:
- single-select list + optional editor for `Other`
@@ -68,8 +68,8 @@
## Limits & Caps
- `questions` must contain at least 1 item (`askSchema` in `packages/coding-agent/src/tools/ask.ts`).
- `ask.timeout` default is `30` seconds; `0` disables timeout (`packages/coding-agent/src/config/settings-schema.ts`).
- Prompt guidance says provide 2-5 options, but code does not enforce that (`packages/coding-agent/src/prompts/tools/ask.md`).
- `ask.timeout` default is `0` seconds, which disables timeout (`packages/coding-agent/src/config/settings-schema.ts`). Configured non-zero values are seconds.
- Prompt guidance says provide 2-5 options, but code only requires the `options` array field and does not enforce a minimum or maximum length (`packages/coding-agent/src/prompts/tools/ask.md`).
- Timeout only applies to the option picker; once the user chooses `Other`, the editor has no timeout (`packages/coding-agent/src/prompts/tools/ask.md`).
## Errors
+6 -6
View File
@@ -7,12 +7,12 @@
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/ast-edit.md`
- Key collaborators:
- `crates/pi-natives/src/ast.rs` — native rewrite planning and file mutation
- `crates/pi-natives/src/language/mod.rs` — language aliases and extension inference
- `crates/pi-ast/src/language/mod.rs` — language aliases and extension inference used by the native wrapper.
- `packages/coding-agent/src/tools/path-utils.ts` — path/glob parsing and multi-path resolution
- `packages/coding-agent/src/tools/resolve.ts` — preview/apply queueing
- `packages/coding-agent/src/tools/render-utils.ts` — parse-error dedupe and display caps
- `packages/coding-agent/src/utils/file-display-mode.ts` — hashline vs line-number diff references
- `packages/coding-agent/src/hashline/hash.ts` — stable hashline diff anchors
- `packages/hashline/src/format.ts` — stable hashline header formatting for preview anchors
- `packages/natives/native/index.d.ts` — JS-visible native binding contract
## Inputs
@@ -33,7 +33,7 @@ Shared AST pattern grammar and language catalog: see [`ast_grep`](./ast-grep.md#
## Outputs
- Single-shot preview result from `ast_edit` itself.
- Model-facing `content` is one text block showing proposed edits, grouped by file for directory/multi-file runs.
- Each change renders as two lines: `-REF|before` and `+REF|after` in hashline mode, or `-LINE:COLUMN before` / `+LINE:COLUMN after` when hashlines are off.
- Each change renders as two lines. Hashline mode uses `-LINE:before` / `+LINE:after` under a `¶PATH#TAG` header; plain mode uses `-LINE:COLUMN before` / `+LINE:COLUMN after`.
- Only the first line of each `before`/`after` snippet is shown, truncated to 120 characters in the wrapper.
- `Limit reached; narrow paths.` and formatted parse issues are appended when applicable.
- If no rewrites match, text is `No replacements made` plus formatted parse issues when present.
@@ -62,7 +62,7 @@ Shared AST pattern grammar and language catalog: see [`ast_grep`](./ast-grep.md#
- parses each file, skips files with syntax-error trees, collects `replace_by(...)` edits for every match, enforces replacement and file caps, and returns textual before/after slices plus source ranges.
7. The TS wrapper deduplicates parse errors, groups changes by file, and renders preview diff lines.
8. If preview found replacements and `applied` is false, `queueResolveHandler(...)` registers a forced `resolve` action and injects a `resolve-reminder` steering message.
9. On `resolve(action: "apply")`, the queued callback reruns the same rewrite set with `dryRun: false`, recomputes counts, and rejects the apply as an error if the live result no longer matches the preview (`stalePreview`).
9. On `resolve(action: "apply")`, the queued callback reruns the same rewrite set with `dryRun: false`, recomputes counts, and returns an error result if the live result no longer matches the preview (`stalePreview`). The current implementation compares replacement totals and per-file counts after the rerun; if the new run has already written different counts, the result is marked error.
10. On a non-stale apply, the callback returns `Applied N replacements in M files.`; on discard, `resolve` returns a discard message without mutating files.
## Modes / Variants
@@ -110,7 +110,7 @@ Shared AST pattern grammar and language catalog: see [`ast_grep`](./ast-grep.md#
- With `failOnParseError: false` (the wrapper always uses this), pattern compile failures and file parse failures become `parseErrors` instead of aborting the whole run.
- If every rewrite pattern fails to compile, native `ast_edit` returns a successful zero-replacement result with `parseErrors` populated.
- Files containing tree-sitter error nodes are skipped for rewriting; they do not get partial edits.
- Apply can fail after a successful preview if the preview becomes stale. The resolve callback compares replacement totals and per-file counts and returns an error result rather than applying a mismatched preview silently.
- Apply can fail after a successful preview if the preview becomes stale. The resolve callback compares replacement totals and per-file counts and returns an error result rather than silently reporting success for a mismatched preview.
## Notes
- `ast_edit` does not expose the native `lang`, `strictness`, `selector`, `maxReplacements`, `failOnParseError`, or `timeoutMs` fields to the model. The runtime fixes the call shape to a preview-first, smart-strictness, best-effort parse mode.
@@ -118,4 +118,4 @@ Shared AST pattern grammar and language catalog: see [`ast_grep`](./ast-grep.md#
- Idempotency is not enforced syntactically. A rewrite like `foo($A) -> foo($A)` previews zero changes because output equals input; a rewrite that keeps matching its own output may still produce replacements on repeated calls.
- Rewrites are accumulated per file, then applied from the end of the file backward after an overlap check. Independent matches can coexist; overlapping matches abort the run.
- Native rewrite rule order is by pattern-string sort, not by the original `ops` array order, because `normalize_rewrite_map(...)` sorts the `(pattern, rewrite)` pairs.
- Preview/apply parity is validated only by totals and per-file counts, not by a byte-for-byte diff of every replacement payload.
- Preview/apply parity is validated by totals and per-file counts after the apply rerun, not by a byte-for-byte diff of every replacement payload.
+2 -2
View File
@@ -7,7 +7,7 @@
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/ast-grep.md`
- Key collaborators:
- `crates/pi-natives/src/ast.rs` — native scan, parse, match engine
- `crates/pi-natives/src/language/mod.rs` — language aliases and extension inference
- `crates/pi-ast/src/language/mod.rs` — language aliases and extension inference used by the native wrapper.
- `packages/coding-agent/src/tools/path-utils.ts` — path/glob parsing and multi-path resolution
- `packages/coding-agent/src/tools/render-utils.ts` — parse-error dedupe and display caps
- `packages/coding-agent/src/tools/match-line-format.ts` — hashline match rendering
@@ -30,7 +30,7 @@ Pattern grammar and language support exposed to the model:
- Metavariable names must be uppercase and must stand for whole AST nodes, not partial tokens or string fragments.
- Reusing the same metavariable requires identical code at each occurrence.
- Patterns must parse as one valid AST node for the inferred target language.
- Supported canonical languages come from `SupportLang::all_langs()` in `crates/pi-natives/src/language/mod.rs`: `astro`, `bash`, `c`, `cmake`, `cpp`, `csharp`, `dart`, `clojure`, `css`, `diff`, `dockerfile`, `elixir`, `erlang`, `go`, `graphql`, `haskell`, `hcl`, `html`, `ini`, `java`, `javascript`, `json`, `just`, `julia`, `kotlin`, `lua`, `make`, `markdown`, `nix`, `objc`, `ocaml`, `odin`, `perl`, `php`, `powershell`, `protobuf`, `python`, `r`, `regex`, `ruby`, `rust`, `scala`, `solidity`, `sql`, `starlark`, `svelte`, `swift`, `toml`, `tlaplus`, `tsx`, `typescript`, `verilog`, `vue`, `xml`, `yaml`, `zig`.
- Supported canonical languages come from `SupportLang::all_langs()` in `crates/pi-ast/src/language/mod.rs`: `astro`, `bash`, `c`, `cmake`, `cpp`, `csharp`, `dart`, `clojure`, `css`, `diff`, `dockerfile`, `elixir`, `erlang`, `go`, `graphql`, `haskell`, `hcl`, `html`, `ini`, `java`, `javascript`, `json`, `just`, `julia`, `kotlin`, `lua`, `make`, `markdown`, `nix`, `objc`, `ocaml`, `odin`, `perl`, `php`, `powershell`, `protobuf`, `python`, `r`, `regex`, `ruby`, `rust`, `scala`, `solidity`, `sql`, `starlark`, `svelte`, `swift`, `toml`, `tlaplus`, `tsx`, `typescript`, `verilog`, `vue`, `xml`, `yaml`, `zig`.
## Outputs
- Single-shot tool result.
+34 -23
View File
@@ -20,7 +20,7 @@
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `command` | `string` | Yes | Shell command text to execute. A leading `cd <path> && ...` is rewritten into `cwd` only when `cwd` was omitted. |
| `env` | `Record<string, string>` | No | Extra environment variables. Keys must match `^[A-Za-z_][A-Za-z0-9_]*$` or the tool throws. Values also go through internal-URL expansion. |
| `env` | `Record<string, string>` | No | Extra environment variables. Keys must match `^[A-Za-z_][A-Za-z0-9_]*$` or the tool throws. Values go through internal-URL expansion and are passed as environment values, not shell text. |
| `timeout` | `number` | No | Timeout in seconds. Default `300`; clamped to `1..3600` by `clampTimeout("bash", ...)`. |
| `cwd` | `string` | No | Working directory, resolved against `session.cwd` via `resolveToCwd`. Must exist and be a directory. |
| `pty` | `boolean` | No | Request PTY mode. Default `false`. PTY is used only when `pty: true`, `PI_NO_PTY !== "1"`, and the tool context has a UI. |
@@ -32,67 +32,78 @@ The tool returns a single `text` content block plus optional `details`.
- Success, foreground:
- `content[0].text`: command output, or `(no output)` when the command produced nothing.
- `details.timeoutSeconds`: effective timeout after clamping.
- `details.requestedTimeoutSeconds`: only present when the requested timeout was clamped.
- `details.requestedTimeoutSeconds`: present when the requested timeout differed from the effective timeout.
- `details.wallTimeMs`: elapsed wall-clock milliseconds for completed local/client-terminal runs.
- `details.terminalId`: present when execution was routed through a client terminal bridge.
- `details.exitCode`: present when the command completed with a non-zero exit code.
- `details.meta.truncation`: present when output was truncated in memory; includes `artifactId` when full output spilled to an artifact.
- non-zero exits return a tool result marked `isError` with output plus `Command exited with code <n>`; they are not thrown.
- Success, background start (`async: true` or auto-background):
- `content[0].text`: optional preview tail, timeout notice if any, then `Background job <id> started: <label>` with follow-up instructions.
- `details.async`: `{ state: "running", jobId, type: "bash" }`.
- Background progress / completion:
- delivered through `onUpdate` / async job manager, not the initial return.
- running updates contain tail text and `details.async.state: "running"` only after the job is considered backgrounded.
- completion/failure updates carry final text and `details.async.state: "completed" | "failed"`.
- completion/failure updates carry final text and `details.async.state: "completed" | "failed"`. A non-zero exit is recorded as a failed background job.
- Failure:
- the tool throws `ToolError` / `ToolAbortError`; non-zero exits are surfaced as errors, not success results.
- unfinished execution (`cancelled`, timeout, missing exit status), validation failures, and intercepted commands throw `ToolError` / `ToolAbortError`.
Stdout and stderr are merged before the model sees them. Non-zero exit codes are appended to the thrown error text as `Command exited with code <n>`.
Stdout and stderr are merged before the model sees them. Definite non-zero exit codes are appended to the returned error result text as `Command exited with code <n>`.
## Flow
1. `BashTool.execute()` in `packages/coding-agent/src/tools/bash.ts` reads `command`, normalizes `env`, and defaults `timeout` to `300`.
2. If `cwd` is absent, it rewrites a leading `cd <path> && ...` into the structured `cwd` field and strips that prefix from `command`.
3. If `async: true` is requested while `async.enabled` is off, it throws `ToolError` before any execution.
4. If `bashInterceptor.enabled` is on, `checkBashInterception()` runs against both the original command and the `cd`-stripped command. A matching enabled rule throws before URL expansion or execution.
5. `expandInternalUrls()` rewrites supported internal URLs inside `command`, each `env` value, and protocol-looking `cwd` values. Command/env replacements are shell-escaped unless `noEscape` is requested by the caller path.
5. `expandInternalUrls()` rewrites supported internal URLs inside `command`, each `env` value, and protocol-looking `cwd` values. Command replacements are shell-escaped; `env` and `cwd` replacements use raw filesystem/string values because they are not interpolated into shell text.
6. `resolveToCwd()` resolves `cwd` against `session.cwd`; `fs.stat()` verifies that the target exists and is a directory.
7. `clampTimeout("bash", requestedTimeoutSec)` enforces `TOOL_TIMEOUTS.bash` (`default: 300`, `min: 1`, `max: 3600`). When clamped, `#buildCompletedResult()` / `#buildBackgroundStartResult()` append a notice line.
8. Execution path splits:
1. `async: true` -> `#startManagedBashJob()` registers a session async job and returns immediately.
2. Non-PTY with `bash.autoBackground.enabled` and an async job manager -> starts a managed job, waits up to `min(thresholdMs, timeoutMs - 1000)`, and either returns the completed result or converts the run into a background job.
3. Otherwise runs foreground execution.
9. Foreground non-PTY calls `executeBash()` from `packages/coding-agent/src/exec/bash-executor.ts`.
3. Non-PTY client-terminal bridge, when the session advertises terminal capability and `pty` is false -> creates a remote terminal, streams/polls current output, and releases the terminal after completion.
4. Otherwise runs foreground execution.
9. Foreground non-PTY without client terminal calls `executeBash()` from `packages/coding-agent/src/exec/bash-executor.ts`.
10. Foreground PTY calls `runInteractiveBashPty()` from `packages/coding-agent/src/tools/bash-interactive.ts`.
11. Both paths allocate an output artifact first when `session.allocateOutputArtifact` is available. The artifact path/id are passed into the sink so large output can spill to disk.
11. Local non-PTY and PTY paths allocate an output artifact first when `session.allocateOutputArtifact` is available. The artifact path/id are passed into the sink so large output can spill to disk.
12. `executeBash()` loads shell settings, optional shell snapshot, and shell minimizer settings, then runs via a persistent native `Shell` session or one-shot `executeShell()`. `docs/bash-tool-runtime.md` covers that path in detail.
13. `runInteractiveBashPty()` creates a `PtySession`, overlays an xterm-backed console UI, forwards user key input into the PTY, captures output through `OutputSink`, and kills the PTY on dismiss/dispose.
14. On completion, `#buildCompletedResult()` formats `(no output)` when needed, attaches truncation metadata from the `OutputSink` summary, and re-checks exit status / timeout / cancellation before returning.
15. On non-zero exit, timeout, missing exit status, or cancellation, `#buildResultText()` throws with the captured output included in the error message.
14. Client-terminal bridge mode calls `session.getClientBridge().createTerminal(...)`, emits `terminalId` updates, polls output until exit/timeout/abort, maps signal exits to `137`, and releases the handle in `finally`.
15. On completion, `#buildCompletedResult()` formats `(no output)` when needed, attaches truncation metadata from the output summary, appends wall-time/timeout/exit notices, and re-checks unfinished status before returning.
16. On timeout, missing exit status, or cancellation, the tool throws with captured output included when available.
## Modes / Variants
1. Foreground non-PTY
- Default path.
1. Foreground non-PTY local
- Default path when no client terminal bridge is available.
- Uses `executeBash()`.
- Streams tail-only updates through `streamTailUpdates()` and `TailBuffer(DEFAULT_MAX_BYTES)`.
2. Foreground PTY
2. Foreground non-PTY client terminal
- Used when `session.getClientBridge()?.capabilities.terminal` is true, `createTerminal` exists, and `pty` is false.
- Streams current terminal output via polling updates with `details.terminalId`.
- Enforces the same timeout and abort behavior, then releases the terminal handle.
3. Foreground PTY
- Requires `pty: true`, UI context, and `PI_NO_PTY !== "1"`.
- Uses `runInteractiveBashPty()` and a `PtySession` overlay.
- Supports interactive input; `Esc` kills the session from the overlay.
3. Explicit background job
4. Explicit background job
- Requires `async: true` and `async.enabled`.
- Registers a job with `session.asyncJobManager` and returns `{ state: "running", jobId }` immediately.
4. Auto-backgrounded non-PTY job
5. Auto-backgrounded non-PTY job
- Requires `bash.autoBackground.enabled`, no PTY, and an async job manager.
- Starts like a foreground managed job, then backgrounds it when it outlives the wait window.
5. Intercepted command
6. Intercepted command
- No subprocess created.
- Returns a `ToolError` pointing the model at `read`, `search`, `find`, `edit`, or `write`.
## Side Effects
- Filesystem
- Validates `cwd` with `fs.stat()`.
- May allocate and write artifact files for full output (`bash`) and minimizer-preserved raw output (`bash-original`).
- May allocate and write artifact files for full local output (`bash`) and minimizer-preserved raw output (`bash-original`).
- `expandInternalUrls(..., { ensureLocalParentDirs: true })` creates parent directories for `local://` paths before execution.
- Subprocesses / native bindings
- Non-PTY uses native shell execution via `@oh-my-pi/pi-natives` (`Shell.run()` or `executeShell()`).
- Subprocesses / native bindings / client terminal
- Non-PTY local execution uses native shell execution via `@oh-my-pi/pi-natives` (`Shell.run()` or `executeShell()`).
- PTY uses native `PtySession.start()`.
- Client-terminal mode delegates process execution to the connected client terminal capability.
- Session state
- Reads session settings for async, auto-background, interceptor, tool availability, and shell configuration.
- Registers jobs with `session.asyncJobManager` for explicit/auto background runs.
@@ -126,15 +137,15 @@ Stdout and stderr are merged before the model sees them. Non-zero exit codes are
- Internal URL expansion:
- unsupported scheme, unknown skill, path traversal, missing router support, or router resolution failures all throw `ToolError` from `packages/coding-agent/src/tools/bash-skill-urls.ts`.
- Execution:
- non-zero exit -> thrown `ToolError` containing captured output plus `Command exited with code <n>`.
- non-zero exit -> returned tool result marked `isError`, with `details.exitCode` and text ending in `Command exited with code <n>`.
- missing exit code -> thrown `ToolError` with `Command failed: missing exit status`.
- timeout -> thrown `ToolError`; PTY uses `Command timed out after <n> seconds`, non-PTY executor returns cancelled output that `BashTool` converts to an error.
- timeout -> thrown `ToolError`; PTY/client-terminal modes use `Command timed out after <n> seconds`, non-PTY executor returns cancelled output that `BashTool` converts to an error.
- user abort -> `ToolAbortError` when the caller signal is aborted.
- Artifact allocation / artifact save failures are swallowed in `saveBashOriginalArtifact()` and `OutputSink.#createFileSink()`; execution continues without that artifact.
## Notes
- `strict = true` and `concurrency = "exclusive"` are set on `BashTool`; the tool does not run concurrently with another bash tool call in the same session.
- `command` and `env` URL expansions shell-escape replacements; `cwd` expansion uses `noEscape: true` because it becomes a filesystem path argument, not shell text.
- `command` URL expansions shell-escape replacements; `env` and `cwd` expansion use `noEscape: true` because they become environment values / filesystem paths, not shell text.
- `checkBashInterception()` blocks only when the matching rule's `tool` name is present in `ctx.toolNames`; missing tools disable their corresponding rule.
- Default interceptor rules come from `DEFAULT_BASH_INTERCEPTOR_RULES` in `packages/coding-agent/src/config/settings-schema.ts`:
- `cat|head|tail|less|more` -> `read`
+5 -5
View File
@@ -46,7 +46,7 @@
| --- | --- | --- | --- |
| `url` | `string` | No | Navigate after the tab is ready. Existing reusable tabs also navigate when `url` is supplied. |
| `viewport` | `{ width: number; height: number; scale?: number }` | No | Requested viewport. For headless launch this becomes the initial viewport; for a page it is applied with `page.setViewport()`. `scale` maps to Puppeteer `deviceScaleFactor`. |
| `wait_until` | `"load" \| "domcontentloaded" \| "networkidle0" \| "networkidle2"` | No | Navigation wait condition. Defaults to `"networkidle2"` where omitted. |
| `wait_until` | `"load" \| "domcontentloaded" \| "networkidle0" \| "networkidle2"` | No | Navigation wait condition. Defaults to `"load"` where omitted, including `open` navigation and later `tab.goto(...)`. |
| `dialogs` | `"accept" \| "dismiss"` | No | Installs a page `dialog` handler that auto-accepts or auto-dismisses dialogs. Omitted means no handler. |
| `app` | `{ path?: string; cdp_url?: string; args?: string[]; target?: string }` | No | Selects browser kind. No `app` uses the session `browser.headless` setting. `app.path` is resolved against the session cwd and used as the executable path for spawn/attach reuse. `app.cdp_url` connects to an existing CDP endpoint. `args` are appended only when spawning `app.path`. `target` is only used for attached/spawned-app page selection. |
@@ -95,9 +95,9 @@ The tool returns one result per call; no streaming partial output is emitted fro
5. `open` acquires a tab through `acquireTab()` (`packages/coding-agent/src/tools/browser/tab-supervisor.ts`):
- same-name + same-browser + alive tab is reused unless `dialogs` changed;
- same-name but different browser handle, dead state, or changed dialog policy forces release and recreation;
- reusing with a new `url` navigates by issuing `await tab.goto(...)` through the worker.
- reusing with a new `url` navigates by issuing `await tab.goto(...)` through the worker, defaulting to `waitUntil: "load"` when `wait_until` is omitted.
6. New tabs build a `WorkerInitPayload` in `buildInitPayload()`:
- headless mode sends `url`, `waitUntil`, `viewport`, `dialogs`, and timeout;
- headless mode sends `url`, `waitUntil`, `viewport`, `dialogs`, and timeout; the worker defaults missing `waitUntil` to `"load"`.
- attach mode resolves a page with `pickElectronTarget()`, gets its target id, and sends `targetId` plus `dialogs`.
7. `acquireTab()` spawns a dedicated Bun `Worker` from `tab-worker-entry.ts`; if that fails it falls back to inline execution in the main thread (`spawnInlineWorker()`), preserving behavior but losing protection against synchronous infinite loops.
8. `WorkerCore.#init()` (`packages/coding-agent/src/tools/browser/tab-worker.ts`) connects back to the browser websocket endpoint. Headless mode opens a new page, applies stealth patches, applies viewport, installs dialog handling if requested, and optionally navigates. Attach mode resolves the requested target page and optionally installs dialog handling.
@@ -190,7 +190,7 @@ The tool returns one result per call; no streaming partial output is emitted fro
- A timed-out `run` aborts the worker execution path and can tear down the tab.
## Limits & Caps
- Tool timeout clamp: default `30` s, min `1` s, max `30` s (`TOOL_TIMEOUTS.browser` in `packages/coding-agent/src/tools/tool-timeouts.ts`).
- Tool timeout clamp: default `30` s, min `1` s, max `300` s (`TOOL_TIMEOUTS.browser` in `packages/coding-agent/src/tools/tool-timeouts.ts`).
- Supervisor grace period around init/run/close: `750` ms (`GRACE_MS` in `packages/coding-agent/src/tools/browser/tab-supervisor.ts`).
- Puppeteer protocol timeout for launch/connect operations: `60_000` ms (`BROWSER_PROTOCOL_TIMEOUT_MS` in `packages/coding-agent/src/tools/browser/launch.ts`).
- Connected-browser CDP readiness wait: `5_000` ms before `puppeteer.connect()` (`packages/coding-agent/src/tools/browser/registry.ts`).
@@ -211,7 +211,7 @@ The tool returns one result per call; no streaming partial output is emitted fro
- `No page targets available on the attached browser`
- `No page target matched "...". Available pages:\n...`
- `Target ... is no longer available on the attached browser`
- Spawned-app path validation requires an absolute executable path, not an app bundle path.
- Spawned-app path validation requires an absolute executable path after cwd resolution, not an app bundle directory path.
- Spawn/attach failures are wrapped into `ToolError`s such as `Timed out waiting for CDP endpoint ...`, `Failed to attach to ...`, or `Connected to ... but puppeteer.connect failed: ...`.
- `tab` helper errors are user-visible `ToolError`s, including unsupported selector prefix, stale/unknown element id, invalid drag target, missing upload files, non-`<select>` for `tab.select()`, non-file-input for `tab.uploadFile()`, and screenshot selector misses.
- On run timeout, the worker reports `Browser code execution timed out after <ms>ms`; the supervisor may escalate to `Browser code execution hung past grace; tab killed` if the worker does not respond after the grace window.
-71
View File
@@ -1,71 +0,0 @@
# calc
> Evaluates one or more arithmetic expressions and returns formatted numeric results.
## Source
- Entry: `packages/coding-agent/src/tools/calculator.ts`
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/calculator.md`
- Key collaborators:
- `packages/coding-agent/src/tui.ts` — status lines and tree-list rendering
- `packages/coding-agent/src/tools/render-utils.ts` — preview limits and formatting helpers
## Inputs
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `calculations` | `Calculation[]` | Yes | Batch of expressions to evaluate in order. |
### `Calculation`
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `expression` | `string` | Yes | Arithmetic expression string. |
| `prefix` | `string` | Yes | Prepended verbatim to the rendered numeric result. |
| `suffix` | `string` | Yes | Appended verbatim to the rendered numeric result. |
## Outputs
- Single-shot result.
- `content[0].text` is the newline-joined `prefix + value + suffix` string for each calculation.
- `details.results` is an array of `{ expression, value, output }`.
- On renderer fallback, if `details` is missing but `content[0].text` exists, the TUI tries to pair each output line with the original expressions from call args.
## Flow
1. `execute()` wraps evaluation in `untilAborted(...)`.
2. For each entry, `evaluateExpression(...)` tokenizes the expression, parses it with a recursive-descent parser, rejects non-finite outputs, and normalizes `-0` to `0`.
3. `tokenizeExpression(...)` accepts whitespace, parentheses, operators, and number literals; any other character throws immediately.
4. `ExpressionParser` applies precedence in this order: `+ -`, `* / %`, unary `+ -`, exponentiation `**`, parentheses/literals.
5. Exponentiation is right-associative (`2 ** 3 ** 2` parses as `2 ** (3 ** 2)`).
6. Each numeric result is formatted with `String(value)` and wrapped with the provided `prefix` and `suffix`.
7. The tool returns text output plus structured `details`.
## Side Effects
- Background work / cancellation
- Supports abort via `untilAborted(...)`.
- Session state
- None.
- Filesystem / Network / Subprocesses
- None.
## Limits & Caps
- Supported operators: `+`, `-`, `*`, `/`, `%`, `**` (`packages/coding-agent/src/tools/calculator.ts`).
- Supported numeric literals:
- decimal integers/floats, including leading-dot forms like `.5`
- scientific notation like `1e10`, `2.5E-3`
- hexadecimal `0x...`
- binary `0b...`
- octal `0o...`
- Results must be finite; `Infinity` and `NaN` are rejected.
- The renderer collapses long result lists using `PREVIEW_LIMITS.COLLAPSED_ITEMS` from `packages/coding-agent/src/tools/render-utils.ts`.
## Errors
- Invalid characters: e.g. `Invalid character "x" in expression`.
- Malformed numbers: invalid prefixed literal, invalid exponent, invalid number.
- Syntax errors: `Unexpected token in expression`, `Unexpected end of expression`, `Missing closing parenthesis`, `Expression is empty`.
- Non-finite arithmetic: `Expression result is not a finite number`.
- Any evaluation error aborts the whole batch; the tool does not return partial successes.
## Notes
- Despite the schema example showing `sqrt(16)`, the parser does not support functions, identifiers, units, or constants; only numeric literals, operators, and parentheses are accepted.
- Precision is plain JavaScript `number` semantics throughout, including floating-point rounding behavior.
- `/` and `%` use JavaScript numeric operators directly; there is no integer-only mode or unit handling.
- Unary operators bind tighter than `*`/`/`/`%` but looser than exponentiation because unary parsing delegates to `#parsePower()`.
-1
View File
@@ -81,4 +81,3 @@ You are in an active checkpoint. You MUST call rewind with your investigation fi
- blob-store contents
- SQLite history rows from `packages/coding-agent/src/session/history-storage.ts`
- auth or agent records from `packages/coding-agent/src/session/agent-storage.ts`
- If the turn ends with `stopReason === "aborted"` while a checkpoint is active, `AgentSession` clears `#checkpointState` and `#pendingRewindReport` instead of preserving a half-finished checkpoint.
+8 -5
View File
@@ -44,8 +44,8 @@
| `port` | `number` | No | Remote attach port. If no adapter is forced, attach prefers `debugpy` when `port` is present. |
| `host` | `string` | No | Remote attach host for `attach`. |
| `levels` | `number` | No | Max stack frames for `stack_trace`. |
| `memory_reference` | `string` | No | Memory reference/address for `disassemble`, `read_memory`, `write_memory`. `disassemble` also accepts it via `instruction_reference` fallback logic in `resolveDisassemblyReference()`. |
| `instruction_reference` | `string` | No | Instruction breakpoint reference; required for instruction breakpoint actions. |
| `memory_reference` | `string` | No | Memory reference/address for `disassemble`, `read_memory`, `write_memory`. `disassemble` uses this when provided; otherwise it falls back to the current stopped location's instruction-pointer reference if the adapter supplied one. |
| `instruction_reference` | `string` | No | Instruction breakpoint reference; required for instruction breakpoint actions. Not used by `disassemble`. |
| `instruction_count` | `number` | No | Required for `disassemble`. |
| `instruction_offset` | `number` | No | Instruction offset for `disassemble`. |
| `count` | `number` | No | Byte count for `read_memory`. Required there. |
@@ -70,7 +70,7 @@
- `set_data_breakpoint` / `remove_data_breakpoint`: `data_id`
- `evaluate`: `expression`
- `variables`: `variable_ref` or `scope_id`
- `disassemble`: capability `supportsDisassembleRequest`, plus `instruction_count`
- `disassemble`: capability `supportsDisassembleRequest`, plus `instruction_count`, and either `memory_reference` or a current stopped location with `instructionPointerReference`
- `read_memory`: capability `supportsReadMemoryRequest`, plus `memory_reference` and `count`
- `write_memory`: capability `supportsWriteMemoryRequest`, plus `memory_reference` and `data`
- `modules`: capability `supportsModulesRequest`
@@ -108,6 +108,7 @@ The agent tool returns a standard `toolResult()` payload from `packages/coding-a
Streaming/UI behavior:
- The tool renderer merges call and result (`mergeCallAndResult: true`) and renders inline.
- `debug.ts` itself does not emit progress updates through `_onUpdate`; result delivery is single-shot.
- Approval is action-sensitive: read-only actions (`output`, `threads`, `stack_trace`, `scopes`, `variables`, `disassemble`, `read_memory`, `loaded_sources`, `modules`, `sessions`) request read approval; all other actions request exec approval.
- The interactive selector is UI-driven instead of model-driven. It swaps TUI components, appends status lines to the chat pane, opens files in external viewers, or writes archives/temp files.
Side-channel artifacts outside the model tool result:
@@ -168,7 +169,7 @@ Side-channel artifacts outside the model tool result:
- `threads` — fetches current threads.
- `scopes` — frame scopes for an explicit `frame_id` or the current stopped frame.
- `variables` — variables for `variable_ref` or `scope_id`.
- `disassemble` — require `supportsDisassembleRequest`; disassembles around a memory reference.
- `disassemble` — require `supportsDisassembleRequest`; disassembles around `memory_reference`, or around the current stopped instruction pointer when no memory reference is supplied.
- `read_memory` — require `supportsReadMemoryRequest`; returns address, base64 data, unreadable-byte count.
- `write_memory` — require `supportsWriteMemoryRequest`; writes base64 data and reports bytes written.
- `modules` — require `supportsModulesRequest`; optional pagination via `start_module` / `module_count`.
@@ -249,6 +250,8 @@ Side-channel artifacts outside the model tool result:
- `attach requires pid or port`
- `set_breakpoint requires file+line or function`
- `variables requires variable_ref or scope_id`
- `instruction_count is required for disassemble`
- `disassemble requires memory_reference unless the current stop location has an instruction pointer reference`
- `memory_reference is required for read_memory`
- `count is required for read_memory`
- `data is required for write_memory`
@@ -280,7 +283,7 @@ Side-channel artifacts outside the model tool result:
- Session summaries expose `needsConfigurationDone`; this is derived from adapter capabilities and whether `configurationDone` has been sent.
- Source breakpoint file paths are normalized with `path.resolve()` before caching and sending to the adapter.
- `evaluate` defaults to `repl`, so the tool can forward raw debugger commands when the adapter supports them.
- `disassemble` resolves its target from `memory_reference` first, then `instruction_reference`; it throws if neither is present.
- `disassemble` resolves its target from `memory_reference` first, then the current stopped session's `instructionPointerReference`; it throws if neither is present.
- `RawSseDebugBuffer.recordEvent()` increments `totalEvents` before bounded retention. A snapshot can therefore show fewer retained records than total observed events.
- Raw SSE buffer listener failures are swallowed so viewer bugs do not break capture.
- `createDebugLogSource()` walks daily log files newest-first, but `loadOlderLogs()` reverses each requested slice before concatenation so older chunks prepend in chronological order.
+73 -58
View File
@@ -7,8 +7,8 @@
- Model-facing prompt: `packages/hashline/src/prompt.md`
- Key collaborators:
- `packages/coding-agent/src/utils/edit-mode.ts` — selects active edit mode
- `packages/hashline/src/grammar.lark` — hashline grammar
- `packages/hashline/src/format.ts` — sigils and header constants (`¶`, `#`, `@@`, `+`, `&`, `,`)
- `packages/hashline/src/grammar.lark` — canonical constrained-decoding grammar
- `packages/hashline/src/format.ts` — sigils and header constants (`¶`, `#`, `+`, `replace`, `delete`, `insert`)
- `packages/hashline/src/input.ts` — parses `¶PATH#TAG` sections
- `packages/hashline/src/tokenizer.ts` / `packages/hashline/src/parser.ts` — tokenizes and parses ops
- `packages/hashline/src/apply.ts` — applies parsed edits to file text
@@ -22,37 +22,47 @@
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `input` | `string` | Yes | One or more file sections. Anchored sections start with `¶PATH#TAG`; hashless `¶PATH` is allowed only for new-file creation or BOF/EOF-only inserts. Optional `*** Begin Patch` / `*** End Patch` envelope is ignored if present. |
| `input` | `string` | Yes | One or more file sections. Anchored sections must start with `¶PATH#TAG`; `TAG` is the four-hex snapshot tag emitted by the latest `read`/`search`/`write`/successful `edit`. Optional `*** Begin Patch` / `*** End Patch` envelope is ignored if present. |
Patch language inside `input`:
- **File header**: `¶PATH#TAG` (or `¶PATH` for new-file / virtual-only hunks). `TAG` is three uppercase-hex chars minted by the session snapshot store.
- **Hunk header**: bare `A B` selects original lines A..B. Two numbers are REQUIRED — single-line ranges are written `A A` (`5 5`), not `5`. The range separator is normally whitespace; the parser also silently accepts `A-B`, `A..B`, and `A…B` (unicode ellipsis). Virtual variants `BOF` and `EOF` target positions before line 1 / after the last line.
- **Body rows** (one per line, immediately under the hunk header):
- `+TEXT` — add the literal line `TEXT` verbatim, including all leading whitespace.
- `+` alone — add one blank line.
- `&A..B` — re-emit original file lines A..B. Use this to keep some of the lines you selected. `&A` is accepted as `&A..A`.
- **Semantics**:
- The new content of the selected range is just the body rows top-to-bottom.
- **Empty body deletes the range entirely.**
- `BOF` / `EOF` with empty body is a no-op (nothing to insert).
- **File header**: `¶PATH#TAG`. `TAG` is four uppercase-hex chars minted by the session snapshot store.
- **Operations**:
- `replace N..M:` — replace original lines N..M with the body rows below.
- `replace block N:` — replace the whole tree-sitter block beginning on line N (its header line through its closing line) with the body rows. The line span is resolved at apply time from the file's parse tree; point N at the line that opens the construct. Errors (and steers to `replace N..M:`) when the language is unsupported, line N is blank or a closing delimiter, no node begins there, or the resolved block has a syntax error.
- `delete N..M` — delete original lines N..M. No body.
- `delete block N` — delete the whole tree-sitter block beginning on line N (resolved like `replace block N`). No body. Same resolution failure modes and `delete N..M` fallback.
- `insert before N:` — insert body rows immediately before line N.
- `insert after N:` — insert body rows immediately after line N.
- `insert head:` — insert body rows at the start of the file.
- `insert tail:` — insert body rows at the end of the file.
- **Body rows**:
- Only body-bearing headers end in `:`.
- Every body row is `+TEXT`; `+` alone adds a blank line.
- `delete` never has body rows.
- There is no repeat row kind. To keep a line, leave it out of every range; split edits into multiple hunks when needed.
- `-` rows are invalid. Literal text beginning with `-` or `+` must be written as `+-text` / `++text`.
Anchors come from `read`/`search` output. `read` emits a `¶PATH#TAG` header from the session snapshot store and lines as `LINE:TEXT`; copy the header into the edit section and copy only the line number into hunk headers.
### Tolerated input shapes (lenient parsing)
Because models reproduce nearby shapes (`read` output, `apply_patch` envelopes, unified-diff hunks), the parser is liberal about a handful of harmless variants:
The canonical grammar is strict, but the hand parser accepts a few non-dangerous variants:
- `A` (bare single number) — REJECTED. The parser throws `single-number hunk header "A" is no longer accepted`. Spell single-line ranges as `A A`.
- `A-B`, `A..B`, `A…B` — accepted as `A B` (any of hyphen, double-dot, or unicode ellipsis works as a silent separator).
- `&A` — accepted as `&A..A`.
- Bare body rows with no `+`/`&` prefix are auto-prepended with `+` and a `BARE_BODY_AUTO_PIPED_WARNING` is appended, BUT only when every row in that block is uniformly bare. Mixed `+`/raw blocks still throw.
- `+&A..B` rows (model mistakenly prefixed a repeat with `+`) are silently rerouted as `&A..B` repeats with `PLUS_PREFIXED_REPEAT_WARNING`.
- Identical-range hunks in the same patch are coalesced last-wins with `REPLACE_PAIR_COALESCED_WARNING`.
- An overlapping bare hunk followed by a concrete hunk is treated as a stale "before then after" pair; the bare hunk is dropped with `REPLACE_PAIR_COALESCED_OVERLAP_WARNING`.
- `replace N:` — accepted as `replace N..N:`.
- `delete N` — accepted as single-line delete.
- Missing trailing colon on `replace` or `insert` — accepted.
- `replace N-M:`, `replace N…M:`, and `replace N M:` — accepted as `replace N..M:`.
- Bare body rows with no `+` prefix are auto-prepended with `+` and a `BARE_BODY_AUTO_PIPED_WARNING` is appended.
- `*** Begin Patch` / `*** End Patch` envelopes are silently consumed. `*** Abort` terminates parsing silently — ops parsed before the marker still apply, no warning surfaced.
- `*** Update File:` / `*** Add File:` / `*** Delete File:` / `*** Move to:` apply_patch sentinels throw an `apply_patch sentinel … is not valid in hashline` error.
- `@@`-bracketed hunk headers (whether the apply_patch `@@ context @@` form or the unified-diff `@@ -N,M +N,M @@` shape) are rejected with an explicit "drop the `@@ ... @@` brackets" message — hashline hunks are bare `A B` lines.
- Some malformed `¶` headers are recovered after stripping apply-patch path noise such as `Update File:` / `Add File:` and extra `***`, but the recovered header still needs a valid four-hex tag for the patcher to apply it.
- `*** Update File:` / `*** Add File:` / `*** Delete File:` / `*** Move to:` apply_patch sentinels inside the diff body throw an `apply_patch sentinel … is not valid in hashline` error.
- `@@`-bracketed hunk headers are rejected with guidance to write a verb header.
- Bare `N` and bare `N M` / `N..M` headers are rejected with guidance to write `replace` or `delete`.
- `delete N..M:` and any body rows under `delete` / `delete block` are rejected.
- Empty `replace` / `insert` / `replace block` hunks are rejected.
- `-` body rows are rejected with `MINUS_ROW_REJECTED`.
- `replace block N:` / `delete block N` require a wired tree-sitter resolver; `replace block` additionally needs at least one `+TEXT` body row, while `delete block` takes none. An unresolvable block (unsupported language, blank/closing-delimiter line, no node beginning on N, or a syntax error in the resolved block) is rejected on the apply/final-preview path; the streaming preview silently drops it instead.
## Outputs
- Single-shot tool result; hashline mode does not use a `resolve` preview/apply handshake.
@@ -80,7 +90,7 @@ Warnings:
Reference file (the exact shape `read` returns):
```text
¶a.ts#0A3
¶a.ts#0A3B
1:const X = "a";
2:const Y = X;
3:
@@ -92,91 +102,96 @@ Reference file (the exact shape `read` returns):
Replace line 1 with two lines:
```text
¶a.ts#0A3
1
¶a.ts#0A3B
replace 1..1:
+const X = "b";
+export const Y = X;
```
Insert BELOW line 5 (keep line 5, add after):
Insert below line 5:
```text
¶a.ts#0A3
5
&5
¶a.ts#0A3B
insert after 5:
+console.log(X + Y);
```
Insert ABOVE line 5 (add before, keep line 5):
Insert above line 5:
```text
¶a.ts#0A3
5
¶a.ts#0A3B
insert before 5:
+console.log(X + Y);
&5
```
Delete lines 4..5 entirely:
```text
¶a.ts#0A3
4 5
¶a.ts#0A3B
delete 4..5
```
Insert at start and end of file:
```text
¶a.ts#0A3
BOF
¶a.ts#0A3B
insert head:
+// header
EOF
insert tail:
+// trailer
```
Multi-file:
```text
¶src/a.ts#0A3
4
¶src/a.ts#0A3B
replace 4..4:
+const enabled = true;
¶src/b.ts#1F7
20
¶src/b.ts#1F7C
delete 20
```
## Limits & Caps
- File snapshot tags are exactly three uppercase-hex chars minted by the per-session snapshot store.
- File snapshot tags are exactly four uppercase-hex chars minted by the per-session snapshot store.
- The visible mismatch report shows 2 lines of context on each side (`MISMATCH_CONTEXT`) in `packages/hashline/src/messages.ts`.
- Stale-anchor recovery uses `fuzzFactor: 0` in `packages/hashline/src/recovery.ts`.
- `HL_FILE_PREFIX` is `¶`, `HL_PAYLOAD_REPLACE` is `+`, `HL_PAYLOAD_REPEAT` is `&`, `HL_RANGE_SEP` is `..` (repeat-row bodies only), and `HL_FILE_HASH_SEP` is `#` (`packages/hashline/src/format.ts`). Hunk headers carry no sigil; the range is just two whitespace-separated line numbers.
- `HL_FILE_PREFIX` is `¶`, `HL_PAYLOAD_REPLACE` is `+`, `HL_RANGE_SEP` is `..`, `HL_FILE_HASH_SEP` is `#`, and hunk keyword constants are `replace` / `delete` / `insert` (`packages/hashline/src/format.ts`).
## Errors
- Missing section header:
- `input must begin with "¶PATH#HASH" on the first non-blank line for anchored edits; got: ...`
- Missing tag for anchored edit:
- Missing tag for any section:
- `Missing hashline snapshot tag for anchored edit to <path>; use ¶<path>#tag from your latest read/search output.`
- Stray payload line:
- `line N: payload line has no preceding hunk header. Use an \`A B\` (or \`BOF\` / \`EOF\`) line above the body. Got "...".`
- Raw body row with no `+` / `&` prefix in a mixed-prefix block:
- `line N: payload row in a hashline hunk must start with + or &A..B. Got "...".`
- `line N: payload line has no preceding hunk header. Use \`replace N..M:\`, \`delete N..M\`, or \`insert before|after|head|tail:\` above the body. Got "...".`
- Minus row:
- ``line N: `-` rows are not valid; hashline ranges already name the lines being changed. To insert a literal line starting with `-`, write `+-…`.``
- Empty body-bearing hunk:
- `line N: \`replace N..M:\` needs at least one \`+TEXT\` body row. To delete lines, use \`delete N..M\`.`
- `line N: \`insert\` needs at least one \`+TEXT\` body row.`
- `line N: \`replace block N:\` needs at least one \`+TEXT\` body row. To delete a block, use \`delete N..M\` with the block's line range.`
- Unresolvable `replace block N:` (apply / final-preview path only):
- `line N: \`replace block X:\` could not resolve a syntactic block beginning on line X. The language may be unsupported, the line may be blank or a closing delimiter, or the block may not parse. Use \`replace X..M:\` with the block's explicit end line instead.`
- Delete with body:
- `line N: \`delete N..M\` does not take body rows. Remove the body, or use \`replace N..M:\`.`
- `line N: \`delete block N\` does not take body rows. Remove the body, or use \`replace block N:\` to replace the block.`
- Range out of order:
- `line N: range A..B ends before it starts.`
- Overlapping hunks on the same anchor:
- `line N: anchor line X is already targeted by another hunk on line Y. Issue ONE hunk per range; payload is only the final desired content, never a before/after pair.`
- apply_patch / unified-diff contamination:
- `line N: apply_patch sentinel "*** …" is not valid in hashline. File sections start with \`¶path#HASH\` (no \`Update File:\` / \`Add File:\` keyword). Hunks are bare \`A B\` lines with \`+TEXT\` / \`&A..B\` body rows.`
- `line N: unified-diff hunk header (\`@@ -N,M +N,M @@\`) is not valid in hashline. Hashline hunks are bare \`A B\` lines (or \`BOF\` / \`EOF\` keywords).`
- `line N: \`@@\`-bracketed hunk header "@@ …" is not valid in hashline. Drop the \`@@ ... @@\` brackets and write the range directly: \`5 7\` (\`BOF\` / \`EOF\` for virtual positions).`
- `line N: single-number hunk header "N" is no longer accepted. Spell single-line ranges as \`N N\` (two numbers); hashline hunks are bare \`A B\` lines (or \`BOF\` / \`EOF\`).`
- `line N: apply_patch sentinel "*** …" is not valid in hashline. File sections start with \`¶path#HASH\` (no \`Update File:\` / \`Add File:\` keyword). Use \`replace N..M:\`, \`delete N..M\`, or \`insert before|after|head|tail:\` ops.`
- `line N: unified-diff hunk header (\`@@ -N,M +N,M @@\`) is not valid in hashline. Use \`replace N..M:\`, \`delete N..M\`, or \`insert before|after|head|tail:\` ops.`
- `line N: \`@@\`-bracketed hunk header "@@ …" is not valid in hashline. Drop the \`@@ ... @@\` brackets and write a verb header such as \`replace N..M:\`.`
- `line N: hunk headers need a verb. Use \`replace N..N:\` to replace, or \`delete N\` to delete.`
- `line N: bare range hunk header "N M" is not valid. Hunk headers need a verb: write \`replace N..M:\` or \`delete N..M\`.`
- Out-of-range anchor:
- `Line N does not exist (file has M lines)`
- Stale snapshot tag throws `MismatchError`. The error contains re-read guidance and nearby current file lines as `*LINE:TEXT` / ` LINE:TEXT`.
- Stale snapshot tag: the `Patcher` first attempts snapshot-based recovery. When recovery cannot prove a valid result it throws `MismatchError`, which distinguishes recognized-but-drifted hashes from never-recorded hashes. The error includes the current file hash plus context around each anchor.
- No-op edit:
- `Edits to <path> parsed and applied cleanly, but produced no change: your body row(s) are byte-identical to the file at the targeted lines. The bug is somewhere else — re-read the file before issuing another edit. Do NOT widen the payload or add lines; verify the anchor first.`
- Recovery failure is silent internally: if cache-based merge cannot prove a valid result, the mismatch error is surfaced unchanged.
## Warnings
- `Detected two identical-range hashline hunks; kept only the second hunk. …` (`REPLACE_PAIR_COALESCED_WARNING`)
- `Detected an overlapping bare hashline hunk immediately followed by a concrete hunk; dropped the earlier bare hunk. …` (`REPLACE_PAIR_COALESCED_OVERLAP_WARNING`)
- `Auto-prefixed bare body row(s) with +. Always start payload rows with +TEXT (literal) or &A..B (repeat) …` (`BARE_BODY_AUTO_PIPED_WARNING`)
- `A body row started with `+&A..B`. `+` (literal text) and `&A..B` (repeat) are sibling row kinds …` (`PLUS_PREFIXED_REPEAT_WARNING`)
- `Auto-prefixed bare body row(s) with +. Body rows must be +TEXT literal lines …` (`BARE_BODY_AUTO_PIPED_WARNING`)
- Recovery banners: `RECOVERY_EXTERNAL_WARNING`, `RECOVERY_SESSION_CHAIN_WARNING`, `RECOVERY_SESSION_REPLAY_WARNING` (`packages/hashline/src/messages.ts`).
+59 -15
View File
@@ -9,10 +9,11 @@
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/eval.md`
- Key collaborators:
- `packages/coding-agent/src/eval/backend.ts` — backend execution contract
- `packages/coding-agent/src/eval/js/index.ts` — JS backend adapter
- `packages/coding-agent/src/eval/js/executor.ts` — JS execution + output sink
- `packages/coding-agent/src/eval/js/context-manager.ts` — persistent VM contexts, prelude, tool bridge
- `packages/coding-agent/src/eval/js/prelude.txt` — JS global helpers
- `packages/coding-agent/src/eval/agent-bridge.ts` — host-side `agent()` bridge into the subagent executor
- `packages/coding-agent/src/eval/js/executor.ts` — JS backend adapter
- `packages/coding-agent/src/eval/js/worker-core.ts` — JS execution, VM context, display/log capture
- `packages/coding-agent/src/eval/js/shared/prelude.txt` — JS global helper installer
- `packages/coding-agent/src/eval/js/shared/helpers.ts` — JS filesystem/text/env helper implementations
- `packages/coding-agent/src/eval/py/index.ts` — Python backend adapter
- `packages/coding-agent/src/eval/py/executor.ts` — kernel session retention, reset, cleanup
- `packages/coding-agent/src/eval/py/kernel.ts` — Jupyter gateway/kernel protocol, display capture
@@ -56,13 +57,12 @@ Final result from `EvalTool.execute()` is single-shot, but `onUpdate` streams pa
Returned shape:
- `content`: one text block containing combined cell output, or `(no text output)` / `(no output)` when only rich outputs exist.
- `content`: one text block containing combined cell output, `(displayed N image(s); no text output)` when only images exist, or `(no output)` when nothing visible was produced; image outputs are appended as additional image content blocks.
- `details` (`EvalToolDetails` from `packages/coding-agent/src/eval/types.ts`):
- `cells`: per-cell code, status (`pending`/`running`/`complete`/`error`), output, duration, exit code, status events, markdown flag
- `language`: first backend used
- `languages`: distinct backends used, in first-use order
- `jsonOutputs`: structured values emitted via `display(...)`
- `images`: image payloads emitted by Python rich display or JS `display({ type: "image", ... })`
- `statusEvents`: aggregated helper/tool status events
- `notice`: backend fallback notice (currently unused; reserved for future per-cell notices)
- `meta`: truncation metadata
@@ -75,7 +75,7 @@ Renderer behavior in `packages/coding-agent/src/tools/eval.ts`:
- markdown outputs are rendered with the Markdown component instead of plain text
- `jsonOutputs` render as a tree, collapsed or expanded depending on UI state
- timeout / truncation notices render as dim metadata lines
- images are carried in `details.images`; generic tool UI image handling renders them outside the text block
- images are returned as content image blocks; live updates may also carry `details.images` while execution is in progress
Side-channel artifacts:
@@ -92,7 +92,7 @@ Side-channel artifacts:
3. The tool allocates an `OutputSink`, a `TailBuffer`, per-cell result objects, and a `sessionAbortController`. `session.trackEvalExecution?.(...)` can wrap the whole run for external cancellation tracking.
4. It resolves the executor session id from `session.getEvalSessionId?.()`, falling back to `defaultEvalSessionId(session)`. Subagents inherit the parent's id so both sides share the same JS VM and Python kernel for each backend.
5. Cells execute sequentially within one eval tool call. For each cell, `execute()`:
- clamps `(cell.timeout ?? 30) * 1000` ms through `clampTimeout("eval", ...)`
- clamps `cell.timeout ?? 30` seconds through `clampTimeout("eval", ...)`
- builds a combined abort signal from the tool signal, the timeout, and the session abort controller
- marks the cell `running` and emits an update
- calls the backend’s `execute()` with `cwd`, `sessionId`, `sessionFile`, `kernelOwnerId`, `deadlineMs`, `reset` (defaults to `false`), artifact info, and chunk callback
@@ -120,7 +120,7 @@ If the requested backend is disabled or unavailable, the tool throws `ToolError`
### JavaScript runtime
Implemented in `packages/coding-agent/src/eval/js/context-manager.ts` and `packages/coding-agent/src/eval/js/prelude.txt`.
Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/coding-agent/src/eval/js/shared/prelude.txt`, and `packages/coding-agent/src/eval/js/shared/helpers.ts`.
- Persistent worker-backed VM sessions keyed by `js:${sessionId}`
- `reset: true` calls `resetVmContext(sessionKey)` before the cell executes; reset is destructive for all live runs on that JS session
@@ -131,7 +131,16 @@ Implemented in `packages/coding-agent/src/eval/js/context-manager.ts` and `packa
- `display`, `print`
- `read`, `write`, `append`, `sort`, `uniq`, `counter`, `diff`, `tree`, `env`, `output`
- `tool.<name>(args)` proxy for arbitrary session tool calls
- JS helpers are async because they cross the VM/tool boundary
- `llm(prompt, opts?)` for oneshot, stateless LLM calls (see _Oneshot LLM helper_ below)
- `agent(prompt, opts?)` for a single subagent call, plus `parallel()` / `pipeline()` bounded-pool helpers (see _Subagent helper_ below)
- JS helpers that touch the host/runtime boundary are async and `await`able; pure text helpers (`sort`, `uniq`, `counter`) return synchronously but may still be safely awaited.
- JS helper signatures use a trailing options object rather than Python keyword arguments:
- `await read(path, { offset?, limit? })`
- `await tree(path = ".", { maxDepth?, hidden? })`
- `sort(text, { reverse?, unique? })`, `uniq(text, { count? })`, `counter(items, { limit?, reverse? })`
- `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
- `await parallel([() => agent("a"), () => agent("b")])`
- `await pipeline(items, stage1, stage2)`
- `display(value)` behavior:
- plain objects/arrays become JSON outputs
- `{ type: "image", data, mimeType }` becomes an image output
@@ -152,7 +161,7 @@ Implemented in `packages/coding-agent/src/eval/py/executor.ts`, `packages/coding
- initialize cwd / env / `sys.path`
- execute `PYTHON_PRELUDE`
- Python cells run in the runner's persistent asyncio event loop, so top-level `await` works; the prompt warns not to use `asyncio.run(...)`
- The Python prelude defines helpers with the same surface as JS where practical, including `tool.<name>(args)` through a per-run loopback bridge
- The Python prelude defines helpers with the same surface as JS where practical, including `tool.<name>(args)`, `llm(...)`, and `agent(...)` through a per-run loopback bridge
- Synchronous statement blocks run in the default executor with ContextVar state copied in; the GIL still serializes bytecode execution, but awaited regions can interleave with sibling cells
- Kernel `display_data` / `execute_result` messages map to:
- `application/x-omp-status` → status event
@@ -163,6 +172,36 @@ Implemented in `packages/coding-agent/src/eval/py/executor.ts`, `packages/coding
- `text/html` → HTML converted to markdown with `htmlToBasicMarkdown()`
- Interactive stdin is rejected: `input_request` sends an empty reply, marks `stdinRequested`, and the executor returns exit code `1`
### Oneshot LLM helper (`llm`)
Both runtimes expose `llm()` — a single stateless completion against a model tier. It is intentionally minimal: no conversation history, no agent-visible tools, pure text in / text (or object) out. Implemented host-side in `packages/coding-agent/src/eval/llm-bridge.ts` and routed through the existing tool bridge under the reserved name `__llm__`.
- Signatures:
- JS: `await llm(prompt, { model?, system?, schema? })`
- Python: `llm(prompt, *, model="default", system=None, schema=None)`
- `model` selects a tier (default `"default"`):
- `"smol"` → `pi/smol` role (fast / cheap)
- `"default"` → the session's active model, falling back to the `pi/default` role
- `"slow"` → `pi/slow` role; requests high reasoning effort only on reasoning-capable models
- `system` (optional) supplies a system prompt.
- `schema` (optional) is a plain JSON-Schema object. When present, the model is forced to call a single synthetic `respond` tool with that schema (loose, non-strict), and the helper returns the parsed object. When absent, the helper returns the completion string.
- Errors surface as exceptions: unresolved tier, missing API key, an `error`/`aborted` stop reason, or empty output each raise.
### Subagent helper (`agent`)
Both runtimes expose `agent()` — a single subagent invocation routed through `packages/coding-agent/src/eval/agent-bridge.ts` into the same `runSubprocess(...)` path used by the `task` tool. It uses the current eval session's spawn policy and inherits the parent eval executor id, so parent and subagent code share JS/Python runtime state.
- Signatures:
- JS: `await agent(prompt, { agentType?, model?, context?, label?, schema? })`
- Python: `agent(prompt, *, agent_type="task", model=None, context=None, label=None, schema=None)`
- `agentType` / `agent_type` defaults to the bundled `task` agent and resolves through normal agent discovery, so project and user agents work.
- `model` overrides the selected agent's model. Without it, normal per-agent settings and the agent frontmatter model apply.
- `context` supplies shared background; `label` controls the `agent://<id>` output label prefix.
- `schema` passes a JSON Schema to the subagent structured-output path. When present, the helper parses the final JSON text and returns an object.
- Spawn restrictions use `session.getSessionSpawns()` exactly like the `task` tool. Eval-driven subagent recursion is capped at depth 3.
- JS and Python both expose `parallel(thunks)` and `pipeline(items, ...stages)`; both use a bounded async/threaded pool whose width tracks the `task.maxConcurrency` setting (the same ceiling the `task` tool uses; `0` = run every item at once), preserve item order, and propagate rejections. The width is fetched live from the host via the `__concurrency__` bridge, so the helpers no longer take a `concurrency` argument.
- Errors surface as exceptions: unknown or disabled agent, disallowed spawn, recursion cap, subagent failure, or invalid structured output all fail the eval cell.
### Multi-language call behavior
A single tool call can mix Python and JS cells. Persistence is per language runtime:
@@ -174,7 +213,8 @@ A single tool call can mix Python and JS cells. Persistence is per language runt
## Side Effects
- Filesystem
- JS/Python prelude helpers can read, write, append, diff, and traverse files under the session cwd or absolute paths.
- JS/Python prelude helpers can read, write, append, diff, and traverse filesystem paths under the session cwd or absolute paths.
- JS helper `read()` rejects protocol URIs (`://`) and directory paths; use `tool.read(...)` for internal URLs or reader-mode behavior.
- Output may spill to an artifact file via `OutputSink`.
- Network
- Python backend speaks NDJSON to a local `python3` subprocess over stdin/stdout (no network).
@@ -182,12 +222,14 @@ A single tool call can mix Python and JS cells. Persistence is per language runt
- Subprocesses / native bindings
- Python availability check runs `<python> -c ...`.
- Python backend spawns one `python -u runner.py` subprocess per kernel; cancellation sends `SIGINT`. Details in `docs/python-repl.md`.
- `agent()` runs one in-process subagent via the task executor; that subagent may use its configured tools.
- Session state
- `session.assertEvalExecutionAllowed?.()` can block execution.
- `session.trackEvalExecution?.(...)` can register cancellable eval work.
- `session.getSessionFile?.()`, `session.getEvalSessionId?.()`, and `session.getEvalKernelOwnerId?.()` influence VM/kernel reuse and artifact lookup.
- JS VM contexts persist across eval calls until reset/disposal.
- Python retained kernels persist until reset, owner cleanup, or process exit.
- `agent()` allocates `agent://<id>` output artifacts and reuses the parent's eval executor id.
- User-visible prompts / interactive UI
- none; stdin requests are rejected programmatically
- Background work / cancellation
@@ -203,6 +245,8 @@ A single tool call can mix Python and JS cells. Persistence is per language runt
- Output truncation window: 50KB default (`DEFAULT_MAX_BYTES` in `packages/coding-agent/src/session/streaming-output.ts`)
- Output line cap inside truncation helpers: 3000 lines (`DEFAULT_MAX_LINES` in `packages/coding-agent/src/session/streaming-output.ts`)
- Streaming tail buffer for live updates: `DEFAULT_MAX_BYTES * 2` = 100KB (`packages/coding-agent/src/tools/eval.ts`)
- JS/Python `parallel()` / `pipeline()` helper pool width: the `task.maxConcurrency` setting (default 32; `0` = unbounded), resolved live via the `__concurrency__` bridge (`packages/coding-agent/src/eval/concurrency-bridge.ts`)
- Eval-driven `agent()` recursion cap: task depth 3 (`EVAL_AGENT_MAX_DEPTH`)
- Python retained kernel idle timeout: 5 minutes (`IDLE_TIMEOUT_MS` in `packages/coding-agent/src/eval/py/executor.ts`)
- Python retained kernel cap: 4 sessions (`MAX_KERNEL_SESSIONS` in `packages/coding-agent/src/eval/py/executor.ts`)
- Python retained kernel cleanup sweep: every 30s (`CLEANUP_INTERVAL_MS` in `packages/coding-agent/src/eval/py/executor.ts`)
@@ -234,11 +278,11 @@ A single tool call can mix Python and JS cells. Persistence is per language runt
## Notes
- Backend selection is now strictly explicit per cell: `language` must be `"py"` or `"js"`. The previous `*** Cell` header parser, the `eval.lark` constrained grammar, and the sniffer-based fallback have all been removed.
- Backend selection is strictly explicit per cell: `language` must be `"py"` or `"js"`. The previous `*** Cell` header parser, the `eval.lark` constrained grammar, and the sniffer-based fallback have all been removed.
- `EvalTool.customFormat` no longer exists. Tool calls flow through the standard JSON schema; there is no Lark-constrained sampling path.
- `tool.<name>()` exists in both JS and Python. Python calls route through a per-run loopback bridge keyed by the current cell id.
- JS helper paths reject protocol URIs (`://`) in `resolvePath()`; the JS prelude is filesystem-only unless the code calls `tool.read(...)` or another tool explicitly.
- Python helper `output(...)` depends on `PI_SESSION_FILE`; it fails outside a session-backed run.
- JS helper paths reject protocol URIs (`://`) in `resolveRegularFile()` for `read()`, and resolve other paths against the session cwd or absolute filesystem path. Use `tool.read(...)` or another tool explicitly for internal URLs.
- Python helper `output(...)` depends on `PI_ARTIFACTS_DIR` or `PI_SESSION_FILE`; it fails outside a session-backed run.
- `display()` can produce text and structured outputs from the same value; the renderer prefers markdown over `text/plain` when both exist.
- JS static imports are rewritten only at top level. Nested imports stay invalid and surface normal JS syntax/runtime errors.
- `EvalTool` is `concurrency = "exclusive"` within one agent session, but parent and subagent sessions can run eval concurrently when they share an inherited executor id.
+29 -23
View File
@@ -18,15 +18,17 @@
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `paths` | `string[]` | Yes | One or more globs, files, or directories. Empty strings are rejected. Multiple entries may be merged into one brace-union search when their base paths can be resolved together. |
| `paths` | `string[]` | Yes | One or more globs, files, directories, or internal URLs with backing files. Empty strings are rejected. Single entries accidentally joined with comma, semicolon, or whitespace are expanded only after existence validation; existing paths containing delimiters stay intact. Multiple entries may be merged into one brace-union search when their base paths can be resolved together. |
| `hidden` | `boolean` | No | Whether hidden files are included. Defaults to `true` (`hidden ?? true`). |
| `limit` | `number` | No | Max returned paths. Defaults to `1000`. Must be a finite positive number; non-integers are floored. |
| `gitignore` | `boolean` | No | Whether `.gitignore` is respected during local native globbing. Defaults to `true`; set `false` to include gitignored files. |
| `limit` | `number` | No | Max returned paths. Defaults to `200`; finite positive inputs are floored then clamped to `1..200`. |
| `timeout` | `number` | No | Timeout in seconds. Defaults to `5`; clamped to `0.5..60`. On timeout, returns partial matches collected so far with a timeout notice and `truncated: true`. |
## Outputs
The tool returns a single text block plus structured `details`.
- Success text: newline-delimited paths, one per line, relative to the session cwd when possible; absolute when outside cwd. Exact file inputs return that file path as one line.
- Empty result text: `No files found matching pattern`.
- Success text: matching paths grouped by directory. Each non-root group starts with `# <dir>/` and then lists basenames; root-level matches are listed without a header. Directory matches carry a trailing `/`. Exact file inputs return that file path as one line.
- Empty result text: `No files found matching pattern`, optionally followed by a timeout or missing-path notice.
- Multi-path partial miss: appends `Skipped missing paths: ...` after the result block, or after the empty-result line.
- `details` may include:
- `scopePath`: display form of the searched root or merged roots.
@@ -36,25 +38,27 @@ The tool returns a single text block plus structured `details`.
- `resultLimitReached`: reached result limit.
- `missingPaths`: skipped missing inputs in multi-path calls.
- `truncation` / `meta.limits`: structured truncation and limit metadata for renderers.
- Streaming: when the runtime supplies `onUpdate`, the local implementation emits incremental newline-delimited text snapshots during globbing, throttled to 200 ms.
- Streaming: when the runtime supplies `onUpdate`, the local implementation emits incremental newline-delimited text snapshots during globbing, throttled to 200 ms. Final output is grouped; streaming snapshots are not.
## Flow
1. `FindTool.execute()` normalizes each `paths` entry with `normalizePathLikeInput()` and `/\\/g -> "/"` (`packages/coding-agent/src/tools/find.ts`). Empty normalized entries fail with `` `paths` must contain non-empty globs or paths ``.
2. For multi-path local calls, `partitionExistingPaths(..., parseFindPattern)` (`packages/coding-agent/src/tools/path-utils.ts`) stats each base path. Missing entries are skipped; if all are missing, the tool throws `Path not found: ...`. Single missing paths still hard-fail.
3. The tool tries `resolveExplicitFindPatterns()` to merge multiple inputs into one search rooted at a common base path. If that does not apply, it parses one input with `parseFindPattern()`.
4. `parseFindPattern()` determines `(basePath, globPattern, hasGlob)`:
1. `FindTool.execute()` expands delimiter-flattened local `paths` entries with `expandDelimitedPathEntries(..., parseFindPattern)` unless custom operations are injected. The splitter validates candidate parts by statting their parsed base paths, keeps existing delimiter-containing paths intact, accepts comma/semicolon splits when at least one part resolves, and accepts whitespace splits only when every part resolves.
2. The tool normalizes each resulting entry with `normalizePathLikeInput()` and `/\\/g -> "/"` (`packages/coding-agent/src/tools/find.ts`). Empty normalized entries fail with `` `paths` must contain non-empty globs or paths ``.
3. For multi-path local calls, `partitionExistingPaths(..., parseFindPattern)` (`packages/coding-agent/src/tools/path-utils.ts`) stats each base path. Missing entries are skipped; if all are missing, the tool throws `Path not found: ...`. Single missing paths still hard-fail.
4. The tool tries `resolveExplicitFindPatterns()` to merge multiple inputs into one search rooted at a common base path. If that does not apply, it parses one input with `parseFindPattern()`.
5. `parseFindPattern()` determines `(basePath, globPattern, hasGlob)`:
- no glob chars (`*`, `?`, `[`, `{`) => search that path with implicit `**/*`.
- glob in the first segment => search from `.` and, unless the pattern already starts with `**/`, prefix it with `**/`.
- glob later in the path => split at the first glob-bearing segment.
5. `resolveToCwd()` converts the base path to an absolute path under the session cwd. A resolved `/` is rejected with `Searching from root directory '/' is not allowed`.
6. `limit` is defaulted to `DEFAULT_LIMIT` (`1000`) and validated as a positive finite integer. `hidden` defaults to `true`. The tool also creates a 5 s timeout via `AbortSignal.timeout(GLOB_TIMEOUT_MS)`.
7. Execution then branches:
6. `resolveToCwd()` converts the base path to an absolute path under the session cwd. A resolved `/` is rejected with `Searching from root directory '/' is not allowed`.
7. `limit` defaults to `DEFAULT_LIMIT` (`200`), must be positive and finite, is floored, then clamped to `MAX_LIMIT` (`200`). `hidden` and `gitignore` both default to `true`. `timeout` is converted to milliseconds and clamped to `500..60_000` before building an `AbortSignal.timeout(...)`.
8. Execution then branches:
- **Custom operations branch**: if `FindToolOptions.operations.glob` exists, the tool checks existence with `operations.exists()`, short-circuits exact-file inputs via `operations.stat()` when available, then calls `operations.glob(globPattern, searchPath, { ignore: ["**/node_modules/**", "**/.git/**"], limit })`.
- **Built-in local branch**: the tool stats `searchPath`. Exact-file inputs return immediately. Directory inputs call `natives.glob()` with `fileType: File`, `hidden`, `maxResults: limit`, `sortByMtime: true`, `gitignore: true`, and the combined abort signal.
8. In the local branch, optional `onMatch` callbacks convert each match to a cwd-relative display path and emit throttled progress updates.
9. After native glob returns, JS sorts `result.matches` by `mtime` descending (`(b.mtime ?? 0) - (a.mtime ?? 0)`) before formatting paths.
10. `buildResult()` applies `applyListLimit()` to cap the array again at `limit`, joins paths with `\n`, then runs `truncateHead()` with `maxLines: Number.MAX_SAFE_INTEGER`. In practice this leaves the 50 KB byte cap in place while disabling the default 3000-line cap.
11. `toolResult()` packages text plus `details`, and records result-limit / truncation metadata for renderers.
- **Built-in local branch**: the tool stats `searchPath`. Exact-file inputs return immediately. Directory inputs call `natives.glob()` with `hidden`, `maxResults: effectiveLimit`, `sortByMtime: true`, `gitignore: useGitignore`, and the combined abort signal.
9. In the local branch, optional `onMatch` callbacks convert each match to a cwd-relative display path and emit throttled progress updates.
10. After native glob returns, JS sorts `result.matches` by `mtime` descending (`(b.mtime ?? 0) - (a.mtime ?? 0)`) before formatting paths.
11. `buildResult()` applies `applyListLimit()` to cap the array again at `effectiveLimit`, formats paths with `formatFindGroupedOutput()`, appends notices, then runs `truncateHead()` with `maxLines: Number.MAX_SAFE_INTEGER`. In practice this leaves the 50 KB byte cap in place while disabling the default 3000-line cap.
12. `toolResult()` packages text plus `details`, and records result-limit / truncation metadata for renderers.
## Modes / Variants
- **Exact file path**: if the parsed input has no glob and the resolved path stats as a file, output is that one path.
@@ -62,6 +66,7 @@ The tool returns a single text block plus structured `details`.
- **Single glob path**: one input parsed by `parseFindPattern()`.
- **Merged multi-path search**: multiple inputs resolved by `resolveExplicitFindPatterns()` into one brace-union glob rooted at a common base path.
- **Partial multi-path search with missing inputs**: local multi-path calls skip missing base paths and surface them as `missingPaths` / `Skipped missing paths: ...`.
- **Internal URL input**: supported when the internal router resolves the URL to a backing file. Internal URL globs are rejected.
- **Custom delegated search**: uses injected `FindOperations` instead of local fs + native glob.
## Side Effects
@@ -74,11 +79,12 @@ The tool returns a single text block plus structured `details`.
- Emits structured progress updates when `onUpdate` is provided.
- Adds truncation / limit metadata to the tool result.
- Background work / cancellation
- Local globbing is cancellable through the caller abort signal plus an internal 5 s timeout.
- Local globbing is cancellable through the caller abort signal plus the configured internal timeout.
## Limits & Caps
- Default result limit: `1000` (`DEFAULT_LIMIT` in `packages/coding-agent/src/tools/find.ts`).
- Local glob timeout: `5000` ms (`GLOB_TIMEOUT_MS` in `packages/coding-agent/src/tools/find.ts`).
- Default result limit: `200` (`DEFAULT_LIMIT` in `packages/coding-agent/src/tools/find.ts`).
- Maximum result limit: `200` (`MAX_LIMIT`); larger inputs are clamped.
- Local glob timeout: default `5000` ms, clamped to `500..60_000` ms.
- Output byte cap: `50 * 1024` bytes (`DEFAULT_MAX_BYTES` in `packages/coding-agent/src/session/streaming-output.ts`).
- Default generic line cap in `truncateHead()` is `3000`, but `find` overrides `maxLines` to `Number.MAX_SAFE_INTEGER`, so byte size — not line count — is the practical output truncation cap.
- Streaming update throttle: `200` ms between `onUpdate` emissions.
@@ -91,7 +97,7 @@ The tool returns a single text block plus structured `details`.
- `Searching from root directory '/' is not allowed`
- `Limit must be a positive number`
- `Path is not a directory: ...`
- `find timed out after 5s`
- timeout result text is `find timed out after <seconds>s; returning <N> partial matches — increase timeout or narrow pattern` and is returned as a successful, truncated partial result rather than an error.
- If the caller aborts, the local branch converts `AbortError` into `ToolAbortError`.
- Non-`ENOENT` stat failures and other unexpected errors are rethrown.
- Empty matches are not errors; they return the no-files text result.
@@ -99,8 +105,8 @@ The tool returns a single text block plus structured `details`.
## Notes
- Reach for `find` for filename / path discovery. Reach for `search` when the selection criterion is file contents or regex matches; `search` takes a `pattern` and returns anchored content matches, while `find` only returns matching paths (`packages/coding-agent/src/prompts/tools/find.md`, `packages/coding-agent/src/prompts/tools/search.md`).
- Bare top-level globs are made recursive. `*.ts` is parsed as base `.` plus glob `**/*.ts`; `src/*.ts` stays rooted at `src` with a non-recursive `*.ts` segment; `src/**/*.ts` preserves explicit recursion.
- `.gitignore` is always enabled in the built-in local branch (`gitignore: true`). There is no model-facing flag to disable it.
- `.gitignore` defaults to enabled in the built-in local branch. Use `gitignore: false` to disable it for native traversal.
- `hidden` defaults to `true`; hidden-file exclusion is opt-out, not opt-in.
- Multi-path missing-input tolerance only applies in the built-in local branch. The custom-operations branch hard-fails the first missing `searchPath` it checks.
- The custom `FindOperations.glob()` hook receives `ignore` and `limit`, but not the `hidden` flag or an explicit `.gitignore` toggle. A remote delegate must account for that itself if it wants parity with the local branch.
- Built-in local globbing asks the native layer for `fileType: File`, so recursive directory searches yield files, not directories. Directory outputs are only possible through exact-path passthrough or custom delegates that return them.
- Built-in local globbing does not force `fileType: File`; it can return files and directories from native glob. Directory outputs also occur through exact-path passthrough or custom delegates that return them.
+27 -23
View File
@@ -32,7 +32,10 @@
| `reviewer` | `string[]` | No | Used only by `pr_create`; each entry becomes `--reviewer`. |
| `assignee` | `string[]` | No | Used only by `pr_create`; each entry becomes `--assignee`. |
| `label` | `string[]` | No | Used only by `pr_create`; each entry becomes `--label`. |
| `query` | `string` | No | Used by all `search_*` ops. Required there. |
| `query` | `string` | No | Used by all `search_*` ops. Required by local validation only for `search_code`; the other search ops compose it with optional date/repo/type qualifiers and send the result to GitHub. |
| `since` | `string` | No | Lower date bound for `search_issues`, `search_prs`, `search_commits`, and `search_repos`. Accepts relative durations (`3d`, `12h`, `2w`, `2mo`, `1y`), `YYYY-MM-DD`, or an ISO datetime. Rejected for `search_code`. |
| `until` | `string` | No | Upper date bound for `search_issues`, `search_prs`, `search_commits`, and `search_repos`. Same formats as `since`. Rejected for `search_code`. |
| `dateField` | `"created" \| "updated"` | No | Date qualifier field for issue/PR/repo search. Defaults to `created`; repo search maps `updated` to GitHub's `pushed:` qualifier. Ignored for commit search, which always uses `committer-date:`. |
| `limit` | `number` | No | Used by all `search_*` ops. Defaults to `10`, floored, clamped to `50`, and must be `> 0`. |
| `run` | `string` | No | Used only by `run_watch`. Must be a numeric run ID or full GitHub Actions run URL. |
| `tail` | `number` | No | Used only by `run_watch`. Defaults to `15`, floored, clamped to `200`, and must be `> 0`. |
@@ -61,12 +64,12 @@ The tool returns a single text result built by `buildTextResult()` in `packages/
- maps common auth/repo-context failures into tool-facing `ToolError` messages;
- `json()` rejects empty or invalid JSON.
5. Read-style ops (`repo_view`, `search_*`) fetch JSON and format Markdown-like text summaries. Single-issue and single-PR views were moved out of the tool and now resolve through the `issue://` / `pr://` internal URL schemes, which share the same SQLite cache.
7. PR diffs moved out of the tool. `pr://<N>/diff` lists changed files, `pr://<N>/diff/<i>` slices a single file, and `pr://<N>/diff/all` returns the full unified diff — see `docs/tools/read.md`. All three variants share one `gh pr diff` invocation through the `pr-diff` cache row.
8. `pr_checkout` resolves PR metadata first, then enters `git.withRepoLock()` before any git mutation so parallel checkout calls for the same primary repo do not race on shared `.git` state.
9. `pr_push` reads PR head metadata back from git branch config, derives a refspec, then pushes with `git.push()`.
10. `pr_create` shells out once, then best-effort re-reads the created PR for a richer summary.
11. `run_watch` chooses either run mode (`run` supplied) or commit mode (`run` omitted), polls GitHub Actions APIs every 3 seconds, emits streaming updates, and may save a full failed-log artifact before returning.
12. Final text goes through `toolResult().text(...)`; if `session.allocateOutputArtifact()` returns a slot, failed-log text is persisted with `Bun.write()`.
6. PR diffs moved out of the tool. `pr://<N>/diff` lists changed files, `pr://<N>/diff/<i>` slices a single file, and `pr://<N>/diff/all` returns the full unified diff — see `docs/tools/read.md`. All three variants share one `gh pr diff` invocation through the `pr-diff` cache row.
7. `pr_checkout` resolves PR metadata first, then enters `git.withRepoLock()` before any git mutation so parallel checkout calls for the same primary repo do not race on shared `.git` state.
8. `pr_push` reads PR head metadata back from git branch config, derives a refspec, then pushes with `git.push()`.
9. `pr_create` shells out once, then best-effort re-reads the created PR for a richer summary.
10. `run_watch` chooses either run mode (`run` supplied) or commit mode (`run` omitted), polls GitHub Actions APIs every 3 seconds, emits streaming updates, and may save a full failed-log artifact before returning.
11. Final text goes through `toolResult().text(...)`; if `session.allocateOutputArtifact()` returns a slot, failed-log text is persisted with `Bun.write()`.
## Modes / Variants
@@ -142,21 +145,21 @@ Push target resolution reads the `branch.<name>.ompPrHeadRef`, `pushRemote`/`rem
| Aspect | Value |
| --- | --- |
| Required fields | `op`, `query` |
| Optional fields | `repo`, `limit` |
| `gh` command | `gh api -X GET /search/issues -f q="<query> [repo:<repo>] is:issue" -F per_page=<limit>` |
| Required fields | `op` |
| Optional fields | `repo`, `query`, `limit`, `since`, `until`, `dateField` |
| `gh` command | `gh api -X GET /search/issues -f q="<query> [date qualifier] [repo:<repo>] is:issue" -F per_page=<limit>` |
| Batching | None |
| Output | `# GitHub issues search`, echoed query, optional repo, result count, then one bullet per issue with repo/state/author/labels/timestamps/URL. |
`repo` defaults to the current checkout's `owner/repo` via `resolveSearchRepoScope()` when omitted. The default is suppressed when the query already contains a leading `repo:`/`org:`/`user:`/`owner:` qualifier or when `gh repo view` fails to resolve the current checkout (e.g. outside a github remote).
`repo` defaults to the current checkout's `owner/repo` via `resolveSearchRepoScope()` when omitted. The default is suppressed when the composed query already contains a leading `repo:`/`org:`/`user:`/`owner:` qualifier or when `gh repo view` fails to resolve the current checkout (e.g. outside a github remote).
### `search_prs`
| Aspect | Value |
| --- | --- |
| Required fields | `op`, `query` |
| Optional fields | `repo`, `limit` |
| `gh` command | `gh api -X GET /search/issues -f q="<query> [repo:<repo>] is:pr" -F per_page=<limit>` |
| Required fields | `op` |
| Optional fields | `repo`, `query`, `limit`, `since`, `until`, `dateField` |
| `gh` command | `gh api -X GET /search/issues -f q="<query> [date qualifier] [repo:<repo>] is:pr" -F per_page=<limit>` |
| Batching | None |
| Output | Same shape as `search_issues`, labeled as pull requests. |
@@ -172,15 +175,15 @@ Push target resolution reads the `branch.<name>.ompPrHeadRef`, `pushRemote`/`rem
| Batching | None |
| Output | `# GitHub code search`, result count, then one bullet per match with path, repo, short commit SHA, URL, and first normalized text-match fragment line when present. |
`repo` defaults to the current checkout's `owner/repo` as in `search_issues`.
`repo` defaults to the current checkout's `owner/repo` as in `search_issues`. `since` and `until` are explicitly rejected for this op because GitHub code search has no supported date qualifier.
### `search_commits`
| Aspect | Value |
| --- | --- |
| Required fields | `op`, `query` |
| Optional fields | `repo`, `limit` |
| `gh` command | `gh api -X GET /search/commits -f q="<query> [repo:<repo>]" -F per_page=<limit>` |
| Required fields | `op` |
| Optional fields | `repo`, `query`, `limit`, `since`, `until`, `dateField` (accepted but ignored; commit searches use `committer-date`) |
| `gh` command | `gh api -X GET /search/commits -f q="<query> [committer-date qualifier] [repo:<repo>]" -F per_page=<limit>` |
| Batching | None |
| Output | `# GitHub commits search`, result count, then one bullet per commit: short SHA + first commit-message line, repo, author, date, URL. |
@@ -190,13 +193,13 @@ Push target resolution reads the `branch.<name>.ompPrHeadRef`, `pushRemote`/`rem
| Aspect | Value |
| --- | --- |
| Required fields | `op`, `query` |
| Optional fields | `limit` |
| `gh` command | `gh api -X GET /search/repositories -f q="<query>" -F per_page=<limit>` |
| Required fields | `op` |
| Optional fields | `query`, `limit`, `since`, `until`, `dateField` |
| `gh` command | `gh api -X GET /search/repositories -f q="<query> [date qualifier]" -F per_page=<limit>` |
| Batching | None |
| Output | `# GitHub repositories search`, result count, then one bullet per repo with first description line, language, stars, forks, open issues, visibility, archive/fork flags, updated time, URL. |
`repo` is intentionally not used for this op.
`repo` is intentionally not used for this op. If `query`, `since`, and `until` are all omitted, the tool sends an empty GitHub repository-search query and the GitHub API may reject it.
### `run_watch`
@@ -260,8 +263,9 @@ Watch flow:
- otherwise stderr/stdout text, or fallback `GitHub CLI command failed: gh ...`
- `json()` also throws on empty stdout or invalid JSON.
- Local validation errors throw `ToolError`, including:
- missing required per-op fields (`query`, `title unless fill=true`)
- missing required per-op fields (`query` for `search_code`, `title unless fill=true`)
- invalid numeric `limit` / `tail`
- invalid `since` / `until` date bound
- invalid `run` format
- `fill` combined with `title` or `body`
- missing git repo / branch / HEAD context for checkout, push, or watch
+5 -4
View File
@@ -46,10 +46,10 @@ TUI rendering adds presentation-only truncation from `packages/coding-agent/src/
6. `readImageMetadata(...)` in `packages/utils/src/mime.ts` inspects file headers only. Supported detected MIME types are `image/png`, `image/jpeg`, `image/gif`, and `image/webp`.
7. If `images.autoResize` is true, `loadImageInput(...)` calls `resizeImage(...)`. Resize failures are swallowed there and the original bytes are kept.
8. If MIME detection returned no supported image type, `execute(...)` throws `ToolError("inspect_image only supports PNG, JPEG, GIF, and WEBP files detected by file content.")`.
9. The tool calls `completeSimple(...)` with one user message containing two content parts in order:
9. The tool calls `instrumentedCompleteSimple(...)` with one user message containing two content parts in order:
- `{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType }`
- `{ type: "text", text: params.question }`
10. `systemPrompt` is a one-element array rendered from `packages/coding-agent/src/prompts/tools/inspect-image-system.md`.
10. `systemPrompt` is a one-element array rendered from `packages/coding-agent/src/prompts/tools/inspect-image-system.md`; telemetry is tagged with oneshot kind `inspect_image`.
11. If the model response stop reason is `error` or `aborted`, the tool maps that to `ToolError`.
12. `extractResponseText(...)` concatenates only `text` content blocks from the assistant message, trims the result, and fails if nothing remains.
13. Success returns the text plus `details`; `inspectImageToolRenderer` formats the result for the TUI.
@@ -65,11 +65,11 @@ TUI rendering adds presentation-only truncation from `packages/coding-agent/src/
- Resolves and reads the target image from disk.
- Stats the file once with `Bun.file(...).stat()` and reads it fully with `fs.readFile(...)`.
- Network
- Sends the final base64 image payload plus question text to the selected model through `completeSimple(...)`.
- Sends the final base64 image payload plus question text to the selected model through `instrumentedCompleteSimple(...)` / the configured simple completion implementation.
- Session state
- Reads session settings, active model preferences, cwd, and model registry.
- Background work / cancellation
- Passes the caller `AbortSignal` into `completeSimple(...)`.
- Passes the caller `AbortSignal` into `instrumentedCompleteSimple(...)` and the configured simple completion implementation.
- Image preprocessing is local and not cancellation-aware in these helpers.
## Limits & Caps
@@ -112,6 +112,7 @@ TUI rendering adds presentation-only truncation from `packages/coding-agent/src/
Failures surface as thrown `ToolError`s from `execute(...)`; the normal success return shape is not used for error reporting.
## Notes
- The tool schema is not marked strict in `InspectImageTool`; callers should still treat only `path` and `question` as supported inputs because the implementation reads no other fields.
- The model-facing prompt path on disk is `packages/coding-agent/src/prompts/tools/inspect-image.md`; the assignment's underscore form does not exist.
- Format support is based on file content, not filename extension. Renaming a non-image file to `.png` does not make it valid.
- `resolveReadPath(...)` tries macOS-specific path variants: shell-unescaped spaces, AM/PM narrow no-break-space filenames, NFD normalization, and curly-quote variants.
+8 -6
View File
@@ -28,7 +28,7 @@
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `op` | `"send"` | Yes | Sends one message to one peer or to `"all"`. |
| `to` | `string` | Yes | Peer id such as `0-Main`, or `"all"` for broadcast. Whitespace is trimmed. |
| `to` | `string` | Yes | Peer id such as `Main`, or `"all"` for broadcast. Whitespace is trimmed. |
| `message` | `string` | Yes | Message body. Whitespace is trimmed; empty-after-trim is rejected. |
| `awaitReply` | `boolean` | No | Wait for prose replies. Defaults to `true` for direct messages and `false` for `to: "all"`. |
@@ -44,7 +44,7 @@
## Flow
1. `IrcTool.createIf` only constructs the tool when `irc.enabled` is on and the session has both an `AgentRegistry` and `getAgentId` (`packages/coding-agent/src/tools/irc.ts`).
2. Tool discovery adds another gate in `packages/coding-agent/src/tools/index.ts`: if the caller is `0-Main` and `async.enabled` is off, `irc` is hidden because the main agent cannot talk to concurrent peers in sync mode.
2. Tool discovery adds another gate in `packages/coding-agent/src/tools/index.ts`: if the caller is `Main` and `async.enabled` is off, `irc` is hidden because the main agent cannot talk to concurrent peers in sync mode.
3. `execute` resolves the process-global registry and sender id. Missing either returns a text error result instead of throwing.
4. `op: "list"` calls `registry.listVisibleTo(senderId)`, which exposes every other agent in flat namespace whose status is `running` or `idle` (`packages/coding-agent/src/registry/agent-registry.ts`).
5. `list` formats human-readable lines and returns `channels` as `['all', ...peerIds]`. These are logical targets only; there is no channel join state.
@@ -58,7 +58,8 @@
- queues just the incoming message for later history injection when `awaitReply === false`, or
- renders `packages/coding-agent/src/prompts/system/irc-incoming.md`, runs `runEphemeralTurn` with `toolChoice: "none"`, emits an auto-reply event, then queues both incoming and reply messages for history injection.
11. Deferred injection waits until the recipient is no longer streaming; `#flushPendingBackgroundExchanges` appends the custom messages through normal `message_start`/`message_end` external events so persistence and listeners see them.
12. `send` aggregates `delivered`, `replies`, `failed`, and `notFound`, then returns one text summary plus matching `details`.
12. Dispatch waits are bounded by `irc.timeoutMs` (default `120_000` ms). A value of `0` disables the local timeout; parent aborts still abort the dispatch.
13. `send` aggregates `delivered`, `replies`, `failed`, and `notFound`, then returns one text summary plus matching `details`.
## Modes / Variants
- `list`: enumerate visible peers and logical channels.
@@ -79,7 +80,7 @@
- Auto-replies are generated from `packages/coding-agent/src/prompts/system/irc-incoming.md` and explicitly forbid tool use.
- Background work / cancellation
- `send` starts one background `respondAsBackground` call per target.
- The caller's `AbortSignal` is forwarded into each background reply turn.
- The caller's `AbortSignal` is forwarded into each background reply turn. `irc.timeoutMs` creates a per-recipient `AbortController` and reports timeout failures per target.
- Network
- No IRC server connection.
- When `awaitReply: true`, the recipient may make model-provider API calls through `runEphemeralTurn`.
@@ -93,7 +94,8 @@
- Visibility scope: only peers in status `running` or `idle` are addressable via `listVisibleTo`.
- Reply execution:
- No tools are available in auto-reply turns (`toolChoice: "none"` in `runEphemeralTurn`).
- No internal timeout, retry, backoff, rate limit, or reply length cap is defined in `irc.ts`; behavior relies on the underlying model stream and any upstream API limits.
- `irc.timeoutMs` defaults to `120_000`; `0` disables the timeout, non-finite values fall back to the default, and positive values are truncated and clamped to at least `1` ms.
- No retry, backoff, rate limit, or reply length cap is defined in `irc.ts`; behavior otherwise relies on the underlying model stream and any upstream API limits.
- Flush scheduling: deferred history injection polls every `50` ms while the recipient is still streaming (`#scheduleBackgroundExchangeFlush` in `packages/coding-agent/src/session/agent-session.ts`).
## Errors
@@ -105,7 +107,7 @@
- unknown op: `Unknown irc op.`
- Unknown, self-addressed, non-running, and non-idle direct targets are reported under `details.notFound` and in the text footer `Unknown / unavailable peers:`.
- If a target has no attached session, it is treated as not found.
- Exceptions thrown by `respondAsBackground` or `runEphemeralTurn` are caught per-target and surfaced under `details.failed` as `{ id, error }`; other recipients still complete.
- Exceptions thrown by `respondAsBackground`, `runEphemeralTurn`, abort handling, or timeout handling are caught per-target and surfaced under `details.failed` as `{ id, error }`; other recipients still complete.
- If no target succeeds, `send` still returns normally with `No recipients received the message.` and optional `failed`/`notFound` metadata.
## Notes
+8 -7
View File
@@ -30,9 +30,11 @@ For normal file-like reads, `splitPathAndSel()` in `packages/coding-agent/src/to
| Suffix | Meaning |
| --- | --- |
| `:raw` | Raw/verbatim mode. Disables structural summaries and line prefixes. |
| `:N` / `:LN` | Start at 1-indexed line `N`, open-ended. |
| `:conflicts` | Render unresolved Git merge-conflict regions for a local file. |
| `:N` / `:LN` / `:N-` | Start at 1-indexed line `N`, open-ended. |
| `:A-B` / `:LA-LB` | Inclusive 1-indexed line range. |
| `:A+C` / `:LA+LC` | `C` lines starting at `A`; tool converts this to end line `A + C - 1`. |
| `:R1,R2,...` | Multiple ranges, sorted and merged before reading (for example `:5-16,960-973`). |
| `:range:raw` or `:raw:range` | Same line selection, but raw output. |
Validation in `parseLineRangeChunk()`:
@@ -42,7 +44,7 @@ Validation in `parseLineRangeChunk()`:
Selector parsing intentionally falls through for unrecognized trailing `:...`; archive and SQLite paths consume their own colon syntax.
URL selectors are parsed separately in `packages/coding-agent/src/tools/fetch.ts` and support only `:raw`, `:N`, `:A-B`, and `:A+C` — no optional `L` prefix there.
URL selectors are parsed separately in `packages/coding-agent/src/tools/fetch.ts`, but use the same line-range parser for `:raw`, `:N`, `:A-B`, `:A+C`, `:5-10,20-30`, and `:range:raw` / `:raw:range`. Because URL ports also use `:`, add a trailing slash before a selector on a host/port URL, e.g. `https://example.com/:80`.
## Outputs
- Single-shot `AgentToolResult` built through `toolResult()` in `packages/coding-agent/src/tools/tool-result.ts`.
@@ -95,7 +97,7 @@ URL selectors are parsed separately in `packages/coding-agent/src/tools/fetch.ts
### Local text files
- No selector: if summarization is enabled and the file is small enough, `#trySummarize()` calls `summarizeCode()`.
- Guards: file size `<= 2 MiB` (`MAX_SUMMARY_BYTES`), line count `<= 20_000` (`MAX_SUMMARY_LINES`).
- Summary output keeps selected declarations and replaces elided spans with `...`. When at least one span is elided, the text content ends with a footer like `[NN lines across MM elided regions; read <path>:raw or a line range like <path>:1-9999 for verbatim content]` so the agent has a concrete recovery selector instead of a bare marker.
- Summary output keeps selected declarations and replaces elided spans with `...` or merged brace-pair lines containing `..`. When at least one span is elided, the text content ends with a footer like `[NN lines elided; re-read needed ranges, e.g. <path>:5-16,40-80]` using concrete ranges from the actual elisions.
- When an elided block sits between matching brace lines, `#renderSummary()` may merge them into one anchored line rather than emitting separate opener/closer lines.
- Explicit selector or summarization miss: streamed text read.
- Default open-ended limit is `min(session setting read.defaultLimit, DEFAULT_MAX_LINES)`.
@@ -104,8 +106,8 @@ URL selectors are parsed separately in `packages/coding-agent/src/tools/fetch.ts
- hashline numbered output when edit mode is hashline, read is not raw, source is mutable, edit tool exists, and `readHashLines !== false`
- otherwise optional line numbers when `readLineNumbers === true`
- raw mode suppresses both
- Prefix format in hashline mode is a `¶PATH#TAG` header followed by `LINE:TEXT`, e.g. `¶src/foo.ts#0a` and `41:def alpha():`, from the session snapshot store plus `formatNumberedLine()` / `formatHashlineHeader()`.
- The `edit`/hashline path consumes that header plus bare line numbers later; the two-hex tag is opaque and only meaningful in the session snapshot store that minted it. Immutable sources and `:raw` intentionally suppress hashline headers.
- Prefix format in hashline mode is a `¶PATH#TAG` header followed by `LINE:TEXT`, e.g. `¶src/foo.ts#0A1B` and `41:def alpha():`, from the session snapshot store plus `formatNumberedLine()` / `formatHashlineHeader()`.
- The `edit`/hashline path consumes that header plus bare line numbers later; the four-hex tag is opaque and only meaningful in the session snapshot store that minted it. Immutable sources and `:raw` intentionally suppress hashline headers.
### Directory listings
- `#readDirectory()` calls `buildDirectoryTree()` with:
@@ -207,7 +209,7 @@ URL selectors are parsed separately in `packages/coding-agent/src/tools/fetch.ts
- `parseReadUrlTarget()` accepts `http://`, `https://`, or `www.` targets.
- Plain URL reads call `executeReadUrl()` in `packages/coding-agent/src/tools/fetch.ts`.
- `:raw` means raw HTML/body fallback path; plain URL reads prefer rendered/reader-friendly output.
- `:N`, `:A-B`, `:A+C` do not refetch. They page over cached output from the prior or current URL render.
- `:N`, `:A-B`, `:A+C`, and comma-separated multi-ranges do not refetch when cached output is usable. They page over cached output from the prior or current URL render.
- URL render pipeline in `renderUrl()`:
1. normalize scheme (`https://` added for bare `www.`)
2. try special handlers for known sites unless raw
@@ -290,7 +292,6 @@ Notes: ...
- URL fetch failure does not throw when HTTP fetch succeeds but `response.ok === false`; it returns a failed URL read with `method: "failed"` and explanatory notes.
## Notes
- `readSchema` examples include `https://example.com:L1-L40`, but URL selector parsing in `packages/coding-agent/src/tools/fetch.ts` does not accept `L` prefixes.
- Hashline anchors are suppressed for raw reads and immutable internal resources because there is no editable backing target for later `edit` consumption.
- `splitPathAndSel()` intentionally treats unknown trailing `:...` as part of the path so `archive.zip:inner/file` and `db.sqlite:table:key` still work.
- `resolveReadPath()` contains macOS-specific filename fallbacks for screenshot timestamps, NFD Unicode normalization, and curly apostrophes.
+57 -33
View File
@@ -1,79 +1,103 @@
# recall
> Search the active Hindsight bank and return raw matching memories.
> Search the active long-term memory backend and return matching memories.
## Source
- Entry: `packages/coding-agent/src/tools/hindsight-recall.ts`
- Entry: `packages/coding-agent/src/tools/memory-recall.ts`
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/recall.md`
- Key collaborators:
- Hindsight collaborators:
- `packages/coding-agent/src/hindsight/state.ts` — session state, recall query defaults, prompt-side auto-recall.
- `packages/coding-agent/src/hindsight/content.ts` — result formatting and UTC timestamp formatting.
- `packages/coding-agent/src/hindsight/client.ts` — HTTP `recall` call and error mapping.
- `packages/coding-agent/src/hindsight/bank.ts` — bank id and tag-filter scoping.
- `docs/tools/retain.md` — shared backend, storage, seeding, and mental-model bootstrap.
- Mnemopi collaborators:
- `packages/coding-agent/src/mnemopi/state.ts` — scoped local recall and result formatting with ids.
- `packages/coding-agent/src/mnemopi/config.ts` — local bank scoping and recall limits.
- `docs/tools/retain.md` — shared backend, storage, scoping, and retention behavior.
## Inputs
| Field | Type | Required | Description |
|---|---|---:|---|
| `query` | `string` | Yes | Natural-language search query. The tool passes it through unchanged. |
| `query` | `string` | Yes | Natural-language search query. The tool passes it through unchanged except Mnemopi `per-project-tagged` may run an internal shared-bank fallback query. |
## Outputs
Returns a single-shot tool result.
When matches exist:
- `content[0].type = "text"`
- `content[0].text = "Found <n> relevant memories (as of YYYY-MM-DD HH:MM UTC):\n\n<bullet list>"`
- each bullet is `- <text> [<type>] (<mentioned_at>)`; the type and timestamp suffixes appear only when those fields are present
- `content[0].text = "Found <n> relevant memory/memories (as of YYYY-MM-DD HH:MM UTC):\n\n<bullet list>"`
- `details = {}`
Hindsight bullet format comes from `formatMemories(...)`:
- each bullet is `- <text> [<type>] (<mentioned_at>)`; the type and timestamp suffixes appear only when those fields are present.
Mnemopi bullet format comes from `formatScopedRecallWithIds(...)`:
- each bullet is `- <content> (id: <id>|id unavailable) [<source>] (<YYYY-MM-DD>) c:<score>`; optional source, date, and score suffixes appear only when present.
When no matches exist:
- `content[0].text = "No relevant memories found."`
- `details = {}`
## Flow
1. `HindsightRecallTool.createIf(...)` only exposes the tool when `memory.backend == "hindsight"`.
2. `execute(...)` wraps the whole operation in `untilAborted(...)` from `@oh-my-pi/pi-utils`.
3. It reads the active `HindsightSessionState`; missing state throws `Hindsight backend is not initialised for this session.`
4. It calls `state.client.recall(...)` with:
- `bankId` from session bootstrap,
- the model-supplied `query`,
- `budget`, `maxTokens`, and `types` from `HindsightConfig`,
- tag filters from the bank scope (`recallTags`, `recallTagsMatch`).
5. `HindsightApi.recall(...)` POSTs `/v1/default/banks/{bank_id}/memories/recall`.
6. Results are formatted into a plain-text list with `formatMemories(...)`; empty results map to the fixed no-match string.
7. Failures are logged with `logger.warn("recall failed", ...)` and rethrown.
1. `MemoryRecallTool.createIf(...)` exposes the tool when `memory.backend` is either `"hindsight"` or `"mnemopi"`.
2. `execute(...)` wraps the operation in `untilAborted(...)`.
3. If the backend is `mnemopi`:
- it reads `session.getMnemopiSessionState()` and throws if the backend was not started;
- it calls `state.recallResultsScoped(params.query)`;
- scoped recall queries each configured recall bank with `recallEnhanced(query, recallLimit, { includeFacts: true, channelId: bank })`, merges/deduplicates results by id/content, sorts them, and truncates to `recallLimit`;
- in `per-project-tagged`, the shared bank may receive one extra fallback query with project-bank literal tokens stripped so broad global memories still match;
- results are formatted with ids for later `memory_edit` use.
4. If the backend is `hindsight`:
- it reads `session.getHindsightSessionState()` and throws if the backend was not started;
- it calls `state.client.recall(...)` with `bankId`, query, configured `budget`, `maxTokens`, `types`, and bank-scope tag filters;
- `HindsightApi.recall(...)` POSTs `/v1/default/banks/{bank_id}/memories/recall`;
- results are formatted into a plain-text list with `formatMemories(...)`.
5. Backend failures are logged with `logger.warn("recall failed", ...)` and rethrown as `Error` instances when needed.
## Modes / Variants
- Tool path: explicit query-only recall. The tool does not compose context from recent turns; that richer path is reserved for backend auto-recall in `HindsightSessionState.beforeAgentStartPrompt(...)` / `maybeRecallOnAgentStart(...)`.
- Bank scoping is inherited from the active `HindsightSessionState`:
- Tool path: explicit query-only recall. It does not compose context from recent turns.
- Backend auto-recall has a richer query-composition path in `HindsightSessionState.beforeAgentStartPrompt(...)` / `maybeRecallOnAgentStart(...)` and `MnemopiSessionState.beforeAgentStartPrompt(...)` / `maybeRecallOnAgentStart(...)`.
- Hindsight bank scoping:
- `global` — no tag filter.
- `per-project` — separate bank id per cwd basename.
- `per-project-tagged` — shared bank id plus `project:<cwd basename>` filter with `tagsMatch = "any"`, so project-tagged and untagged global memories can both surface.
- Session scope: reads cross-session server-side memories, but uses per-session cached config and scope.
- Mnemopi bank scoping:
- `global` — recall reads the shared bank.
- `per-project` — recall reads the project bank.
- `per-project-tagged` — recall reads the project bank and shared bank, then merges results.
- Session scope: reads cross-session memory data, using the active session's cached config and scope.
## Side Effects
- Network
- `POST /v1/default/banks/{bank_id}/memories/recall` via `packages/coding-agent/src/hindsight/client.ts`.
- Session state (transcript, memory, jobs, checkpoints, registries)
- None on success. Unlike backend auto-recall, this tool does not update `lastRecallSnippet` or refresh the system prompt.
- Hindsight: `POST /v1/default/banks/{bank_id}/memories/recall`.
- Mnemopi: none unless configured local runtime providers perform embedding/LLM work during recall.
- Session state
- None on success for the explicit tool path. Unlike backend auto-recall, this tool does not update `lastRecallSnippet` or refresh the system prompt.
- Background work / cancellation
- Aborts through `untilAborted(...)` if the tool call signal is cancelled.
## Limits & Caps
- Client default budget for raw `HindsightApi.recall(...)` is `"mid"`; this tool overrides from config in `packages/coding-agent/src/hindsight/state.ts`.
- Default recall settings from `packages/coding-agent/src/config/settings-schema.ts`:
- Tool availability requires `memory.backend` to be `"hindsight"` or `"mnemopi"`; default `memory.backend` is `"off"`.
- Hindsight client default budget for raw `HindsightApi.recall(...)` is `"mid"`; this tool overrides from config.
- Hindsight recall settings:
- `hindsight.recallBudget = "mid"`
- `hindsight.recallMaxTokens = 1024`
- `hindsight.recallTypes = ["world", "experience"]`
- The explicit tool path does not apply `hindsight.recallContextTurns` or `hindsight.recallMaxQueryChars`; those caps only affect backend auto-recall query composition.
- Mnemopi recall settings:
- `mnemopi.recallLimit = 8`
- `mnemopi.scoping` selects which local bank(s) are searched
- The explicit tool path does not apply `hindsight.recallContextTurns`, `hindsight.recallMaxQueryChars`, `mnemopi.recallContextTurns`, or `mnemopi.recallMaxQueryChars`; those caps only affect backend auto-recall query composition.
## Errors
- Throws `Hindsight backend is not initialised for this session.` when no state exists.
- HTTP and fetch failures become `HindsightError` from `packages/coding-agent/src/hindsight/client.ts` with `statusCode` and parsed `details` when available.
- Non-`Error` failures are normalized to `new Error(String(err))` before rethrow.
- Throws `Mnemopi backend is not initialised for this session.` when `memory.backend == "mnemopi"` but no state exists.
- Throws `Hindsight backend is not initialised for this session.` when `memory.backend == "hindsight"` but no state exists.
- Hindsight HTTP and fetch failures become `HindsightError` with `statusCode` and parsed `details` when available.
- Mnemopi recall target failures inside `collectScopedRecallResults(...)` are caught per bank and logged only when `mnemopi.debug` is enabled; if all targets fail, the tool can return `No relevant memories found.`
- Non-`Error` failures caught by the tool are normalized to `new Error(String(err))` before rethrow.
## Notes
- Shared backend details are in `docs/tools/retain.md`: server-side storage, subagent aliasing, bank scoping, mission setup, and mental-model bootstrap.
- Mental models are not fetched by this tool. They may still already be present in the agent's developer instructions because the backend caches a `<mental_models>` block separately from recall results.
- The tool returns raw memory hits; it does not synthesize across them. Use `reflect` for that path.
- Shared backend details are in `docs/tools/retain.md`: storage, subagent aliasing, bank scoping, mission setup, and mental-model behavior.
- Hindsight mental models are not fetched by this tool. They may already be present in the agent's developer instructions because the backend caches a `<mental_models>` block separately from recall results.
- Mnemopi developer instructions may include a `<memories>` block from auto-recall; this explicit tool does not update that block.
- The tool returns memory hits; it does not synthesize across them. Use `reflect` for that path.
-155
View File
@@ -1,155 +0,0 @@
# recipe
> Run a task exposed by a detected project task runner.
## Source
- Entry: `packages/coding-agent/src/tools/recipe/index.ts`
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/recipe.md`
- Key collaborators:
- `packages/coding-agent/src/tools/recipe/runner.ts` — op parsing, task resolution, prompt model.
- `packages/coding-agent/src/tools/recipe/render.ts` — shell-style call/result rendering.
- `packages/coding-agent/src/tools/recipe/runners/index.ts` — runner registration order.
- `packages/coding-agent/src/tools/recipe/runners/just.ts` — detect `just` recipes from justfiles.
- `packages/coding-agent/src/tools/recipe/runners/pkg.ts` — detect `package.json` scripts and workspaces.
- `packages/coding-agent/src/tools/recipe/runners/cargo.ts` — detect Cargo run/test targets.
- `packages/coding-agent/src/tools/recipe/runners/make.ts` — parse make targets from makefiles.
- `packages/coding-agent/src/tools/recipe/runners/task.ts` — detect Taskfile tasks via `task --list-all`.
- `packages/coding-agent/src/tools/bash.ts` — actual command execution, truncation, cwd/env handling.
## Inputs
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `op` | `string` | Yes | Single string containing the task selector plus trailing arguments. The first whitespace-delimited token selects the task; the remainder is appended verbatim to the resolved runner command. Examples from schema/prompt: `test`, `build --release`, `pkg-a/test`, `crate/bin/server`, `pkg:test --watch`. |
### `op` grammar
```text
op := S* head (S+ tail)?
head := explicit-runner / implicit-task
explicit-runner := runner-id ":" task-token
implicit-task := task-token
runner-id := detected runner id (`just` | `pkg` | `cargo` | `make` | `task`)
task-token := first non-whitespace token; may contain `/`
tail := remaining characters after the first whitespace run
```
Resolution rules from `resolveRunnerAndTask()`:
- Leading whitespace is ignored; an empty `op` throws `ToolError` with the available task list.
- Only the first token is parsed structurally. Everything after the first whitespace run becomes `tail` and is appended to the command unchanged.
- If `head` contains `:` and the prefix matches a detected runner id, the suffix must exactly match a task in that runner.
- Otherwise `head` is treated as a task name and matched across all detected runners.
- If exactly one runner has that task, it is used.
- If multiple runners have that task, the call is rejected and the error tells the model to use `<runner-id>:<task>`.
- Namespaced task names generated by runners use `/`, not `:`. `/` is part of the task name, not a parser separator.
## Outputs
- Delegates directly to `BashTool.execute()` and returns the same `AgentToolResult<BashToolDetails>` shape.
- Success path: one text content block containing merged command output (`result.output` from bash execution, or `(no output)`), plus any timeout clamp notice appended after a blank line.
- Recipe does not return separate `stdout`, `stderr`, or `exitCode` fields. `stdout`/`stderr` are already merged into the text block by bash execution; `exitCode` is only observed indirectly (success requires `0`, non-zero becomes an error).
- Error path: throws `ToolError`; for non-zero exits the message is the merged output followed by `Command exited with code <n>`.
- `details` may include:
- `timeoutSeconds`: effective timeout used by bash.
- `requestedTimeoutSeconds`: only when bash clamped a requested timeout; recipe never sets one itself.
- `meta`: output truncation metadata from bash execution.
- `async`: defined by bash background execution paths, but recipe does not expose an `async` input.
- When bash output is truncated, the full text is stored in an artifact and referenced via bash truncation metadata.
- Call/result rendering in the TUI uses bash shell rendering with a resolved title, command preview, and optional task cwd.
## Flow
1. `RecipeTool.createIf()` in `packages/coding-agent/src/tools/recipe/index.ts` checks `session.settings.get("recipe.enabled")`; disabled returns `null`.
2. It probes every runner in `RUNNERS` from `packages/coding-agent/src/tools/recipe/runners/index.ts` with `Promise.all(...)` in this order: `just`, `pkg`, `cargo`, `make`, `task`.
3. Each runner returns either `null` or a `DetectedRunner { id, label, commandPrefix, tasks }`; runners with zero tasks are discarded.
4. If no runners remain, the tool is not registered.
5. Constructor stores detected runners, instantiates `BashTool`, renders the model-facing description by passing `buildPromptModel(runners)` into `packages/coding-agent/src/prompts/tools/recipe.md`, and builds shell renderers from `createRecipeToolRenderer()`.
6. On execution, `RecipeTool.execute()` calls `resolveCommand(op, this.#runners)`.
7. `resolveCommand()` in `packages/coding-agent/src/tools/recipe/runner.ts`:
1. `parseOp()` trims only leading whitespace, extracts the first non-whitespace token as `head`, and keeps the remainder as `tail`.
2. `resolveRunnerAndTask()` resolves `head` either as `runnerId:taskName` or as an unqualified task name.
3. It throws `ToolError` for empty ops, missing explicit tasks, ambiguous task names, or unknown tasks; all error variants include the available task list.
4. It builds the final shell command with `buildCommand(commandPrefix, commandName, tail)`, joining non-empty parts with spaces.
5. If the task defines `cwd`, that relative path is returned alongside the command.
8. `RecipeTool.execute()` forwards `{ command, cwd }` into `BashTool.execute()`; recipe does not pass timeout, env, async, or pty options.
9. `BashTool.execute()` resolves internal URLs, validates/normalizes cwd against `session.cwd`, clamps timeout, applies bash interception rules, runs the command, and formats the final result.
## Modes / Variants
- Tool enablement:
- Disabled by `recipe.enabled` setting: tool is absent.
- Enabled but no detected tasks: tool is absent.
- Task selection:
- Unqualified task name: succeeds only when exactly one detected runner owns that task.
- Explicit runner-qualified task: `<runner-id>:<task>`.
- Runner detection paths:
- `just`: requires `just` on `PATH`, a justfile, and successful `just --dump --dump-format=json`.
- `pkg`: requires a readable root `package.json`; picks a package manager command from lockfiles or `bun` availability; discovers root scripts and workspace package scripts.
- `cargo`: requires `cargo` on `PATH`, `Cargo.toml`, and successful `cargo metadata --no-deps --format-version=1`.
- `make`: requires `make` on `PATH` and a makefile; parses targets statically.
- `task`: requires `task` on `PATH`, a Taskfile, and successful `task --list-all --json`.
- Execution path:
- Always the synchronous `bash` call surface from recipe inputs.
- Bash may still auto-background long-running work if `bash.autoBackground.enabled` and session async job support are enabled.
## Side Effects
- Filesystem
- Reads manifests from the session cwd during detection: justfiles, `package.json`, workspace `package.json` files, `Cargo.toml`, makefiles, `Taskfile.yml` / `Taskfile.yaml`.
- Command execution runs in `session.cwd` or a task-specific relative cwd resolved under it.
- Bash may allocate output artifacts for truncated command output.
- Subprocesses / native bindings
- Detection may spawn `just --dump --dump-format=json`, `cargo metadata --no-deps --format-version=1`, and `task --list-all --json`.
- Execution spawns the resolved shell command through `BashTool` / `executeBash()`.
- Session state (transcript, memory, jobs, checkpoints, registries)
- Tool availability depends on session settings.
- Constructor prompt text is specialized to detected runners/tasks.
- Bash execution may create async job records and output artifacts if bash auto-background triggers.
- User-visible prompts / interactive UI
- The model-facing tool description lists detected runners and up to 20 tasks per runner.
- TUI rendering shows a shell-style preview using the resolved title/command/cwd.
- Background work / cancellation
- Detection is parallelized across runners.
- Runtime command execution honors the passed abort signal through `BashTool`.
## Limits & Caps
- Prompt task listing is capped at `PROMPT_TASK_LIMIT = 20` per runner in `packages/coding-agent/src/tools/recipe/runner.ts`; this affects the rendered tool description, not execution.
- Recipe itself defines no timeout input; delegated bash execution therefore uses bash's default `timeout = 300` seconds from `packages/coding-agent/src/tools/bash.ts`.
- Bash clamps timeouts to the configured bash range (`clampTimeout("bash", ...)` in `packages/coding-agent/src/tools/bash.ts`), but recipe cannot request a custom value.
- `pkg` workspace discovery normalizes workspace globs to `.../package.json` and sorts matched package files lexicographically before task generation.
- `cargo` deduplicates generated task names with a `Set`, so duplicate targets collapse to one recipe task.
## Errors
- Detection failures in runner modules are mostly soft-failed:
- Missing binaries, missing manifests, parse failures, or non-zero probe exits usually return `null` and log with `logger.debug(...)`.
- Result: the affected runner disappears instead of surfacing an error to the model.
- Invocation failures are hard errors from `resolveRunnerAndTask()`:
- Empty `op`.
- Explicit runner prefix with missing/empty task.
- Ambiguous unqualified task name across runners.
- Unknown task name.
- Execution failures come from `BashTool.execute()`:
- Invalid cwd.
- Bash interceptor blocks.
- Aborts/timeouts.
- Non-zero exit codes.
- Missing exit status.
- All `resolveRunnerAndTask()` errors include the current available task list to help the model retry.
## Notes
- `RecipeTool` sets `concurrency = "exclusive"`; calls do not run concurrently with other exclusive tools.
- Tool registration is all-or-nothing per runner: a detected runner with zero tasks is dropped.
- Runner ids are fixed string literals from the runner modules: `just`, `pkg`, `cargo`, `make`, `task`.
- `buildPromptModel()` includes each task's rendered command (`commandPrefix` + `commandName`) and relative cwd when present; the prompt therefore exposes the exact shell form recipe will run.
- `pkg` task names:
- Root `package.json` scripts keep bare names like `test`.
- Workspace scripts are always namespaced as `<package-name-or-dir>/<script>` and set `cwd` to that package directory.
- Script names are shell-quoted into `commandName`, so a task like `build` becomes `bun run 'build'` / `npm run 'build'` / similar.
- `pkg` command prefix selection prefers lockfiles in this order: `bun.lock`/`bun.lockb`, `pnpm-lock.yaml`, `yarn.lock`, `package-lock.json`/`npm-shrinkwrap.json`; otherwise it falls back to `bun run` if `bun` exists, else `npm run`.
- `cargo` task names are generated from metadata targets:
- Single-package manifests: `bin/<name>`, `example/<name>`, `test/<name>`.
- Multi-package workspaces: `<package>/bin/<name>`, `<package>/example/<name>`, `<package>/test/<name>`.
- Each task overrides `commandPrefix` to the full `cargo run ... --bin|--example` or `cargo test ... --test` prefix, and `commandName` to the quoted target name.
- `make` target parsing is static text parsing, not `make -qp` output:
- Recognizes makefiles named `Makefile`, `makefile`, `GNUmakefile`.
- Uses `.PHONY` lines to decide whether to include undocumented file targets; without any `.PHONY`, all parsed targets are exposed.
- If `.PHONY` exists, documented non-phony targets are kept with ` (file target)` appended to `doc`.
- `just` detection ignores private recipes and preserves declared parameter names only for prompt display; execution still accepts arbitrary `tail` text.
- `task` detection uses `desc` first, then `summary`, for task documentation.
- Recipe has no env input of its own. Commands inherit whatever environment `BashTool` supplies for normal bash execution in the session.
+56 -32
View File
@@ -1,73 +1,97 @@
# reflect
> Ask the Hindsight server to synthesize an answer over the active memory bank.
> Synthesize an answer over the active long-term memory backend.
## Source
- Entry: `packages/coding-agent/src/tools/hindsight-reflect.ts`
- Entry: `packages/coding-agent/src/tools/memory-reflect.ts`
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/reflect.md`
- Key collaborators:
- Hindsight collaborators:
- `packages/coding-agent/src/hindsight/bank.ts` — best-effort bank mission initialization.
- `packages/coding-agent/src/hindsight/state.ts` — session state, shared bank scope, recall/reflect config.
- `packages/coding-agent/src/hindsight/client.ts` — HTTP `reflect` call and error mapping.
- `docs/tools/retain.md` — shared backend, storage, seeding, and mental-model bootstrap.
- Mnemopi collaborators:
- `packages/coding-agent/src/mnemopi/state.ts` — scoped local recall and context formatting.
- `docs/tools/retain.md` — shared backend, storage, scoping, and mental-model behavior.
## Inputs
| Field | Type | Required | Description |
|---|---|---:|---|
| `query` | `string` | Yes | Question to answer from long-term memory. |
| `context` | `string` | No | Extra guidance sent to the Hindsight reflect endpoint. |
| `context` | `string` | No | Extra guidance. Hindsight sends it as `context`; Mnemopi appends trimmed context to the recall query under `Additional context:`. |
## Outputs
Returns a single-shot tool result:
Returns a single-shot tool result.
Hindsight:
- `content[0].type = "text"`
- `content[0].text = response.text?.trim() || "No relevant information found to reflect on."`
- `details = {}`
- The tool returns the Hindsight server's synthesized text directly; it does not expose raw recall hits.
The tool returns the Hindsight server's synthesized text directly; it does not expose raw recall hits.
Mnemopi:
- if no scoped recall results exist: `content[0].text = "No relevant information found to reflect on."`
- otherwise: `content[0].text = "Based on recalled memories:\n\n<formatted context>"`
- `details = {}`
- The local path performs recall plus formatting; it does not call a separate synthesis endpoint.
## Flow
1. `HindsightReflectTool.createIf(...)` only exposes the tool when `memory.backend == "hindsight"`.
1. `MemoryReflectTool.createIf(...)` exposes the tool when `memory.backend` is either `"hindsight"` or `"mnemopi"`.
2. `execute(...)` runs under `untilAborted(...)`.
3. It reads the active `HindsightSessionState`; missing state throws `Hindsight backend is not initialised for this session.`
4. Before reflecting, it calls `ensureBankMission(...)` with the current `bankId`, config, and process-local `missionsSet`.
5. `ensureBankMission(...)` best-effort `PUT`s `/v1/default/banks/{bank_id}` with `reflect_mission` and optional `retain_mission` exactly once per bank/process; failures are swallowed.
6. It calls `state.client.reflect(...)` with the model `query`, optional `context`, configured recall budget, and bank-scope tag filters.
7. `HindsightApi.reflect(...)` POSTs `/v1/default/banks/{bank_id}/reflect` and defaults its own budget to `"low"` when callers omit one; this tool always passes the configured budget.
8. Blank or whitespace-only responses are replaced with `No relevant information found to reflect on.`
9. Failures are logged with `logger.warn("reflect failed", ...)` and rethrown.
3. If the backend is `mnemopi`:
- it reads `session.getMnemopiSessionState()` and throws if the backend was not started;
- if `context` has non-whitespace content, it recalls with `<query>\n\nAdditional context:\n<context>`; otherwise it recalls with `query`;
- it calls `state.recallResultsScoped(...)` using the same local scoping and merge behavior as `recall`;
- if results exist, it renders them through `state.formatContextScoped(...)` and prefixes `Based on recalled memories:`.
4. If the backend is `hindsight`:
- it reads `session.getHindsightSessionState()` and throws if the backend was not started;
- it calls `ensureBankMission(...)` with the current `bankId`, config, and process-local `missionsSet`;
- `ensureBankMission(...)` best-effort `PUT`s `/v1/default/banks/{bank_id}` with `reflect_mission` and optional `retain_mission` exactly once per bank/process; failures are swallowed;
- it calls `state.client.reflect(...)` with `query`, optional `context`, configured recall budget, and bank-scope tag filters;
- `HindsightApi.reflect(...)` POSTs `/v1/default/banks/{bank_id}/reflect` and defaults its own budget to `"low"` when callers omit one; this tool always passes the configured budget;
- blank or whitespace-only responses are replaced with `No relevant information found to reflect on.`
5. Backend failures are logged with `logger.warn("reflect failed", ...)` and rethrown as `Error` instances when needed.
## Modes / Variants
- Tool path: one reflect request, optionally focused by `context`.
- Bank scoping is inherited from the active `HindsightSessionState`:
- Hindsight tool path: one remote reflect request, optionally focused by `context`.
- Mnemopi tool path: one local scoped recall followed by context formatting.
- Hindsight bank scoping:
- `global` — no tag filter.
- `per-project` — separate bank id per cwd basename.
- `per-project-tagged` — shared bank id plus `project:<cwd basename>` filter with `tagsMatch = "any"`.
- Session scope: reads cross-session server-side memories, but does not persist local output.
- Mnemopi bank scoping:
- `global` — reads the shared bank.
- `per-project` — reads the project bank.
- `per-project-tagged` — reads the project bank and shared bank, then merges results.
- Session scope: reads cross-session memory data, but does not persist local output.
## Side Effects
- Network
- Optional `PUT /v1/default/banks/{bank_id}` from `ensureBankMission(...)`.
- `POST /v1/default/banks/{bank_id}/reflect` via `packages/coding-agent/src/hindsight/client.ts`.
- Session state (transcript, memory, jobs, checkpoints, registries)
- Reads session-held bank scope and config only. Does not update `lastRecallSnippet`, the mental-model cache, or the retain queue.
- Hindsight: optional `PUT /v1/default/banks/{bank_id}` from `ensureBankMission(...)`, then `POST /v1/default/banks/{bank_id}/reflect`.
- Mnemopi: none unless configured embedding or LLM providers are used by the local runtime during recall.
- Session state
- Reads session-held backend scope and config only. Does not update `lastRecallSnippet`, Hindsight mental-model cache, or retain queues.
- Background work / cancellation
- Aborts through `untilAborted(...)` if the tool call signal is cancelled.
## Limits & Caps
- Tool availability requires `memory.backend` to be `"hindsight"` or `"mnemopi"`; default `memory.backend` is `"off"`.
- Tool-level params: only `query` is required; `context` is optional.
- Default budget setting comes from `hindsight.recallBudget` in `packages/coding-agent/src/config/settings-schema.ts`; default `"mid"`.
- `reflect` itself has no client-side token cap parameter here; unlike `recall`, the tool does not pass `maxTokens`.
- Mission initialization tracks up to `MISSION_SET_CAP = 10_000` bank ids in `packages/coding-agent/src/hindsight/bank.ts`, then drops the oldest half of the sorted set.
- Hindsight budget setting comes from `hindsight.recallBudget`, default `"mid"`.
- Hindsight `reflect` has no client-side token cap parameter here; unlike `recall`, the tool does not pass `maxTokens`.
- Hindsight mission initialization tracks up to `MISSION_SET_CAP = 10_000` bank ids, then drops the oldest half of the sorted set.
- Mnemopi result count is capped by `mnemopi.recallLimit`, default `8`.
## Errors
- Throws `Hindsight backend is not initialised for this session.` when no state exists.
- HTTP and fetch failures become `HindsightError` from `packages/coding-agent/src/hindsight/client.ts` with `statusCode` and parsed `details` when available.
- `ensureBankMission(...)` failures are silent to the tool caller; only the later reflect request can fail visibly.
- Non-`Error` failures are normalized to `new Error(String(err))` before rethrow.
- Throws `Mnemopi backend is not initialised for this session.` when `memory.backend == "mnemopi"` but no state exists.
- Throws `Hindsight backend is not initialised for this session.` when `memory.backend == "hindsight"` but no state exists.
- Hindsight HTTP and fetch failures become `HindsightError` with `statusCode` and parsed `details` when available.
- Hindsight `ensureBankMission(...)` failures are silent to the tool caller; only the later reflect request can fail visibly.
- Mnemopi recall target failures inside `collectScopedRecallResults(...)` are caught per bank and logged only when `mnemopi.debug` is enabled; if all targets fail, the tool can return the no-information text.
- Non-`Error` failures caught by the tool are normalized to `new Error(String(err))` before rethrow.
## Notes
- Shared backend details are in `docs/tools/retain.md`: server-side storage, subagent aliasing, bank scoping, seed mental models from `packages/coding-agent/src/hindsight/seeds.json`, and mental-model prompt injection.
- `reflect` does not read the cached `<mental_models>` block directly. It queries the Hindsight server over the bank contents. The same session may also have separate mental-model context injected into its developer instructions.
- Reflect mission and retain mission are bank-level server settings, not per-request payload. The tool just ensures they are present best-effort before reflecting.
- Shared backend details are in `docs/tools/retain.md`: storage, subagent aliasing, bank scoping, seed mental models, and prompt injection.
- Hindsight `reflect` does not read the cached `<mental_models>` block directly. It queries the Hindsight server over the bank contents. The same session may also have separate mental-model context injected into its developer instructions.
- Hindsight reflect mission and retain mission are bank-level server settings, not per-request payload. The tool just ensures they are present best-effort before reflecting.
- Mnemopi `reflect` is local recall plus formatting, so its output shape differs from Hindsight's remote synthesized answer.
+49 -31
View File
@@ -1,6 +1,6 @@
# resolve
> Finalizes a queued preview action by applying or discarding it.
> Finalizes a pending action by applying or discarding it.
## Source
- Entry: `packages/coding-agent/src/tools/resolve.ts`
@@ -9,64 +9,82 @@
- `docs/resolve-tool-runtime.md` — preview/apply runtime reference
- `packages/coding-agent/src/extensibility/custom-tools/loader.ts` — forwards custom pending actions into the queue
- `packages/coding-agent/src/tools/ast-edit.ts` — built-in preview producer example
- `packages/coding-agent/src/session/agent-session.ts` — tool-choice queue and invoker access
- `packages/coding-agent/src/session/agent-session.ts` — tool-choice queue, standing resolve handler, and invoker access
## Inputs
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `action` | `"apply" | "discard"` | Yes | Whether to commit or reject the queued preview. |
| `reason` | `string` | Yes | Required explanation passed through to the queued callback. |
| `action` | `"apply" | "discard"` | Yes | Whether to commit or reject the pending action. |
| `reason` | `string` | Yes | Required explanation passed through to the handler. |
| `extra` | `Record<string, unknown>` | No | Free-form metadata passed through to the handler. Plan approval uses this for data such as a title slug; preview-style actions usually ignore it. |
## Outputs
- Single-shot result.
- `execute()` returns whatever the queued invoker returns, with `details` wrapped/augmented to include:
- `execute()` returns whatever the queued or standing invoker returns, with `details` wrapped/augmented to include:
- `action`
- `reason`
- `extra?`
- `sourceToolName?`
- `label?`
- `sourceResultDetails?` — original `result.details` from the apply/reject callback when present
- If `discard` has no custom reject callback, the default success payload is `Discarded: <label>. Reason: <reason>`.
- If `discard` has no custom reject callback, or the reject callback returns `undefined`, the default success payload is `Discarded: <label>. Reason: <reason>`.
- The TUI renderer is inline and merges call+result into one block.
## Flow
1. Preview-producing code calls `queueResolveHandler(...)` with a label, source tool name, and `apply(reason)` callback, plus optional `reject(reason)`.
2. `queueResolveHandler(...)` asks the session for a forced `resolve` tool choice and pushes it into the tool-choice queue with `pushOnce(...)`.
3. The queued entry is marked `now: true`; if the model rejects that forced tool choice, `onRejected` returns `requeue`, so the reminder comes back.
4. `queueResolveHandler(...)` also injects a `resolve-reminder` steering message: `This is a preview. Call the resolve tool to apply or discard these changes.`
5. When `resolve.execute()` runs, it wraps the call in `untilAborted(...)` and fetches the current queue invoker with `session.peekQueueInvoker()`.
6. If no invoker exists, it throws `ToolError("No pending action to resolve. Nothing to apply or discard.")`.
7. Otherwise it invokes the queued callback with `{ action, reason }`.
8. For `apply`, it always executes the producer's `apply(reason)` callback.
9. For `discard`, it executes `reject(reason)` when provided; if that callback is absent or returns `undefined`, `resolve` fabricates the default discard message.
10. Before returning, it merges resolve metadata into `result.details` so renderer/UI code can show the action, label, and originating tool.
1. Preview-producing code can call `queueResolveHandler(...)` with a label, source tool name, `apply(reason, extra?)` callback, and optional `reject(reason, extra?)` callback.
2. Modes can also register a standing resolve handler through `session.setStandingResolveHandler(...)`; `resolve.execute()` consults it only when no queued invoker is active.
3. `queueResolveHandler(...)` asks the session for a forced `resolve` tool choice and pushes it into the tool-choice queue with `pushOnce(...)`.
4. The queued entry is marked `now: true`; if the model rejects that forced tool choice, `onRejected` returns `requeue`, so the reminder comes back.
5. `queueResolveHandler(...)` also injects a `resolve-reminder` steering message:
```text
<system-reminder>
This is a preview. Call the `resolve` tool to apply or discard these changes.
</system-reminder>
```
6. When `resolve.execute()` runs, it wraps the call in `untilAborted(...)` and fetches `session.peekQueueInvoker?.() ?? session.peekStandingResolveHandler?.()`.
7. If no invoker exists, it throws `ToolError("No pending action to resolve. Nothing to apply or discard.")`.
8. Otherwise it invokes the current handler with the full params object.
9. `runResolveInvocation(...)` builds base details from `action`, `reason`, `extra`, `sourceToolName`, and `label`.
10. For `apply`, it calls the producer's `apply(reason, extra)` callback.
11. If `apply` throws, `runResolveInvocation(...)` calls `onApplyError` when present. The queued preview integration uses this to re-push the resolve directive and steering reminder so the action remains pending. Non-`ToolError` exceptions are wrapped as `ToolError("Apply failed: <message>")`.
12. For `discard`, it calls `reject(reason, extra)` when provided. If no reject callback exists or it returns `undefined`, `resolve` fabricates the default discard message.
13. Before returning callback results, it merges resolve metadata into `result.details` so renderer/UI code can show the action, label, and originating tool.
## Modes / Variants
- `apply`: runs the queued `apply(reason)` callback and returns its content.
- `discard` with reject callback: runs `reject(reason)` and returns that callback's content.
- `discard` without reject callback: returns the built-in `Discarded: ...` text payload.
- `apply`: runs the pending action's `apply(reason, extra?)` callback and returns its content.
- `discard` with reject callback: runs `reject(reason, extra?)` and returns that callback's content when non-`undefined`.
- `discard` without reject callback, or with a reject callback returning `undefined`: returns the built-in `Discarded: ...` text payload.
- Queued handler: one in-flight tool-choice queue invoker, used by preview producers such as `ast_edit`.
- Standing handler: long-lived mode-owned handler, used as a fallback when no queue invoker is active.
## Side Effects
- Session state
- Consumes the current pending preview through the session tool-choice queue; there is no separate pending-action stack.
- Adds a `resolve-reminder` steering message when a preview is queued.
- Consumes or invokes the current pending action through the session tool-choice queue or standing handler; `resolve` does not maintain its own stack.
- Adds a `resolve-reminder` steering message when a queued preview is registered.
- On queued apply failure, requeues the same pending action before rethrowing so the model can discard or retry instead of losing the pending preview.
- User-visible prompts / interactive UI
- No direct prompt. The visible effect depends on the preview-producing tool and the resolve renderer.
- The visible effect depends on the preview-producing tool and the resolve renderer.
- Renderer result blocks show `Accept`, `Discard`, or `Failed`, include the pending action label, and display the reason.
- Background work / cancellation
- `untilAborted(...)` lets abort signals interrupt resolution before invoking the callback completes.
- `untilAborted(...)` lets abort signals interrupt resolution before or while the callback awaits.
## Limits & Caps
- Hidden tool: not discoverable in the normal tool index (`packages/coding-agent/src/tools/resolve.ts`, `packages/coding-agent/src/session/agent-session.ts`).
- Exactly one active queue invoker is consulted per call via `session.peekQueueInvoker()`.
- There is no independent queue depth cap in this tool; ordering follows the shared tool-choice queue (`docs/resolve-tool-runtime.md`).
- Hidden tool: `ResolveTool.hidden = true`, and normal requested-tool filtering removes `resolve`; `createTools(...)` adds it separately as a hidden tool.
- Exactly one active queue invoker is consulted per call via `session.peekQueueInvoker()`; if none exists, one standing handler may be consulted via `session.peekStandingResolveHandler()`.
- There is no independent queue depth cap in this tool; ordering follows the shared tool-choice queue and mode-owned standing handler lifecycle.
## Errors
- No pending preview: throws `ToolError("No pending action to resolve. Nothing to apply or discard.")`.
- Any exception from the queued `apply` / `reject` callback propagates through `resolve`.
- No pending action or standing handler: throws `ToolError("No pending action to resolve. Nothing to apply or discard.")`.
- `apply` callback throws `ToolError`: the original `ToolError` propagates.
- `apply` callback throws any other value: `resolve` wraps it as `ToolError("Apply failed: <message>")` after running `onApplyError` when present.
- `reject` callback exceptions propagate without the apply-specific wrapper.
- Aborts during `untilAborted(...)` surface as the underlying abort error from the utility.
## Notes
- `reason` is informational; `resolve` passes it through but does not interpret it.
- `queueResolveHandler(...)` is the canonical built-in integration point; custom tools use `pushPendingAction(...)`, which the loader forwards into the same mechanism.
- The tool only works because another tool already staged a preview and forced a one-shot `resolve` choice.
- `reason` and `extra` are passed through; `resolve` itself does not interpret them.
- `queueResolveHandler(...)` is the canonical built-in preview integration point; custom tools use `pushPendingAction(...)`, which the loader forwards into the same mechanism.
- Standing handlers let modes accept `resolve` invocations without forcing the tool choice every turn.
- `sourceResultDetails` is added only when the apply/reject callback returned a non-null `details` field; custom pending-action `details` are not forwarded automatically by the loader.
+75 -49
View File
@@ -1,11 +1,11 @@
# retain
> Queue durable facts for asynchronous write into the active Hindsight bank.
> Store durable facts through the active long-term memory backend.
## Source
- Entry: `packages/coding-agent/src/tools/hindsight-retain.ts`
- Entry: `packages/coding-agent/src/tools/memory-retain.ts`
- Model-facing prompt: `packages/coding-agent/src/prompts/tools/retain.md`
- Key collaborators:
- Hindsight collaborators:
- `packages/coding-agent/src/hindsight/state.ts` — per-session queue, flush, auto-retain.
- `packages/coding-agent/src/hindsight/backend.ts` — session bootstrap, prompt injection, subagent aliasing.
- `packages/coding-agent/src/hindsight/bank.ts` — bank id derivation, tag scoping, mission setup.
@@ -14,85 +14,111 @@
- `packages/coding-agent/src/hindsight/mental-models.ts` — bank-scoped mental-model seeding and cache rendering.
- `packages/coding-agent/src/hindsight/seeds.json` — built-in mental-model seed definitions.
- `packages/coding-agent/src/hindsight/transcript.ts` — extracts user/assistant turns for auto-retain.
- Mnemopi collaborators:
- `packages/coding-agent/src/mnemopi/backend.ts` — local backend bootstrap, prompt injection, subagent aliasing, enqueue/clear.
- `packages/coding-agent/src/mnemopi/state.ts` — scoped recall/retain state and local writes.
- `packages/coding-agent/src/mnemopi/config.ts` — local SQLite path, bank, scoping, provider settings.
- `packages/mnemopi/src/core/memory.ts` — local memory runtime used by `remember(...)`.
## Inputs
| Field | Type | Required | Description |
|---|---|---:|---|
| `items` | `Array<{ content: string; context?: string }>` | Yes | One or more memories to queue. `minItems: 1`. Each item must be self-contained; `context` is optional per-item provenance. |
| `items` | `Array<{ content: string; context?: string }>` | Yes | One or more memories to store. `minItems: 1`. Each item must be self-contained; `context` is optional per-item provenance. |
## Outputs
Returns a single-shot tool result:
The output depends on the active `memory.backend`.
Hindsight:
- `content[0].type = "text"`
- `content[0].text = "<count> memory queued."` or `"<count> memories queued."`
- `details = { count: number }`
- The write is not confirmed before the tool returns. The queue flushes later; flush failures emit a session warning notice and are not returned to the model.
The write is not confirmed before the tool returns. The queue flushes later; flush failures emit a session warning notice and are not returned to the model.
Mnemopi:
- `content[0].type = "text"`
- `content[0].text = "<count> memory stored."` or `"<count> memories stored."`
- `details = { count: number }`
- The tool calls the local backend synchronously, but `rememberScoped(...)` catches per-item write failures and returns `undefined`; the tool still reports the requested count.
## Flow
1. `HindsightRetainTool.createIf(...)` only exposes the tool when `memory.backend == "hindsight"` in `packages/coding-agent/src/tools/hindsight-retain.ts`.
2. `execute(...)` fetches `session.getHindsightSessionState()` and throws if the Hindsight backend was not started.
3. Each input item is handed to `HindsightSessionState.enqueueRetain(...)` in `packages/coding-agent/src/hindsight/state.ts`.
4. `HindsightRetainQueue.enqueue(...)` appends the item and either:
- flushes immediately when the queue reaches `RETAIN_FLUSH_BATCH_SIZE`, or
- starts a debounce timer for `RETAIN_FLUSH_INTERVAL_MS`.
5. On flush, `HindsightRetainQueue.#doFlush(...)`:
- verifies the session still owns this state,
- calls `ensureBankMission(...)` once per bank/process before writing,
- maps queued items to `MemoryItemInput` with `context ?? config.retainContext`, `metadata.session_id`, and bank-scope tags,
- sends one async `retainBatch(...)` request.
6. The tool returns immediately after enqueueing; it does not await the HTTP write.
1. `MemoryRetainTool.createIf(...)` exposes the tool when `memory.backend` is either `"hindsight"` or `"mnemopi"`.
2. `execute(...)` re-reads `memory.backend` and dispatches to the matching session state.
3. If the backend is `mnemopi`:
- it fetches `session.getMnemopiSessionState()` and throws if the backend was not started;
- for each item, it calls `state.rememberScoped(item.content, ...)` with `source: "coding-agent-retain"`, `importance: 0.75`, `scope: "bank"`, `extract: true`, `extractEntities: true`, `veracity: "tool"`, `memoryType: "fact"`, and metadata `{ session_id, cwd, context, tool: "retain" }`;
- writes go to the scoped retain bank selected by `packages/coding-agent/src/mnemopi/config.ts`.
4. If the backend is `hindsight`:
- it fetches `session.getHindsightSessionState()` and throws if the backend was not started;
- each input item is handed to `HindsightSessionState.enqueueRetain(...)`;
- `HindsightRetainQueue.enqueue(...)` appends the item and either flushes immediately when the queue reaches `RETAIN_FLUSH_BATCH_SIZE`, or starts a debounce timer for `RETAIN_FLUSH_INTERVAL_MS`;
- on flush, `HindsightRetainQueue.#doFlush(...)` verifies ownership, best-effort initializes the bank mission, maps items to `MemoryItemInput` with `context ?? config.retainContext`, `metadata.session_id`, and bank-scope tags, then sends one async `retainBatch(...)` request.
## Modes / Variants
- Tool path: queued batch write only.
- Bank scoping comes from `computeBankScope(...)` in `packages/coding-agent/src/hindsight/bank.ts`:
- Hindsight tool path: queued batch write only.
- Mnemopi tool path: direct local `remember(...)` into the scoped retain bank.
- Hindsight bank scoping from `computeBankScope(...)`:
- `global` — one shared bank, no project tags.
- `per-project` — bank id gets `-<cwd basename>` appended.
- `per-project-tagged` — shared bank plus `project:<cwd basename>` tags on retained memories.
- Mnemopi bank scoping from `resolveBankScope(...)`:
- `global` — retain and recall use the shared bank.
- `per-project` — retain and recall use the project bank.
- `per-project-tagged` — retain writes project-local memories; recall also reads the shared bank.
- Session scope:
- tool-called retains are per-session queued work in `HindsightSessionState`,
- persisted memories are cross-session server-side bank data,
- subagents alias the parent `HindsightSessionState`, so their `retain` calls write into the same bank and queue.
- tool-called retains are per-session work for the active backend;
- persisted Hindsight memories are cross-session server-side bank data;
- persisted Mnemopi memories are local SQLite data;
- subagents alias parent memory state for both supported backends.
## Side Effects
- Filesystem
- None for retained memories. No local memory file is written.
- Hindsight: none for retained memories. No local memory file is written.
- Mnemopi: writes to local SQLite under `mnemopi.dbPath`, defaulting beneath the agent memories directory (`mnemopi/mnemopi.db`) with one database file per scoped bank when needed.
- Network
- `POST /v1/default/banks/{bank_id}/memories` via `retainBatch(...)` in `packages/coding-agent/src/hindsight/client.ts`.
- Optional `PUT /v1/default/banks/{bank_id}` via `ensureBankMission(...)` before first write per bank/process.
- Session state (transcript, memory, jobs, checkpoints, registries)
- Appends to the in-memory `HindsightRetainQueue` on the active `HindsightSessionState`.
- Includes `metadata.session_id` on each retained item.
- Shares parent state for subagents (`aliasOf` path in `packages/coding-agent/src/hindsight/backend.ts`).
- Hindsight: `POST /v1/default/banks/{bank_id}/memories` via `retainBatch(...)`, plus optional `PUT /v1/default/banks/{bank_id}` via `ensureBankMission(...)` before first write per bank/process.
- Mnemopi: none unless configured embedding or LLM providers make calls during extraction.
- Session state
- Hindsight: appends to the in-memory `HindsightRetainQueue`, includes `metadata.session_id`, and shares parent state for subagents.
- Mnemopi: writes through the session's scoped `Mnemopi` instance, includes `session_id`, `cwd`, and optional `context`, and shares scoped resources with subagents.
- User-visible prompts / interactive UI
- On async flush failure, emits `session.emitNotice("warning", ...)`; the model is not told.
- Hindsight async flush failures emit `session.emitNotice("warning", ...)`; the model is not told.
- Mnemopi write failures are logged by `rememberInScope(...)`; the tool response does not expose per-item failures.
- Background work / cancellation
- Flush runs later on timer, queue-size threshold, `agent_end`, backend `enqueue(...)`, or backend `clear(...)`.
- Hindsight flush runs later on timer, queue-size threshold, `agent_end`, backend `enqueue(...)`, or backend `clear(...)`.
- Mnemopi fact/entity extraction may continue in the Mnemopi runtime; backend `enqueue(...)` calls `flushExtractions()` before sleeping sessions.
## Limits & Caps
- Input schema requires `items.length >= 1` in `packages/coding-agent/src/tools/hindsight-retain.ts`.
- Queue flush threshold: `RETAIN_FLUSH_BATCH_SIZE = 16` in `packages/coding-agent/src/hindsight/state.ts`.
- Queue debounce: `RETAIN_FLUSH_INTERVAL_MS = 5_000` in `packages/coding-agent/src/hindsight/state.ts`.
- Queue writes use `retainBatch(..., { async: true })`; the client does not wait for server-side consolidation.
- Shared auto-retain settings on the same backend:
- Input schema requires `items.length >= 1`.
- Tool availability requires `memory.backend` to be `"hindsight"` or `"mnemopi"`; default `memory.backend` is `"off"`.
- Hindsight queue flush threshold: `RETAIN_FLUSH_BATCH_SIZE = 16`.
- Hindsight queue debounce: `RETAIN_FLUSH_INTERVAL_MS = 5_000`.
- Hindsight queue writes use `retainBatch(..., { async: true })`; the client does not wait for server-side consolidation.
- Hindsight auto-retain settings:
- `hindsight.retainEveryNTurns` default `3`
- `hindsight.retainOverlapTurns` default `2`
- `hindsight.retainContext` default `"omp"`
- `hindsight.retainMode` default `"full-session"`
from `packages/coding-agent/src/config/settings-schema.ts`.
- Mnemopi retain settings:
- `mnemopi.retainEveryNTurns` default `4`
- `mnemopi.autoRetain` controls automatic retention of completed conversation turns
- `mnemopi.scoping` selects `global`, `per-project`, or `per-project-tagged`
## Errors
- Throws `Hindsight backend is not initialised for this session.` when no state exists.
- Queue enqueue on disposed state throws `Hindsight retain queue is closed.`
- Flush-time API failures are caught in `HindsightRetainQueue.#doFlush(...)`, logged, and converted into a warning notice instead of a tool error.
- Mission creation failures are swallowed in `ensureBankMission(...)`; writes continue.
- Throws `Mnemopi backend is not initialised for this session.` when `memory.backend == "mnemopi"` but no state exists.
- Throws `Hindsight backend is not initialised for this session.` when `memory.backend == "hindsight"` but no state exists.
- Hindsight queue enqueue on disposed state throws `Hindsight retain queue is closed.`
- Hindsight flush-time API failures are caught in `HindsightRetainQueue.#doFlush(...)`, logged, and converted into a warning notice instead of a tool error.
- Hindsight mission creation failures are swallowed in `ensureBankMission(...)`; writes continue.
- Mnemopi `remember(...)` failures are caught in `MnemopiSessionState.rememberInScope(...)`, logged, and not rethrown to the tool caller.
## Notes
- Storage is server-side. `hindsightBackend.clear(...)` only clears local cache/state and warns that upstream deletion must happen in Hindsight UI or `deleteBank`; see `packages/coding-agent/src/hindsight/backend.ts`.
- Auto-retain uses the same bank but a different path than this tool: `retainSession(...)` extracts plain user/assistant transcript from `packages/coding-agent/src/hindsight/transcript.ts`, strips `<memories>` / `<mental_models>` blocks via `stripMemoryTags(...)`, and calls single-item `retain(...)`.
- `retain` itself does not seed or read mental models. Mental-model bootstrap lives in the shared backend: `HindsightSessionState.runMentalModelLoad(...)` optionally resolves seeds from `packages/coding-agent/src/hindsight/seeds.json`, creates missing models with `ensureMentalModels(...)`, then caches a rendered `<mental_models>` block for prompt injection.
- Built-in seeds are `user-preferences`, `project-conventions`, and `project-decisions`. `projectTagged: true` seeds inherit the active scope's retain tags; untagged seeds read the whole bank.
- Mental-model defaults from `packages/coding-agent/src/config/settings-schema.ts`: `hindsight.mentalModelsEnabled = true`, `hindsight.mentalModelAutoSeed = true`, `hindsight.mentalModelRefreshIntervalMs = 5 * 60 * 1000`, `hindsight.mentalModelMaxRenderChars = 16_000`. First-turn loading waits up to `MENTAL_MODEL_FIRST_TURN_DEADLINE_MS = 1500` in `packages/coding-agent/src/hindsight/mental-models.ts`.
- Seed lifecycle is create-only. Changing `packages/coding-agent/src/hindsight/seeds.json` does not mutate existing server-side models.
- `recall.md` and `reflect.md` rely on the same bank, scoping, and mental-model bootstrap; refer back here for the shared backend behavior.
- Hindsight storage is server-side. `hindsightBackend.clear(...)` only clears local cache/state and warns that upstream deletion must happen in Hindsight UI or `deleteBank`.
- Mnemopi storage is local SQLite. `mnemopiBackend.clear(...)` removes the scoped database files for the active configuration.
- Hindsight auto-retain uses the same bank but a different path than this tool: `retainSession(...)` extracts plain user/assistant transcript, strips `<memories>` / `<mental_models>` blocks, and calls single-item `retain(...)`.
- Mnemopi auto-retain stores prepared transcripts with `source: "coding-agent-transcript"`, `importance: 0.65`, `veracity: "unknown"`, and `memoryType: "episode"`.
- Hindsight mental-model bootstrap lives in the shared backend: `HindsightSessionState.runMentalModelLoad(...)` optionally resolves seeds, creates missing models, then caches a rendered `<mental_models>` block for prompt injection.
- Built-in Hindsight seeds are `user-preferences`, `project-conventions`, and `project-decisions`. `projectTagged: true` seeds inherit the active scope's retain tags; untagged seeds read the whole bank.
- Hindsight mental-model defaults: `hindsight.mentalModelsEnabled = true`, `hindsight.mentalModelAutoSeed = true`, `hindsight.mentalModelRefreshIntervalMs = 5 * 60 * 1000`, `hindsight.mentalModelMaxRenderChars = 16_000`. First-turn loading waits up to `MENTAL_MODEL_FIRST_TURN_DEADLINE_MS = 1500`.
- Hindsight seed lifecycle is create-only. Changing `packages/coding-agent/src/hindsight/seeds.json` does not mutate existing server-side models.
- `recall.md` and `reflect.md` rely on the same backend selection and scoping behavior.
-1
View File
@@ -77,7 +77,6 @@ The returned tool result is not the final rewind. `AgentSession` waits until `tu
- `ToolError("No active checkpoint.")` — thrown when no checkpoint state is present.
- `ToolError("Report cannot be empty.")` — thrown when the trimmed report is empty.
- Missing checkpoint entry IDs during apply do not fail the tool call; `#applyRewind()` catches the error, logs `Rewind branch checkpoint missing, falling back to root`, and branches from root.
- If the agent turn is aborted while a checkpoint is active, `AgentSession` clears checkpoint state rather than applying a delayed rewind.
## Notes
- Checkpoint selection is implicit. `rewind` always targets the single `#checkpointState` captured by the last successful `checkpoint`; there is no checkpoint list, label, or ID parameter.

Some files were not shown because too many files have changed in this diff Show More