Merge branch 'main' into glm53

This commit is contained in:
Can Bölük
2026-08-15 19:36:50 +02:00
committed by GitHub
97 changed files with 2580 additions and 3883 deletions
+6 -2
View File
@@ -74,5 +74,9 @@ out.jsonl
out.html
pi-*.html
# Secrets. Should never be in an image regardless.
.env
# Host virtualenvs — the image installs its own interpreter deps.
**/.venv/
# Secrets. Should never be in an image regardless. Depth-agnostic: a bare
# `.env` misses nested ones such as `python/robomp/.env`.
**/.env
+3 -9
View File
@@ -162,11 +162,11 @@ jobs:
bazelisk --bazelrc="${{ steps.cache.outputs.rc }}" test //crates/...
# Clippy scope mirrors `cargo clippy --workspace` (libraries only, no
# test targets) plus the strict/default split: crates with
# `[lints] workspace = true` get the workspace policy, the vendored
# brush-core fork is exempt (same as run-rs-task.ts's cargo excludes).
# `[lints] workspace = true` get the workspace policy, except
# brush-core (a vendored fork excluded from the Cargo task too).
# pi-builtins allows every clippy group in its own manifest (ported
# brush/uutils/jaq code) but is still held to zero rustc warnings;
# cargo honors that via `[lints]`, bazel via the clippy-ported config.
# Cargo honors that via `[lints]`, Bazel via the clippy-ported config.
- name: Clippy (workspace lint policy on opted-in crates)
run: |
bazelisk query "kind('rust_library|rust_shared_library', //crates/pi-ast/... + //crates/pi-iso/... + //crates/pi-natives/... + //crates/pi-shell/... + //crates/pi-voice/... + //crates/pi-walker/...)" \
@@ -411,12 +411,6 @@ jobs:
- name: Test coding-agent native/unit bucket
env:
OMP_TEST_CONCURRENCY: "4"
# The mupdf/PDF-extraction chunk measures ~7 min on burstable
# runners under a full 8-wide fan-out; the default 600 s chunk
# watchdog SIGKILLed it (release run 30519992654). The watchdog
# exists to catch wedged children, not slow-but-progressing
# chunks — give this bucket a wider budget.
OMP_TEST_CHUNK_TIMEOUT: "1200"
run: bun run ci:test:coding-agent:native
test_smoke:
-1
View File
@@ -64,7 +64,6 @@ pi-*.html
# Generated files
packages/coding-agent/src/export/html/tool-views.generated.js
packages/natives/npm/
packages/coding-agent/src/utils/mupdf-wasm.wasm
/runs/
python/omp-rpc/src/omp_rpc.egg-info/
python/omp-rpc/build/
Generated
+148 -9
View File
@@ -1868,6 +1868,15 @@ version = "1.0.20"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
[[package]]
name = "ecb"
version = "0.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1a8bfa975b1aec2145850fcaa1c6fe269a16578c44705a532ae3edc92b8881c7"
dependencies = [
"cipher",
]
[[package]]
name = "ecdsa"
version = "0.16.9"
@@ -1979,6 +1988,29 @@ dependencies = [
"syn 2.0.119",
]
[[package]]
name = "env_filter"
version = "2.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "900d271a03799a1ee8d1ca9b19893b48ca674a9284fefcfb85f05e74ed314217"
dependencies = [
"log",
"regex",
]
[[package]]
name = "env_logger"
version = "0.11.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "de671bd27a75a797dc9ae289ba1e77276e75e2026408aab65185384e2d5cd3f6"
dependencies = [
"anstream",
"anstyle",
"env_filter",
"jiff",
"log",
]
[[package]]
name = "equivalent"
version = "1.0.2"
@@ -3152,6 +3184,25 @@ dependencies = [
"quick-error",
]
[[package]]
name = "include_dir"
version = "0.7.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
dependencies = [
"include_dir_macros",
]
[[package]]
name = "include_dir_macros"
version = "0.7.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
dependencies = [
"proc-macro2",
"quote",
]
[[package]]
name = "indenter"
version = "0.3.4"
@@ -3640,6 +3691,37 @@ version = "0.4.33"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
[[package]]
name = "lopdf"
version = "0.42.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
dependencies = [
"aes",
"bitflags 2.13.1",
"cbc",
"chrono",
"ecb",
"encoding_rs",
"flate2",
"getrandom 0.4.3",
"indexmap",
"itoa",
"jiff",
"log",
"md-5",
"nom 8.0.0",
"rand 0.10.2",
"rangemap",
"rayon",
"sha2",
"stringprep",
"thiserror 2.0.20",
"time",
"ttf-parser",
"weezl",
]
[[package]]
name = "lscolors"
version = "0.21.0"
@@ -4539,6 +4621,24 @@ dependencies = [
"pkg-config",
]
[[package]]
name = "pdf-inspector"
version = "1.14.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e024ae242c514e2adf6aee186678e0eabdc2e5ecfbb2159186881b4498593cb"
dependencies = [
"env_logger",
"include_dir",
"log",
"lopdf",
"once_cell",
"rayon",
"regex",
"thiserror 2.0.20",
"ttf-parser",
"unicode-normalization",
]
[[package]]
name = "peg"
version = "0.8.6"
@@ -4780,7 +4880,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "17.3.3"
version = "17.3.4"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -4849,7 +4949,7 @@ dependencies = [
[[package]]
name = "pi-builtins"
version = "17.3.3"
version = "17.3.4"
dependencies = [
"ansi-width",
"anyhow",
@@ -4934,7 +5034,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "17.3.3"
version = "17.3.4"
dependencies = [
"async-trait",
"libc",
@@ -4946,7 +5046,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "17.3.3"
version = "17.3.4"
dependencies = [
"anyhow",
"arboard",
@@ -4982,6 +5082,7 @@ dependencies = [
"objc2-core-graphics",
"objc2-foundation",
"parking_lot",
"pdf-inspector",
"phf 0.13.1",
"pi-ast",
"pi-iso",
@@ -5017,7 +5118,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "17.3.3"
version = "17.3.4"
dependencies = [
"anyhow",
"brush-core",
@@ -5045,7 +5146,7 @@ dependencies = [
[[package]]
name = "pi-voice"
version = "17.3.3"
version = "17.3.4"
dependencies = [
"audiopus_sys",
"bytes",
@@ -5060,7 +5161,7 @@ dependencies = [
[[package]]
name = "pi-walker"
version = "17.3.3"
version = "17.3.4"
dependencies = [
"dashmap",
"globset",
@@ -5134,9 +5235,9 @@ dependencies = [
[[package]]
name = "pkg-config"
version = "0.3.33"
version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "19f132c84eca552bf34cab8ec81f1c1dcc229b811638f9d283dceabe58c5569e"
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
[[package]]
name = "platform-info"
@@ -5505,6 +5606,12 @@ dependencies = [
"rand_core 0.10.1",
]
[[package]]
name = "rangemap"
version = "1.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a611d15b50743feb4c76b7d03edcb0e64f399c26961e4efe6975bc398be6aa3d"
[[package]]
name = "rayon"
version = "1.12.0"
@@ -6210,6 +6317,17 @@ dependencies = [
"quote",
]
[[package]]
name = "stringprep"
version = "0.1.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7b4df3d392d81bd458a8a621b8bffbd2302a12ffe288a9d931670948749463b1"
dependencies = [
"unicode-bidi",
"unicode-normalization",
"unicode-properties",
]
[[package]]
name = "strsim"
version = "0.11.1"
@@ -7375,12 +7493,33 @@ version = "2.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dbc4bc3a9f746d862c45cb89d705aa10f187bb96c76001afab07a0d35ce60142"
[[package]]
name = "unicode-bidi"
version = "0.3.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5c1cb5db39152898a79168971543b1cb5020dff7fe43c8dc468b0885f5e29df5"
[[package]]
name = "unicode-ident"
version = "1.0.24"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
[[package]]
name = "unicode-normalization"
version = "0.1.25"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8"
dependencies = [
"tinyvec",
]
[[package]]
name = "unicode-properties"
version = "0.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7df058c713841ad818f1dc5d3fd88063241cc61f49f5fbea4b951e8cf5a8d71d"
[[package]]
name = "unicode-segmentation"
version = "1.13.3"
+2 -1
View File
@@ -16,7 +16,7 @@ members = [
resolver = "3"
[workspace.package]
version = "17.3.3"
version = "17.3.4"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
@@ -229,6 +229,7 @@ clap = { version = "4", features = ["derive"] }
# ──────────────────────────────────────────────────────────────────────────────
# Text Processing & Parsing
# ──────────────────────────────────────────────────────────────────────────────
pdf-inspector = "1"
regex = "1"
similar = "3.1.0"
unicode-segmentation = "1.13"
+6 -2
View File
@@ -75,5 +75,9 @@ out.jsonl
out.html
pi-*.html
# Secrets. Should never be in the image regardless.
.env
# Host virtualenvs — the image installs its own interpreter deps.
**/.venv/
# Secrets. Should never be in the image regardless. Depth-agnostic: a bare
# `.env` misses `python/robomp/.env`, which `COPY . /pi/` would bake in.
**/.env
+6 -2
View File
@@ -78,8 +78,12 @@ out.jsonl
out.html
pi-*.html
# Secrets. Should never be in the image regardless.
.env
# Host virtualenvs — the image installs its own interpreter deps.
**/.venv/
# Secrets. Should never be in the image regardless. Depth-agnostic: a bare
# `.env` misses `python/robomp/.env`, which sits next to the copied src tree.
**/.env
# Robomp-only excludes. Natives + wheel + python + bun + rustup all come
# from PI_BASE; the web-builder stage only needs root manifests + the
+596 -425
View File
File diff suppressed because one or more lines are too long
+40 -44
View File
@@ -21,7 +21,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "17.3.3",
"version": "17.3.4",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -40,7 +40,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "17.3.3",
"version": "17.3.4",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/omptype": "catalog:",
@@ -63,7 +63,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "17.3.3",
"version": "17.3.4",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/omptype": "catalog:",
@@ -76,7 +76,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "17.3.3",
"version": "17.3.4",
"bin": {
"omp": "src/cli.ts",
},
@@ -105,7 +105,6 @@
"@opentelemetry/sdk-metrics": "catalog:",
"@opentelemetry/sdk-trace-base": "catalog:",
"@opentelemetry/sdk-trace-node": "catalog:",
"mupdf": "catalog:",
"puppeteer-core": "catalog:",
},
"devDependencies": {
@@ -134,7 +133,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "17.3.3",
"version": "17.3.4",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -178,7 +177,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "17.3.3",
"version": "17.3.4",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -204,7 +203,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "17.3.3",
"version": "17.3.4",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -212,7 +211,7 @@
},
"packages/omptype": {
"name": "@oh-my-pi/omptype",
"version": "17.3.3",
"version": "17.3.4",
"devDependencies": {
"@ark/attest": "0.56.3",
"@ark/schema": "0.56.2",
@@ -225,7 +224,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "17.3.3",
"version": "17.3.4",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -239,7 +238,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "17.3.3",
"version": "17.3.4",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -264,7 +263,7 @@
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "17.3.3",
"version": "17.3.4",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -300,7 +299,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "17.3.3",
"version": "17.3.4",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
},
@@ -310,7 +309,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "17.3.3",
"version": "17.3.4",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -348,19 +347,19 @@
"@bufbuild/protoc-gen-es": "^2.12.1",
"@huggingface/transformers": "^4.2.0",
"@napi-rs/cli": "3.7.2",
"@oh-my-pi/hashline": "17.3.3",
"@oh-my-pi/omp-stats": "17.3.3",
"@oh-my-pi/omptype": "17.3.3",
"@oh-my-pi/pi-agent-core": "17.3.3",
"@oh-my-pi/pi-ai": "17.3.3",
"@oh-my-pi/pi-catalog": "17.3.3",
"@oh-my-pi/pi-coding-agent": "17.3.3",
"@oh-my-pi/pi-mnemopi": "17.3.3",
"@oh-my-pi/pi-natives": "17.3.3",
"@oh-my-pi/pi-tui": "17.3.3",
"@oh-my-pi/pi-utils": "17.3.3",
"@oh-my-pi/pi-wire": "17.3.3",
"@oh-my-pi/snapcompact": "17.3.3",
"@oh-my-pi/hashline": "17.3.4",
"@oh-my-pi/omp-stats": "17.3.4",
"@oh-my-pi/omptype": "17.3.4",
"@oh-my-pi/pi-agent-core": "17.3.4",
"@oh-my-pi/pi-ai": "17.3.4",
"@oh-my-pi/pi-catalog": "17.3.4",
"@oh-my-pi/pi-coding-agent": "17.3.4",
"@oh-my-pi/pi-mnemopi": "17.3.4",
"@oh-my-pi/pi-natives": "17.3.4",
"@oh-my-pi/pi-tui": "17.3.4",
"@oh-my-pi/pi-utils": "17.3.4",
"@oh-my-pi/pi-wire": "17.3.4",
"@oh-my-pi/snapcompact": "17.3.4",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/api-logs": "^0.220.0",
"@opentelemetry/context-async-hooks": "^2.9.0",
@@ -386,7 +385,6 @@
"ghostty-web": "^0.4.0",
"lint-staged": "^17.0.8",
"lucide-react": "^1.24.0",
"mupdf": "^1.28.0",
"onnxruntime-node": "1.26.0",
"postcss": "^8.5.16",
"prettier": "^3.9.5",
@@ -455,23 +453,23 @@
"@babel/types": ["@babel/types@7.29.8", "", { "dependencies": { "@babel/helper-string-parser": "^7.29.7", "@babel/helper-validator-identifier": "^7.29.7" } }, "sha512-Vj1jF3cPfxg7OAfoI7QnVKLoILlm2JF9pnVHrX8qx7AHMiYWT+NDAA7jChlNgRS4WTLc/fD1lXLmPixluj+3Gg=="],
"@biomejs/biome": ["@biomejs/biome@2.5.7", "", { "optionalDependencies": { "@biomejs/cli-darwin-arm64": "2.5.7", "@biomejs/cli-darwin-x64": "2.5.7", "@biomejs/cli-linux-arm64": "2.5.7", "@biomejs/cli-linux-arm64-musl": "2.5.7", "@biomejs/cli-linux-x64": "2.5.7", "@biomejs/cli-linux-x64-musl": "2.5.7", "@biomejs/cli-win32-arm64": "2.5.7", "@biomejs/cli-win32-x64": "2.5.7" }, "bin": { "biome": "bin/biome" } }, "sha512-zr8K/DcY5tYsQOQwqMJ0AWElo6QgmgNI7idXgXLhevVszlt8RGVpesEJPqx3ThazLaOwjJ5Y8fz3BtH5fGZNsw=="],
"@biomejs/biome": ["@biomejs/biome@2.5.8", "", { "optionalDependencies": { "@biomejs/cli-darwin-arm64": "2.5.8", "@biomejs/cli-darwin-x64": "2.5.8", "@biomejs/cli-linux-arm64": "2.5.8", "@biomejs/cli-linux-arm64-musl": "2.5.8", "@biomejs/cli-linux-x64": "2.5.8", "@biomejs/cli-linux-x64-musl": "2.5.8", "@biomejs/cli-win32-arm64": "2.5.8", "@biomejs/cli-win32-x64": "2.5.8" }, "bin": { "biome": "bin/biome" } }, "sha512-aeAeeJB9fSDc7Gq+2GqpQxA0qBj6gj1k2R6L1cYqGePKP/baIq1WX8y6B+D+nRsO5ViQL22K/8IwbqERW0q1nw=="],
"@biomejs/cli-darwin-arm64": ["@biomejs/cli-darwin-arm64@2.5.7", "", { "os": "darwin", "cpu": "arm64" }, "sha512-vxo/Ls3/PYdQWyLhYYcgMOCzQypAjcY+iihS8M0wW03l16TCLW4zqZzGo75gm1VdCMj38hTVZ31KBWrZ4G9dJw=="],
"@biomejs/cli-darwin-arm64": ["@biomejs/cli-darwin-arm64@2.5.8", "", { "os": "darwin", "cpu": "arm64" }, "sha512-mk1QON9PHllvvLN5gU3f4rMxeh4syK5p9OvKyWH6/W8ueh04uaC8TUXXByhGufWf/y5mQc03ZLM45zU+cmqMjA=="],
"@biomejs/cli-darwin-x64": ["@biomejs/cli-darwin-x64@2.5.7", "", { "os": "darwin", "cpu": "x64" }, "sha512-Cd3Ga61amT/Yl/0x8elP5hhGYaFy4bw6WuysTgf7oo8TA5tJ5A1k+DkVoJ2BHbTVil51gTX9VPzArnrlLJ3Kyg=="],
"@biomejs/cli-darwin-x64": ["@biomejs/cli-darwin-x64@2.5.8", "", { "os": "darwin", "cpu": "x64" }, "sha512-bsGwFMBNyHPyiLSsQcZJxdoRrg1V4JL+d7wEsvUBczlP9U9lwM+7mzQHxI4o1mhBsTmdOBbAb6fHU3Z3snN45w=="],
"@biomejs/cli-linux-arm64": ["@biomejs/cli-linux-arm64@2.5.7", "", { "os": "linux", "cpu": "arm64" }, "sha512-rR2QE0yF2GYSuYuKIa7pKvODGJqnOH+2eDREAM8wV+mWKSkMQKdAp4zXEZfTaxY8PMoNONnpgSWcBCyLDPDOKg=="],
"@biomejs/cli-linux-arm64": ["@biomejs/cli-linux-arm64@2.5.8", "", { "os": "linux", "cpu": "arm64" }, "sha512-XmFiA0WPYFC+uiUDC8WRFzAIH9bo7vwQLav38Uoq4ETC+T/+uBi0TsYGJECkugY3r8USl3jc+Ae2/irAF6F2lQ=="],
"@biomejs/cli-linux-arm64-musl": ["@biomejs/cli-linux-arm64-musl@2.5.7", "", { "os": "linux", "cpu": "arm64" }, "sha512-xPI5yB6XlpDbNkS+bm1t42olw5c4l3UrlOmLg7KtLJvjvkNF/1V4tnUgfkylGIeb3u/T+BzMGYqgQhzjAoJzuQ=="],
"@biomejs/cli-linux-arm64-musl": ["@biomejs/cli-linux-arm64-musl@2.5.8", "", { "os": "linux", "cpu": "arm64" }, "sha512-VcJNbstduTHx83NGAdhp78/JOcP45BZHXL7yNsfI1uGzdUgegAz2s+mSoT7wK6PBNzLoqG0zDOXaz/RQYVtSiw=="],
"@biomejs/cli-linux-x64": ["@biomejs/cli-linux-x64@2.5.7", "", { "os": "linux", "cpu": "x64" }, "sha512-FQgqJhscrqJUFptGaRSUJWlXAExwWcDwLuK49dvKfkQ1bB5SEEyFssnsxQY83Xm6jR0EbbX3+8+D5bfvYqUG2Q=="],
"@biomejs/cli-linux-x64": ["@biomejs/cli-linux-x64@2.5.8", "", { "os": "linux", "cpu": "x64" }, "sha512-S5wcm9OBDvLHodD4PUaN488hCpco9QD/9ZxuYJiw4euWtr/oQvLR72z2ixItH8Wd5BCm6FZaeb+YNvOoM1xHtQ=="],
"@biomejs/cli-linux-x64-musl": ["@biomejs/cli-linux-x64-musl@2.5.7", "", { "os": "linux", "cpu": "x64" }, "sha512-rE5VZi+qtmPgQH+l7jVxYoZ18b/TiHEhulhMpjmCZH1PltSbjRcxNWywC3HZ9tYottG7ORkeTtoscBilKSBm0g=="],
"@biomejs/cli-linux-x64-musl": ["@biomejs/cli-linux-x64-musl@2.5.8", "", { "os": "linux", "cpu": "x64" }, "sha512-kKmiyokeISRGq2FLwvr+TzsgBusfxaZ0FZNLcOYOpCK/78tRrEjeEBLvq3xLZMpqbANgJdRPI7vZX8ZL37u9/w=="],
"@biomejs/cli-win32-arm64": ["@biomejs/cli-win32-arm64@2.5.7", "", { "os": "win32", "cpu": "arm64" }, "sha512-Oq4x0CCwP4jirrcTywXs5kOGZ4v5vuEP+gWrbtjApOA2CL9F3F9GlIdQIci8AKSCa/zURanMRpX/4wQ7Am6hHg=="],
"@biomejs/cli-win32-arm64": ["@biomejs/cli-win32-arm64@2.5.8", "", { "os": "win32", "cpu": "arm64" }, "sha512-nILH0mzm3Hi3iEdd7o7GpB8kBR/mSQwfQG/tyBqyNrY2GFtcgwfV9nV8xLmbtUpMNY/Oi0Ml1XgfR4flOdq+AA=="],
"@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.5.7", "", { "os": "win32", "cpu": "x64" }, "sha512-V+0wu/nrj2S+MhP4EQ0uHNolP0IALEsz45pg0WoKkHfDeh0+ItHwP/p7bX5RPoMOl9NkpHYWdYPhIcy2mACHvQ=="],
"@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.5.8", "", { "os": "win32", "cpu": "x64" }, "sha512-I2czzXTY61f3nFJxXoMDq80t7MivxDEnCjE+8sDKoFfcKMaoQdkqhIFQ3KyY0XLzeSpUBYeNAXgD+iOV/BU0VA=="],
"@bufbuild/protobuf": ["@bufbuild/protobuf@2.13.0", "", {}, "sha512-acq7c49vxfm1ggJ95P70TX7ABDM0vxr1SYD3BB0o0jnBLB4OAqeHyKuN+cD3w80gXEDQ2zxHpR6CUeA+O/aU9g=="],
@@ -1219,8 +1217,6 @@
"ms": ["ms@2.1.3", "", {}, "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="],
"mupdf": ["mupdf@1.28.0", "", {}, "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg=="],
"mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="],
"nanoid": ["nanoid@3.3.18", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w=="],
@@ -1301,17 +1297,17 @@
"sherpa-onnx-darwin-arm64": ["sherpa-onnx-darwin-arm64@1.13.3", "", { "os": "darwin", "cpu": "arm64" }, "sha512-9x86Cbf+BDFONdtCPM3cnjvtAW0ER8tMaHK5pVfz+SHPt8GeuwRXaiR/BzcByFBUyxCgmceO09/WMZOCi44P/g=="],
"sherpa-onnx-darwin-x64": ["sherpa-onnx-darwin-x64@1.13.4", "", { "os": "darwin", "cpu": "x64" }, "sha512-6RGeis9K9gV/UQWOgd6Rf3iqXr2/YsBQswxHaCR4hrYkHfEIpHMfFmRWLt6nJJCOWgYW2xFxEd9yzjrafAV/Pw=="],
"sherpa-onnx-darwin-x64": ["sherpa-onnx-darwin-x64@1.13.5", "", { "os": "darwin", "cpu": "x64" }, "sha512-Ay//Bq4T3RHYhzdYk9O0ahg0UoEgbRMDGZvtpQi2D0oHOcXv0M5tP3OKlx4AvWNXYl0SL+rUObUFe5IIqROrfQ=="],
"sherpa-onnx-linux-arm64": ["sherpa-onnx-linux-arm64@1.13.4", "", { "os": "linux", "cpu": "arm64" }, "sha512-RMjMRqT82BgTXypNNGmLe6ZFYhc3WEvnAGl3DdkK7qB/kuXwkL3iHhV31wAecbnWPsnEpUoD+8cFovWSBzsCuw=="],
"sherpa-onnx-linux-arm64": ["sherpa-onnx-linux-arm64@1.13.5", "", { "os": "linux", "cpu": "arm64" }, "sha512-1Jpkyv+Sg7wl+jCOlttcudVpVdlCNPXMf/RYT4+FU4/VmoMIx9MZbcSu2Nf8gtKq3mnYdkk7HtCK5AA3Dtuvwg=="],
"sherpa-onnx-linux-x64": ["sherpa-onnx-linux-x64@1.13.4", "", { "os": "linux", "cpu": "x64" }, "sha512-WZh5NCkGPFHHpYSd78iN4OnmxQeSTGyt9uZskH+im/NFHQ7elQ7B0sLzCMeRpvJxiIKvd9C6WxIJ4hYaxClfsQ=="],
"sherpa-onnx-linux-x64": ["sherpa-onnx-linux-x64@1.13.5", "", { "os": "linux", "cpu": "x64" }, "sha512-eedC/AMCL2cvwhtPnmRlLHKXuwBmD7o2r+Kb+I0krg4PlQejR9y6O1oKqdLko5vFaWe2A8/rQORKGXE+We3y4Q=="],
"sherpa-onnx-node": ["sherpa-onnx-node@1.13.2", "", { "optionalDependencies": { "sherpa-onnx-darwin-arm64": "^1.13.2", "sherpa-onnx-darwin-x64": "^1.13.2", "sherpa-onnx-linux-arm64": "^1.13.2", "sherpa-onnx-linux-x64": "^1.13.2", "sherpa-onnx-win-ia32": "^1.13.2", "sherpa-onnx-win-x64": "^1.13.2" } }, "sha512-uIH6SA5Or4pb8HlCYWB3K54XkMtzdef4/tkw1amtIf8GB1tt6hQLpur9p2jSFNfTYRyzZ8XrXofxefXQ0A7EUA=="],
"sherpa-onnx-win-ia32": ["sherpa-onnx-win-ia32@1.13.4", "", { "os": "win32", "cpu": "ia32" }, "sha512-/JbPjldrfNv+t+uIS3MlkuhfIf5l3FHUGkRC2oRXgjRqOaVmEyP3vLlQ7dTa4J7raG5oB8c3GoPjuSWSqT9GOQ=="],
"sherpa-onnx-win-ia32": ["sherpa-onnx-win-ia32@1.13.5", "", { "os": "win32", "cpu": "ia32" }, "sha512-aQKTmzsvNyCMOoAs+Clx8Zsp05JSX37hVpts/yiO3+mUTXnbBIMAL3YgG49apz6JN/l/HCJBaEczXj3odOgS8A=="],
"sherpa-onnx-win-x64": ["sherpa-onnx-win-x64@1.13.4", "", { "os": "win32", "cpu": "x64" }, "sha512-R0PWby1VxC14TDZPq7GcfSyXSY6SAFO8Y4JwdCdqouFmeXkZ1L7Is9m98C9KxQ0dN7ZtDzhAmE/43FUs/elXRQ=="],
"sherpa-onnx-win-x64": ["sherpa-onnx-win-x64@1.13.5", "", { "os": "win32", "cpu": "x64" }, "sha512-r1lKWEbFDquVjiij72nhNpmPBeqG7V06ZGx1uL2zWXFCYXmkpqawN5iHdkugVhb8QfYZ8x2tx88eYGHGMn88aA=="],
"signal-exit": ["signal-exit@4.1.0", "", {}, "sha512-bzyZ1e88w9O1iNJbKnOlvYTrWPDl46O1bG0D3XInv+9tkPrxrN8jUUTiFlDkkmKWgn1M6CfIA13SuGqOa9Korw=="],
+1
View File
@@ -44,6 +44,7 @@ inferno.workspace = true
napi.workspace = true
napi-derive.workspace = true
parking_lot.workspace = true
pdf-inspector.workspace = true
phf.workspace = true
flume.workspace = true
pi-ast.workspace = true
+41
View File
@@ -99,6 +99,43 @@ mod platform {
static _NSConcreteStackBlock: *const c_void;
}
// `SessionGetInfo` reports the caller's login-session attributes; consulted
// to detect a GUI/graphic session before touching the GUI-only DeviceCheck
// daemon.
#[link(name = "Security", kind = "framework")]
unsafe extern "C" {
fn SessionGetInfo(session: u32, session_id: *mut u32, attributes: *mut u32) -> i32;
}
/// `callerSecuritySession` — query the session hosting the current process.
const CALLER_SECURITY_SESSION: u32 = u32::MAX;
/// `sessionHasGraphicAccess` attribute bit from `Security/AuthSession.h`.
const SESSION_HAS_GRAPHIC_ACCESS: u32 = 0x0010;
/// Whether the caller's security session has GUI/graphic access.
///
/// `DCDevice.isSupported` synchronously opens an XPC connection to the
/// per-user `DeviceCheck` metadata daemon, which exists only in an
/// interactive GUI login session. From a session without graphic access
/// (SSH, a launchd `LaunchDaemon`, a CI runner, a service account, a
/// sandbox) the connection setup hits `_xpc_api_misuse` and aborts the
/// whole process with `SIGTRAP` before the completion handler can run — so
/// there is no error to return, only a dead process. Gating on the
/// documented session attribute keeps the call out of that trapping path.
///
/// Returns `false` when graphic access is absent *or* the session cannot be
/// queried: degrading to "unsupported" merely drops the attestation header
/// (exactly as on every non-macOS host), whereas optimistically assuming
/// "supported" risks the process-killing trap.
fn session_has_graphic_access() -> bool {
let mut attributes: u32 = 0;
// SAFETY: `SessionGetInfo` writes the caller session's attribute bits
// through the out-pointer; the session-id slot is unused, so it is null.
let status =
unsafe { SessionGetInfo(CALLER_SECURITY_SESSION, ptr::null_mut(), &mut attributes) };
status == 0 && attributes & SESSION_HAS_GRAPHIC_ACCESS != 0
}
/// Outcome delivered once from the completion block to the waiting worker.
enum Completion {
Token(String),
@@ -289,6 +326,10 @@ mod platform {
error: None,
latency_ms: 0.0,
};
if !session_has_graphic_access() {
result.error = Some("DeviceCheck unavailable without a GUI login session".to_owned());
return result;
}
// SAFETY: `c"DCDevice"` is a valid null-terminated class name.
let class = unsafe { objc_getClass(c"DCDevice".as_ptr()) };
if class.is_null() {
+5 -3
View File
@@ -2,7 +2,7 @@
//!
//! # Overview
//! High-performance primitives for clipboard access, grep, file discovery,
//! ANSI-aware text measurement, syntax highlighting, HTML-to-Markdown
//! ANSI-aware text measurement, syntax highlighting, HTML/PDF-to-Markdown
//! conversion, and terminal SIXEL encoding.
//!
//! # Example
@@ -15,7 +15,7 @@
//!
//! # Architecture
//! ```text
//! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/highlight/sixel/text)
//! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/pdf/highlight/sixel/text)
//! ```
#![allow(clippy::trailing_empty_array, reason = "generated by napi macro")]
@@ -41,6 +41,8 @@ pub mod html;
pub mod iofs;
pub mod keys;
pub mod live;
/// PDF inspection and Markdown conversion.
pub mod pdf;
pub mod sixel;
pub mod snapcompact;
pub use pi_ast::language;
@@ -255,7 +257,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV17_3_3")]
#[napi(js_name = "__piNativesV17_3_4")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
+165
View File
@@ -0,0 +1,165 @@
//! PDF inspection and Markdown conversion backed by `pdf-inspector`.
use napi::{Result, bindgen_prelude::Uint8Array};
use napi_derive::napi;
use pdf_inspector::{MarkdownOptions, PdfOptions, process_pdf_mem_with_options};
use crate::task;
/// Markdown and inspection metadata produced from a PDF document.
#[napi(object)]
pub struct PdfMarkdownResult {
/// Extracted document content in Markdown format.
pub markdown: String,
/// Document title from PDF metadata, when present.
pub title: Option<String>,
/// Total number of pages in the document.
pub page_count: u32,
/// One-indexed page numbers whose content requires OCR.
pub pages_needing_ocr: Vec<u32>,
/// Whether the document contains text encoding problems.
pub has_encoding_issues: bool,
}
/// Convert an in-memory PDF to Markdown and return its inspection metadata.
///
/// Conversion copies the typed array before dispatch so JavaScript mutation
/// cannot race the native worker.
///
/// # Errors
/// Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
/// be parsed or converted.
#[napi(js_name = "pdfToMarkdown")]
pub fn pdf_to_markdown(input: Uint8Array) -> task::Promise<PdfMarkdownResult> {
let input = input.to_vec();
task::blocking("pdf.to_markdown", (), move |_| convert_pdf(&input))
}
fn convert_pdf(input: &[u8]) -> Result<PdfMarkdownResult> {
let options = PdfOptions::new()
.markdown(MarkdownOptions { include_page_numbers: true, ..Default::default() });
let converted = process_pdf_mem_with_options(input, options)
.map_err(|error| napi::Error::from_reason(format!("PDF conversion failed: {error}")))?;
let markdown = match converted.markdown {
Some(markdown) => markdown,
None if !converted.pages_needing_ocr.is_empty() => String::new(),
None => {
return Err(napi::Error::from_reason(
"PDF conversion failed: converter returned no Markdown",
));
},
};
Ok(PdfMarkdownResult {
markdown,
title: converted.title,
page_count: converted.page_count,
pages_needing_ocr: converted.pages_needing_ocr,
has_encoding_issues: converted.has_encoding_issues,
})
}
#[cfg(test)]
mod tests {
use super::*;
fn pdf_fixture(page_contents: &[&str], title: Option<&str>) -> Vec<u8> {
let font_id = 3 + page_contents.len() * 2;
let info_id = title.map(|_| font_id + 1);
let mut objects = Vec::with_capacity(font_id + usize::from(info_id.is_some()));
objects.push("<< /Type /Catalog /Pages 2 0 R >>".to_string());
let kids = (0..page_contents.len())
.map(|index| format!("{} 0 R", 3 + index * 2))
.collect::<Vec<_>>()
.join(" ");
objects.push(format!("<< /Type /Pages /Kids [{kids}] /Count {} >>", page_contents.len()));
for (index, content) in page_contents.iter().enumerate() {
let content_id = 4 + index * 2;
objects.push(format!(
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 \
{font_id} 0 R >> >> /Contents {content_id} 0 R >>"
));
objects.push(format!("<< /Length {} >>\nstream\n{content}\nendstream", content.len()));
}
objects.push(
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>"
.to_string(),
);
if let Some(title) = title {
objects.push(format!("<< /Title ({title}) >>"));
}
let mut pdf = b"%PDF-1.4\n".to_vec();
let mut offsets = Vec::with_capacity(objects.len());
for (index, object) in objects.iter().enumerate() {
offsets.push(pdf.len());
pdf.extend_from_slice(format!("{} 0 obj\n{object}\nendobj\n", index + 1).as_bytes());
}
let xref_offset = pdf.len();
pdf.extend_from_slice(
format!("xref\n0 {}\n0000000000 65535 f \n", objects.len() + 1).as_bytes(),
);
for offset in offsets {
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
}
let info = info_id.map_or_else(String::new, |id| format!(" /Info {id} 0 R"));
pdf.extend_from_slice(
format!(
"trailer\n<< /Size {} /Root 1 0 R{info} >>\nstartxref\n{xref_offset}\n%%EOF\n",
objects.len() + 1
)
.as_bytes(),
);
pdf
}
#[test]
fn converts_text_title_and_page_markers() {
let pdf = pdf_fixture(
&[
"BT /F1 12 Tf 72 720 Td (First page text) Tj 0 -18 Td (More first page text) Tj 0 -18 \
Td (End first page) Tj ET",
"BT /F1 12 Tf 72 720 Td (Second page text) Tj 0 -18 Td (More second page text) Tj 0 \
-18 Td (End second page) Tj ET",
],
Some("Fixture Title"),
);
let result = convert_pdf(&pdf).expect("fixture PDF should convert");
assert_eq!(result.title.as_deref(), Some("Fixture Title"));
assert_eq!(result.page_count, 2);
assert!(result.markdown.contains("First page text"), "{}", result.markdown);
assert!(result.markdown.contains("Second page text"), "{}", result.markdown);
assert!(result.markdown.contains("<!-- Page 1 -->"), "{}", result.markdown);
assert!(result.markdown.contains("<!-- Page 2 -->"), "{}", result.markdown);
assert!(!result.has_encoding_issues);
}
#[test]
fn reports_empty_pages_as_needing_ocr() {
let pdf = pdf_fixture(&[""], None);
let result = convert_pdf(&pdf).expect("empty-page PDF should still convert");
assert_eq!(result.page_count, 1);
assert_eq!(result.pages_needing_ocr, vec![1]);
}
#[test]
fn prefixes_malformed_pdf_errors() {
let error = convert_pdf(b"not a PDF")
.err()
.expect("malformed input should fail");
assert!(
error.reason.starts_with("PDF conversion failed:"),
"unexpected error: {}",
error.reason
);
}
}
+42 -46
View File
@@ -121,41 +121,41 @@
url = "https://registry.npmjs.org/@babel/types/-/types-7.29.8.tgz";
hash = "sha512-Vj1jF3cPfxg7OAfoI7QnVKLoILlm2JF9pnVHrX8qx7AHMiYWT+NDAA7jChlNgRS4WTLc/fD1lXLmPixluj+3Gg==";
};
"@biomejs/biome@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/biome/-/biome-2.5.7.tgz";
hash = "sha512-zr8K/DcY5tYsQOQwqMJ0AWElo6QgmgNI7idXgXLhevVszlt8RGVpesEJPqx3ThazLaOwjJ5Y8fz3BtH5fGZNsw==";
"@biomejs/biome@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/biome/-/biome-2.5.8.tgz";
hash = "sha512-aeAeeJB9fSDc7Gq+2GqpQxA0qBj6gj1k2R6L1cYqGePKP/baIq1WX8y6B+D+nRsO5ViQL22K/8IwbqERW0q1nw==";
};
"@biomejs/cli-darwin-arm64@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-darwin-arm64/-/cli-darwin-arm64-2.5.7.tgz";
hash = "sha512-vxo/Ls3/PYdQWyLhYYcgMOCzQypAjcY+iihS8M0wW03l16TCLW4zqZzGo75gm1VdCMj38hTVZ31KBWrZ4G9dJw==";
"@biomejs/cli-darwin-arm64@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-darwin-arm64/-/cli-darwin-arm64-2.5.8.tgz";
hash = "sha512-mk1QON9PHllvvLN5gU3f4rMxeh4syK5p9OvKyWH6/W8ueh04uaC8TUXXByhGufWf/y5mQc03ZLM45zU+cmqMjA==";
};
"@biomejs/cli-darwin-x64@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-darwin-x64/-/cli-darwin-x64-2.5.7.tgz";
hash = "sha512-Cd3Ga61amT/Yl/0x8elP5hhGYaFy4bw6WuysTgf7oo8TA5tJ5A1k+DkVoJ2BHbTVil51gTX9VPzArnrlLJ3Kyg==";
"@biomejs/cli-darwin-x64@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-darwin-x64/-/cli-darwin-x64-2.5.8.tgz";
hash = "sha512-bsGwFMBNyHPyiLSsQcZJxdoRrg1V4JL+d7wEsvUBczlP9U9lwM+7mzQHxI4o1mhBsTmdOBbAb6fHU3Z3snN45w==";
};
"@biomejs/cli-linux-arm64-musl@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-arm64-musl/-/cli-linux-arm64-musl-2.5.7.tgz";
hash = "sha512-xPI5yB6XlpDbNkS+bm1t42olw5c4l3UrlOmLg7KtLJvjvkNF/1V4tnUgfkylGIeb3u/T+BzMGYqgQhzjAoJzuQ==";
"@biomejs/cli-linux-arm64-musl@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-arm64-musl/-/cli-linux-arm64-musl-2.5.8.tgz";
hash = "sha512-VcJNbstduTHx83NGAdhp78/JOcP45BZHXL7yNsfI1uGzdUgegAz2s+mSoT7wK6PBNzLoqG0zDOXaz/RQYVtSiw==";
};
"@biomejs/cli-linux-arm64@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-arm64/-/cli-linux-arm64-2.5.7.tgz";
hash = "sha512-rR2QE0yF2GYSuYuKIa7pKvODGJqnOH+2eDREAM8wV+mWKSkMQKdAp4zXEZfTaxY8PMoNONnpgSWcBCyLDPDOKg==";
"@biomejs/cli-linux-arm64@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-arm64/-/cli-linux-arm64-2.5.8.tgz";
hash = "sha512-XmFiA0WPYFC+uiUDC8WRFzAIH9bo7vwQLav38Uoq4ETC+T/+uBi0TsYGJECkugY3r8USl3jc+Ae2/irAF6F2lQ==";
};
"@biomejs/cli-linux-x64-musl@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-x64-musl/-/cli-linux-x64-musl-2.5.7.tgz";
hash = "sha512-rE5VZi+qtmPgQH+l7jVxYoZ18b/TiHEhulhMpjmCZH1PltSbjRcxNWywC3HZ9tYottG7ORkeTtoscBilKSBm0g==";
"@biomejs/cli-linux-x64-musl@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-x64-musl/-/cli-linux-x64-musl-2.5.8.tgz";
hash = "sha512-kKmiyokeISRGq2FLwvr+TzsgBusfxaZ0FZNLcOYOpCK/78tRrEjeEBLvq3xLZMpqbANgJdRPI7vZX8ZL37u9/w==";
};
"@biomejs/cli-linux-x64@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-x64/-/cli-linux-x64-2.5.7.tgz";
hash = "sha512-FQgqJhscrqJUFptGaRSUJWlXAExwWcDwLuK49dvKfkQ1bB5SEEyFssnsxQY83Xm6jR0EbbX3+8+D5bfvYqUG2Q==";
"@biomejs/cli-linux-x64@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-linux-x64/-/cli-linux-x64-2.5.8.tgz";
hash = "sha512-S5wcm9OBDvLHodD4PUaN488hCpco9QD/9ZxuYJiw4euWtr/oQvLR72z2ixItH8Wd5BCm6FZaeb+YNvOoM1xHtQ==";
};
"@biomejs/cli-win32-arm64@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-win32-arm64/-/cli-win32-arm64-2.5.7.tgz";
hash = "sha512-Oq4x0CCwP4jirrcTywXs5kOGZ4v5vuEP+gWrbtjApOA2CL9F3F9GlIdQIci8AKSCa/zURanMRpX/4wQ7Am6hHg==";
"@biomejs/cli-win32-arm64@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-win32-arm64/-/cli-win32-arm64-2.5.8.tgz";
hash = "sha512-nILH0mzm3Hi3iEdd7o7GpB8kBR/mSQwfQG/tyBqyNrY2GFtcgwfV9nV8xLmbtUpMNY/Oi0Ml1XgfR4flOdq+AA==";
};
"@biomejs/cli-win32-x64@2.5.7" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-win32-x64/-/cli-win32-x64-2.5.7.tgz";
hash = "sha512-V+0wu/nrj2S+MhP4EQ0uHNolP0IALEsz45pg0WoKkHfDeh0+ItHwP/p7bX5RPoMOl9NkpHYWdYPhIcy2mACHvQ==";
"@biomejs/cli-win32-x64@2.5.8" = fetchurl {
url = "https://registry.npmjs.org/@biomejs/cli-win32-x64/-/cli-win32-x64-2.5.8.tgz";
hash = "sha512-I2czzXTY61f3nFJxXoMDq80t7MivxDEnCjE+8sDKoFfcKMaoQdkqhIFQ3KyY0XLzeSpUBYeNAXgD+iOV/BU0VA==";
};
"@bufbuild/protobuf@2.13.0" = fetchurl {
url = "https://registry.npmjs.org/@bufbuild/protobuf/-/protobuf-2.13.0.tgz";
@@ -1738,10 +1738,6 @@
url = "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz";
hash = "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==";
};
"mupdf@1.28.0" = fetchurl {
url = "https://registry.npmjs.org/mupdf/-/mupdf-1.28.0.tgz";
hash = "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg==";
};
"mute-stream@3.0.0" = fetchurl {
url = "https://registry.npmjs.org/mute-stream/-/mute-stream-3.0.0.tgz";
hash = "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw==";
@@ -1927,29 +1923,29 @@
url = "https://registry.npmjs.org/sherpa-onnx-darwin-arm64/-/sherpa-onnx-darwin-arm64-1.13.3.tgz";
hash = "sha512-9x86Cbf+BDFONdtCPM3cnjvtAW0ER8tMaHK5pVfz+SHPt8GeuwRXaiR/BzcByFBUyxCgmceO09/WMZOCi44P/g==";
};
"sherpa-onnx-darwin-x64@1.13.4" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-darwin-x64/-/sherpa-onnx-darwin-x64-1.13.4.tgz";
hash = "sha512-6RGeis9K9gV/UQWOgd6Rf3iqXr2/YsBQswxHaCR4hrYkHfEIpHMfFmRWLt6nJJCOWgYW2xFxEd9yzjrafAV/Pw==";
"sherpa-onnx-darwin-x64@1.13.5" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-darwin-x64/-/sherpa-onnx-darwin-x64-1.13.5.tgz";
hash = "sha512-Ay//Bq4T3RHYhzdYk9O0ahg0UoEgbRMDGZvtpQi2D0oHOcXv0M5tP3OKlx4AvWNXYl0SL+rUObUFe5IIqROrfQ==";
};
"sherpa-onnx-linux-arm64@1.13.4" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-linux-arm64/-/sherpa-onnx-linux-arm64-1.13.4.tgz";
hash = "sha512-RMjMRqT82BgTXypNNGmLe6ZFYhc3WEvnAGl3DdkK7qB/kuXwkL3iHhV31wAecbnWPsnEpUoD+8cFovWSBzsCuw==";
"sherpa-onnx-linux-arm64@1.13.5" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-linux-arm64/-/sherpa-onnx-linux-arm64-1.13.5.tgz";
hash = "sha512-1Jpkyv+Sg7wl+jCOlttcudVpVdlCNPXMf/RYT4+FU4/VmoMIx9MZbcSu2Nf8gtKq3mnYdkk7HtCK5AA3Dtuvwg==";
};
"sherpa-onnx-linux-x64@1.13.4" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-linux-x64/-/sherpa-onnx-linux-x64-1.13.4.tgz";
hash = "sha512-WZh5NCkGPFHHpYSd78iN4OnmxQeSTGyt9uZskH+im/NFHQ7elQ7B0sLzCMeRpvJxiIKvd9C6WxIJ4hYaxClfsQ==";
"sherpa-onnx-linux-x64@1.13.5" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-linux-x64/-/sherpa-onnx-linux-x64-1.13.5.tgz";
hash = "sha512-eedC/AMCL2cvwhtPnmRlLHKXuwBmD7o2r+Kb+I0krg4PlQejR9y6O1oKqdLko5vFaWe2A8/rQORKGXE+We3y4Q==";
};
"sherpa-onnx-node@1.13.2" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-node/-/sherpa-onnx-node-1.13.2.tgz";
hash = "sha512-uIH6SA5Or4pb8HlCYWB3K54XkMtzdef4/tkw1amtIf8GB1tt6hQLpur9p2jSFNfTYRyzZ8XrXofxefXQ0A7EUA==";
};
"sherpa-onnx-win-ia32@1.13.4" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-win-ia32/-/sherpa-onnx-win-ia32-1.13.4.tgz";
hash = "sha512-/JbPjldrfNv+t+uIS3MlkuhfIf5l3FHUGkRC2oRXgjRqOaVmEyP3vLlQ7dTa4J7raG5oB8c3GoPjuSWSqT9GOQ==";
"sherpa-onnx-win-ia32@1.13.5" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-win-ia32/-/sherpa-onnx-win-ia32-1.13.5.tgz";
hash = "sha512-aQKTmzsvNyCMOoAs+Clx8Zsp05JSX37hVpts/yiO3+mUTXnbBIMAL3YgG49apz6JN/l/HCJBaEczXj3odOgS8A==";
};
"sherpa-onnx-win-x64@1.13.4" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-win-x64/-/sherpa-onnx-win-x64-1.13.4.tgz";
hash = "sha512-R0PWby1VxC14TDZPq7GcfSyXSY6SAFO8Y4JwdCdqouFmeXkZ1L7Is9m98C9KxQ0dN7ZtDzhAmE/43FUs/elXRQ==";
"sherpa-onnx-win-x64@1.13.5" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx-win-x64/-/sherpa-onnx-win-x64-1.13.5.tgz";
hash = "sha512-r1lKWEbFDquVjiij72nhNpmPBeqG7V06ZGx1uL2zWXFCYXmkpqawN5iHdkugVhb8QfYZ8x2tx88eYGHGMn88aA==";
};
"sherpa-onnx@1.13.2" = fetchurl {
url = "https://registry.npmjs.org/sherpa-onnx/-/sherpa-onnx-1.13.2.tgz";
+13 -16
View File
@@ -23,19 +23,19 @@
"@bufbuild/protoc-gen-es": "^2.12.1",
"@huggingface/transformers": "^4.2.0",
"@napi-rs/cli": "3.7.2",
"@oh-my-pi/hashline": "17.3.3",
"@oh-my-pi/omp-stats": "17.3.3",
"@oh-my-pi/omptype": "17.3.3",
"@oh-my-pi/pi-agent-core": "17.3.3",
"@oh-my-pi/pi-ai": "17.3.3",
"@oh-my-pi/pi-catalog": "17.3.3",
"@oh-my-pi/pi-coding-agent": "17.3.3",
"@oh-my-pi/pi-mnemopi": "17.3.3",
"@oh-my-pi/pi-natives": "17.3.3",
"@oh-my-pi/pi-tui": "17.3.3",
"@oh-my-pi/pi-utils": "17.3.3",
"@oh-my-pi/pi-wire": "17.3.3",
"@oh-my-pi/snapcompact": "17.3.3",
"@oh-my-pi/hashline": "17.3.4",
"@oh-my-pi/omp-stats": "17.3.4",
"@oh-my-pi/omptype": "17.3.4",
"@oh-my-pi/pi-agent-core": "17.3.4",
"@oh-my-pi/pi-ai": "17.3.4",
"@oh-my-pi/pi-catalog": "17.3.4",
"@oh-my-pi/pi-coding-agent": "17.3.4",
"@oh-my-pi/pi-mnemopi": "17.3.4",
"@oh-my-pi/pi-natives": "17.3.4",
"@oh-my-pi/pi-tui": "17.3.4",
"@oh-my-pi/pi-utils": "17.3.4",
"@oh-my-pi/pi-wire": "17.3.4",
"@oh-my-pi/snapcompact": "17.3.4",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/api-logs": "^0.220.0",
"@opentelemetry/context-async-hooks": "^2.9.0",
@@ -61,7 +61,6 @@
"ghostty-web": "^0.4.0",
"lint-staged": "^17.0.8",
"lucide-react": "^1.24.0",
"mupdf": "^1.28.0",
"onnxruntime-node": "1.26.0",
"postcss": "^8.5.16",
"prettier": "^3.9.5",
@@ -170,8 +169,6 @@
"gen:nix": "bun scripts/gen-nix-bun.ts",
"gen:tool-views": "bun --cwd=packages/collab-web run gen:tool-views",
"gen:bundle": "bun --cwd=packages/coding-agent run gen:bundle",
"gen:mupdf": "bun --cwd=packages/coding-agent run gen:mupdf",
"gen:mupdf:reset": "bun --cwd=packages/coding-agent run gen:mupdf:reset",
"gen:native": "bun --cwd=packages/natives run gen:native",
"gen:native:reset": "bun --cwd=packages/natives run gen:native:reset",
"check-spoofed-versions": "bun scripts/check-spoofed-versions.ts"
+6
View File
@@ -2,6 +2,12 @@
## [Unreleased]
## [17.3.4] - 2026-08-14
### Fixed
- Fixed Codex-compatible V2 remote compaction with an explicit `v2Endpoint` by sending the required feature-negotiation header ([#8524](https://github.com/can1357/oh-my-pi/issues/8524)).
## [17.3.0] - 2026-08-13
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "17.3.3",
"version": "17.3.4",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
@@ -423,6 +423,7 @@ function buildCompactionV2Headers(
}
headers[OPENAI_HEADERS.BETA] = OPENAI_HEADER_VALUES.BETA_RESPONSES;
headers[OPENAI_HEADERS.ORIGINATOR] = OPENAI_HEADER_VALUES.ORIGINATOR_CODEX;
headers[OPENAI_HEADERS.CODEX_BETA_FEATURES] = OPENAI_HEADER_VALUES.REMOTE_COMPACTION_V2;
if (model.useResponsesLite) {
headers[OPENAI_HEADERS.RESPONSES_LITE] = "true";
}
@@ -745,6 +745,7 @@ describe("requestCompactionV2Streaming", () => {
let sessionHeader: string | undefined;
let clientRequestHeader: string | undefined;
let legacySessionHeader: string | undefined;
let betaFeaturesHeader: string | undefined;
const fetchMock: FetchImpl = async (input, init) => {
expect(String(input)).toBe("https://compact.example/v1/responses");
if (!init?.headers || init.headers instanceof Headers || Array.isArray(init.headers)) {
@@ -753,9 +754,11 @@ describe("requestCompactionV2Streaming", () => {
const rawSessionHeader = init.headers.session_id;
const rawClientRequestHeader = init.headers["x-client-request-id"];
const rawLegacySessionHeader = init.headers["session-id"];
const rawBetaFeaturesHeader = init.headers["x-codex-beta-features"];
sessionHeader = typeof rawSessionHeader === "string" ? rawSessionHeader : undefined;
clientRequestHeader = typeof rawClientRequestHeader === "string" ? rawClientRequestHeader : undefined;
legacySessionHeader = typeof rawLegacySessionHeader === "string" ? rawLegacySessionHeader : undefined;
betaFeaturesHeader = typeof rawBetaFeaturesHeader === "string" ? rawBetaFeaturesHeader : undefined;
requestBody = JSON.parse(String(init.body)) as {
model: string;
input: Array<Record<string, unknown>>;
@@ -789,6 +792,7 @@ describe("requestCompactionV2Streaming", () => {
expect(sessionHeader).toBe("session-1");
expect(clientRequestHeader).toBe("session-1");
expect(legacySessionHeader).toBeUndefined();
expect(betaFeaturesHeader).toBeUndefined();
expect(requestBody?.model).toBe("gpt-5-compact");
expect(requestBody?.prompt_cache_key).toBe("cache-1");
expect(requestBody?.input[requestBody.input.length - 1]).toEqual({ type: "compaction_trigger" });
@@ -798,6 +802,52 @@ describe("requestCompactionV2Streaming", () => {
expect(result.usage?.reasoningOutputTokens).toBe(1);
});
test("negotiates Codex V2 compaction for an explicit Responses endpoint", async () => {
const model = buildModel({
id: "gpt-5",
name: "GPT-5",
api: "openai-responses",
provider: "openai",
baseUrl: "https://api.openai.com/v1",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 400000,
maxTokens: 128000,
remoteCompaction: {
enabled: true,
api: "openai-codex-responses",
v2StreamingEnabled: true,
v2Endpoint: "https://compact.example/v1/responses",
},
});
const request = buildCompactionV2Request(
model,
[{ type: "message", role: "user", content: [{ type: "input_text", text: "real user" }] }],
"instructions",
);
let betaFeaturesHeader: string | undefined;
const fetchMock: FetchImpl = async (_input, init) => {
if (!init?.headers || init.headers instanceof Headers || Array.isArray(init.headers)) {
throw new Error("Expected V2 compaction to send headers as a plain object");
}
const rawBetaFeaturesHeader = init.headers["x-codex-beta-features"];
betaFeaturesHeader = typeof rawBetaFeaturesHeader === "string" ? rawBetaFeaturesHeader : undefined;
return sseResponse([
{
type: "response.output_item.done",
output_index: 0,
item: { type: "compaction", encrypted_content: "enc" },
},
{ type: "response.completed", response: { usage: { input_tokens: 1, output_tokens: 1, total_tokens: 2 } } },
]);
};
await requestCompactionV2Streaming(model, "test-key", request, undefined, { fetch: fetchMock });
expect(betaFeaturesHeader).toBe("remote_compaction_v2");
});
test("retries transient V2 stream failures with a fresh request attempt", async () => {
const model = makeOpenAiModel({
remoteCompaction: {
+4 -1
View File
@@ -2,9 +2,12 @@
## [Unreleased]
## [17.3.4] - 2026-08-14
### Fixed
- Fixed `omp usage invalidate` to discard stale OAuth and API-key usage snapshots, then force a cache-bypassing, per-provider serialized refresh so upgraded subscriptions do not silently retain pre-change quota data.
- Fixed `omp usage invalidate` to discard stale OAuth and API-key usage snapshots, then force a cache-bypassing, per-provider serialized refresh with a broker request budget sized for the full unfiltered account batch, so upgraded subscriptions do not silently retain pre-change quota data.
- Fixed quota reporting and Cookie capture guidance for China (Beijing) Alibaba Token Plan credentials ([#8509](https://github.com/can1357/oh-my-pi/issues/8509)).
## [17.3.3] - 2026-08-14
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "17.3.3",
"version": "17.3.4",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+24 -7
View File
@@ -249,12 +249,23 @@ export class AuthBrokerClient {
}
}
fetchUsage(signal?: AbortSignal): Promise<UsageResponse> {
// Validates the envelope (`generatedAt`, `reports[].provider`, `limits`,
// `metadata`) but leaves provider-specific extension fields permissive so
// the broker can ship new shapes ahead of the client. `raw` is accepted
// but normally stripped by the broker before send.
return this.#request<UsageResponse>("GET", "/v1/usage", { schema: "usageResponseSchema", signal });
/**
* Fetch aggregate broker usage with a timeout sized for serialized
* same-provider account probes.
*/
fetchUsage(options: { signal?: AbortSignal; maxAccountsPerProvider?: number } = {}): Promise<UsageResponse> {
const requestedAccountCount = options.maxAccountsPerProvider;
const accountCount =
typeof requestedAccountCount === "number" && Number.isFinite(requestedAccountCount)
? Math.max(1, Math.floor(requestedAccountCount))
: 1;
const perAccountTimeoutMs = Math.max(DEFAULT_TIMEOUT_MS, this.#timeoutMs);
const timeoutMs = perAccountTimeoutMs * (accountCount + 1);
return this.#request<UsageResponse>("GET", "/v1/usage", {
schema: "usageResponseSchema",
signal: options.signal,
timeoutMs,
});
}
/** Recorded usage-limit snapshots from the broker host, oldest first. */
@@ -369,7 +380,13 @@ export class AuthBrokerClient {
async #request<t>(
method: "GET" | "POST" | "DELETE",
path: string,
opts: { schema: AuthBrokerResponseSchemaName; auth?: boolean; body?: unknown; signal?: AbortSignal },
opts: {
schema: AuthBrokerResponseSchemaName;
auth?: boolean;
body?: unknown;
signal?: AbortSignal;
timeoutMs?: number;
},
): Promise<t> {
const response = await this.#fetchRaw(method, path, opts);
const text = await response.text();
+48 -3
View File
@@ -253,6 +253,9 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
#usageInflight?: Promise<UsageReport[] | null>;
#credentialBlockReconcileAfter: Map<string, number> = new Map();
#usageCacheEpoch = 0;
/** Raw broker credentials retained to size aggregate usage requests before account-pool filtering. */
#brokerUsageProviderByCredentialId = new Map<number, Provider>();
#brokerUsageAccountCounts = new Map<Provider, number>();
/** Per-snapshot lookup of oauth credentials by provider; rebuilt when `#snapshot` is replaced. */
#usageFilterLookup?: { snapshot: SnapshotResponse; byProvider: Map<Provider, OAuthCredential[]> };
/** Memoized `#filterUsageReports` output, keyed on (input identity, lookup identity). */
@@ -296,6 +299,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
#applySnapshot(snapshot: SnapshotResponse, generation: number, protectNewBlocks = true): void {
const nowMs = Date.now();
this.#replaceBrokerUsageAccounts(snapshot.credentials);
const previousCredentials = this.#snapshot.credentials;
const credentials = snapshot.credentials
.filter(entry => isCredentialInAccountPool(entry, this.#accountPool))
@@ -427,8 +431,9 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
generation: number,
serverNowMs: number,
): void {
this.#upsertBrokerUsageAccount(entry);
if (!isCredentialInAccountPool(entry, this.#accountPool)) {
this.#removeStreamCredential(entry.id, refresher, generation, serverNowMs);
this.#removeStreamCredential(entry.id, refresher, generation, serverNowMs, { retainBrokerUsageAccount: true });
return;
}
const incoming = this.#normalizeSnapshotEntryBlocks(entry, Date.now());
@@ -446,7 +451,14 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
this.#snapshotReceivedAt = Date.now();
}
#removeStreamCredential(id: number, refresher: RefresherSchedule, generation: number, serverNowMs: number): void {
#removeStreamCredential(
id: number,
refresher: RefresherSchedule,
generation: number,
serverNowMs: number,
options?: { retainBrokerUsageAccount?: boolean },
): void {
if (!options?.retainBrokerUsageAccount) this.#removeBrokerUsageAccount(id);
const removed = this.#snapshot.credentials.find(entry => entry.id === id);
if (removed?.blocks && removed.blocks.length > 0) this.#invalidateUsageCache();
const credentials = this.#snapshot.credentials.filter(entry => entry.id !== id);
@@ -1066,6 +1078,39 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
});
}
#replaceBrokerUsageAccounts(entries: readonly SnapshotEntry[]): void {
this.#brokerUsageProviderByCredentialId.clear();
this.#brokerUsageAccountCounts.clear();
for (const entry of entries) this.#upsertBrokerUsageAccount(entry);
}
#upsertBrokerUsageAccount(entry: Pick<SnapshotEntry, "id" | "provider">): void {
const previous = this.#brokerUsageProviderByCredentialId.get(entry.id);
if (previous === entry.provider) return;
if (previous !== undefined) {
const count = this.#brokerUsageAccountCounts.get(previous) ?? 0;
if (count <= 1) this.#brokerUsageAccountCounts.delete(previous);
else this.#brokerUsageAccountCounts.set(previous, count - 1);
}
this.#brokerUsageProviderByCredentialId.set(entry.id, entry.provider);
this.#brokerUsageAccountCounts.set(entry.provider, (this.#brokerUsageAccountCounts.get(entry.provider) ?? 0) + 1);
}
#removeBrokerUsageAccount(id: number): void {
const provider = this.#brokerUsageProviderByCredentialId.get(id);
if (provider === undefined) return;
this.#brokerUsageProviderByCredentialId.delete(id);
const count = this.#brokerUsageAccountCounts.get(provider) ?? 0;
if (count <= 1) this.#brokerUsageAccountCounts.delete(provider);
else this.#brokerUsageAccountCounts.set(provider, count - 1);
}
#maxBrokerUsageAccounts(): number {
let maximum = 1;
for (const count of this.#brokerUsageAccountCounts.values()) maximum = Math.max(maximum, count);
return maximum;
}
#loadUsageReports(): Promise<UsageReport[] | null> {
const cached = this.#usageCache;
if (cached && Date.now() - cached.fetchedAt < USAGE_CACHE_TTL_MS) {
@@ -1074,7 +1119,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore {
if (this.#usageInflight) return this.#usageInflight;
const epoch = this.#usageCacheEpoch;
const inflight = this.#client
.fetchUsage()
.fetchUsage({ maxAccountsPerProvider: this.#maxBrokerUsageAccounts() })
.then(body => {
if (epoch !== this.#usageCacheEpoch) return this.#loadUsageReports();
this.#usageCache = { reports: body.reports, fetchedAt: Date.now() };
+10 -8
View File
@@ -3274,9 +3274,10 @@ export class AuthStorage {
async #fetchUsageCached(
request: UsageRequestDescriptor,
timeoutMs?: number,
forceRefresh = false,
options: { timeoutMs?: number; forceRefresh?: boolean } = {},
): Promise<UsageReport | null> {
const timeoutMs = options.timeoutMs;
const forceRefresh = options.forceRefresh ?? false;
const cacheKey = this.#buildUsageReportCacheKey(request);
const now = Date.now();
const cached = forceRefresh ? undefined : this.#usageCache.get<UsageReport | null>(cacheKey);
@@ -3797,10 +3798,9 @@ export class AuthStorage {
if (!resolvedApiKey) return null;
usageCredential.apiKey = resolvedApiKey;
}
return this.#fetchUsageCached(
this.#buildUsageRequest(provider, usageCredential, options?.baseUrl),
options?.timeoutMs ?? this.#usageRequestTimeoutMs,
);
return this.#fetchUsageCached(this.#buildUsageRequest(provider, usageCredential, options?.baseUrl), {
timeoutMs: options?.timeoutMs ?? this.#usageRequestTimeoutMs,
});
}
/**
@@ -4007,10 +4007,12 @@ export class AuthStorage {
requests.map(request => {
const forceRefresh = serializedProviders.has(request.provider);
if (!forceRefresh) {
return this.#fetchUsageCached(request, this.#usageRequestTimeoutMs);
return this.#fetchUsageCached(request, { timeoutMs: this.#usageRequestTimeoutMs });
}
const tail = tails.get(request.provider) ?? Promise.resolve();
const current = tail.then(() => this.#fetchUsageCached(request, this.#usageRequestTimeoutMs, true));
const current = tail.then(() =>
this.#fetchUsageCached(request, { timeoutMs: this.#usageRequestTimeoutMs, forceRefresh: true }),
);
tails.set(
request.provider,
current.then(
@@ -86,9 +86,13 @@ export async function loginAlibabaTokenPlan(options: OAuthController): Promise<s
fetch: options.fetch,
});
const cookieRequestHost =
baseUrl === ALIBABA_TOKEN_PLAN_CN_BASE_URL ? "bailian-cs.console.aliyun.com" : "cs-data.qwencloud.com";
const rawCookie = await options.onPrompt({
message:
"Optional quota reporting: open browser DevTools → Network, reload the Token Plan page, filter for api.json, and select the cs-data.qwencloud.com/data/api.json request whose api query ends in /tokenplan/personal/api/v2/usage. Copy Request Headers → Cookie, then paste the complete name=value; ... value here, or press Enter to skip.",
baseUrl === ALIBABA_TOKEN_PLAN_CN_BASE_URL
? "Optional quota reporting: open browser DevTools → Network, reload the Token Plan page, filter for api.json, and select the bailian-cs.console.aliyun.com/data/api.json request whose api query ends in /tokenplan/personal/api/v2/usage. Copy Request Headers → Cookie, then paste the complete name=value; ... value here, or press Enter to skip."
: "Optional quota reporting: open browser DevTools → Network, reload the Token Plan page, filter for api.json, and select the cs-data.qwencloud.com/data/api.json request whose api query ends in /tokenplan/personal/api/v2/usage. Copy Request Headers → Cookie, then paste the complete name=value; ... value here, or press Enter to skip.",
placeholder: "name=value; name=value; ...",
allowEmpty: true,
});
@@ -107,7 +111,7 @@ export async function loginAlibabaTokenPlan(options: OAuthController): Promise<s
})
) {
throw new AIError.ConfigurationError(
"Invalid QwenCloud Cookie header. Copy the complete Cookie request header from the cs-data.qwencloud.com usage request, not a single cookie value.",
`Invalid QwenCloud Cookie header. Copy the complete Cookie request header from the ${cookieRequestHost} usage request, not a single cookie value.`,
);
}
@@ -5,10 +5,10 @@ import { scheduler } from "node:timers/promises";
import { getBundledModels } from "@oh-my-pi/pi-catalog/models";
import {
COPILOT_API_HEADERS,
discoverGitHubCopilotApiEndpoint,
getGitHubCopilotBaseUrl,
isPublicGitHubHost,
normalizeDomain,
normalizeGitHubCopilotApiEndpoint,
normalizeGitHubCopilotEnterpriseDomain,
OPENCODE_HEADERS,
} from "@oh-my-pi/pi-catalog/wire/github-copilot";
@@ -226,27 +226,6 @@ export function refreshGitHubCopilotToken(
};
}
async function discoverGitHubCopilotApiEndpoint(token: string, fetchImpl: FetchImpl): Promise<string | undefined> {
try {
const data = await fetchJson(
"https://api.github.com/copilot_internal/user",
{
headers: {
Accept: "application/json",
Authorization: `token ${token}`,
...OPENCODE_HEADERS,
},
},
fetchImpl,
);
if (!data || typeof data !== "object") return undefined;
const endpoints = (data as { endpoints?: { api?: unknown } }).endpoints;
return typeof endpoints?.api === "string" ? normalizeGitHubCopilotApiEndpoint(endpoints.api) : undefined;
} catch {
return undefined;
}
}
/**
* Enable a model for the user's GitHub Copilot account.
* This is required for some models (like Claude, Grok) before they can be used.
+89 -34
View File
@@ -1,5 +1,8 @@
import { toNumber } from "@oh-my-pi/pi-catalog/utils";
import { parseAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import {
ALIBABA_TOKEN_PLAN_CN_BASE_URL,
parseAlibabaTokenPlanCredential,
} from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import type {
CredentialRankingStrategy,
UsageFetchContext,
@@ -12,21 +15,45 @@ import { isRecord } from "../utils";
import { HOUR_MS, parsePositiveTimestamp, WEEK_MS } from "./shared";
const PROVIDER = "alibaba-token-plan";
const CONSOLE_ORIGIN = "https://home.qwencloud.com";
const DASHBOARD_URL = `${CONSOLE_ORIGIN}/billing/subscription/token-plan-individual`;
const USER_INFO_URL = `${CONSOLE_ORIGIN}/tool/user/info.json`;
const GATEWAY_ACTION = "IntlBroadScopeAspnGateway";
const USAGE_API = "zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/usage";
const USAGE_URL = `https://cs-data.qwencloud.com/data/api.json?product=sfm_bailian&action=${GATEWAY_ACTION}&api=${encodeURIComponent(USAGE_API)}`;
const BROWSER_USER_AGENT =
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/143.0.0.0 Safari/537.36";
const CONSOLE_CORNERSTONE_PARAM = {
domain: "home.qwencloud.com",
consoleSite: "QWENCLOUD",
console: "ONE_CONSOLE",
xsp_lang: "en-US",
protocol: "V2",
productCode: "p_efm",
const INTERNATIONAL_CONSOLE = {
origin: "https://home.qwencloud.com",
dashboardUrl: "https://home.qwencloud.com/billing/subscription/token-plan-individual",
sessionUrl: "https://home.qwencloud.com/tool/user/info.json",
gatewayAction: "IntlBroadScopeAspnGateway",
region: "ap-southeast-1",
usageUrl: `https://cs-data.qwencloud.com/data/api.json?product=sfm_bailian&action=IntlBroadScopeAspnGateway&api=${encodeURIComponent(USAGE_API)}`,
cornerstoneParam: {
domain: "home.qwencloud.com",
consoleSite: "QWENCLOUD",
console: "ONE_CONSOLE",
xsp_lang: "en-US",
protocol: "V2",
productCode: "p_efm",
},
} as const;
const CHINA_CONSOLE = {
origin: "https://bailian.console.aliyun.com",
dashboardUrl: "https://bailian.console.aliyun.com/cn-beijing?tab=plan",
sessionUrl: "https://bailian.console.aliyun.com/cn-beijing?tab=plan",
gatewayAction: "BroadScopeAspnGateway",
region: "cn-beijing",
usageUrl: `https://bailian-cs.console.aliyun.com/data/api.json?action=BroadScopeAspnGateway&product=sfm_bailian&api=${encodeURIComponent(USAGE_API)}`,
cornerstoneParam: {
feURL: "https://bailian.console.aliyun.com/cn-beijing?tab=plan#/efm/subscription/token-plan/personal",
protocol: "V2",
console: "ONE_CONSOLE",
productCode: "p_efm",
switchAgent: 12608464,
switchUserType: 3,
domain: "bailian.console.aliyun.com",
consoleSite: "BAILIAN_ALIYUN",
userNickName: "",
userPrincipalName: "",
xsp_lang: "zh-CN",
},
} as const;
function extractCookieValue(header: string, name: string): string | undefined {
@@ -102,35 +129,56 @@ async function fetchAlibabaTokenPlanUsage(
const credential = parseAlibabaTokenPlanCredential(params.credential.apiKey);
if (!credential?.cookie) return null;
const cookie = credential.cookie;
const isChina = credential.baseUrl === ALIBABA_TOKEN_PLAN_CN_BASE_URL;
const consoleConfig = isChina ? CHINA_CONSOLE : INTERNATIONAL_CONSOLE;
try {
const userResponse = await ctx.fetch(USER_INFO_URL, {
const sessionResponse = await ctx.fetch(consoleConfig.sessionUrl, {
headers: {
Accept: "application/json, text/plain, */*",
Accept: isChina
? "text/html,application/xhtml+xml,application/json;q=0.9,*/*;q=0.8"
: "application/json, text/plain, */*",
Cookie: cookie,
Referer: `${CONSOLE_ORIGIN}/`,
Referer: `${consoleConfig.origin}/`,
"User-Agent": BROWSER_USER_AGENT,
},
redirect: "manual",
signal: params.signal,
});
if (!userResponse.ok) {
ctx.logger?.warn("QwenCloud session lookup failed", { provider: PROVIDER, status: userResponse.status });
if (!sessionResponse.ok) {
ctx.logger?.warn("Alibaba Token Plan session lookup failed", {
provider: PROVIDER,
status: sessionResponse.status,
});
return null;
}
const userPayload: unknown = await userResponse.json();
if (!isRecord(userPayload) || !isRecord(userPayload.data) || typeof userPayload.data.secToken !== "string") {
ctx.logger?.warn("QwenCloud session response invalid", { provider: PROVIDER });
return null;
let secToken: string | undefined;
let accountId: string | undefined;
if (isChina) {
const html = await sessionResponse.text();
secToken = /\bSEC_TOKEN\s*:\s*"([^"]+)"/.exec(html)?.[1];
if (!secToken) {
ctx.logger?.warn("Alibaba Token Plan China session response invalid", { provider: PROVIDER });
return null;
}
} else {
const userPayload: unknown = await sessionResponse.json();
if (!isRecord(userPayload) || !isRecord(userPayload.data) || typeof userPayload.data.secToken !== "string") {
ctx.logger?.warn("QwenCloud session response invalid", { provider: PROVIDER });
return null;
}
secToken = userPayload.data.secToken;
accountId = accountIdFromUserData(userPayload.data);
}
const secToken = userPayload.data.secToken;
const csrf = extractCookieValue(cookie, "login_aliyunid_csrf") ?? extractCookieValue(cookie, "csrf");
const headers: Record<string, string> = {
Accept: "application/json, text/plain, */*",
"Content-Type": "application/x-www-form-urlencoded",
Cookie: cookie,
Origin: CONSOLE_ORIGIN,
Referer: DASHBOARD_URL,
Origin: consoleConfig.origin,
Referer: consoleConfig.dashboardUrl,
"User-Agent": BROWSER_USER_AGENT,
"X-Requested-With": "XMLHttpRequest",
};
@@ -140,16 +188,21 @@ async function fetchAlibabaTokenPlanUsage(
}
const body = new URLSearchParams({
product: "sfm_bailian",
action: GATEWAY_ACTION,
region: "ap-southeast-1",
action: consoleConfig.gatewayAction,
region: consoleConfig.region,
sec_token: secToken,
params: JSON.stringify({
Api: USAGE_API,
Data: { cornerstoneParam: CONSOLE_CORNERSTONE_PARAM },
Data: {
cornerstoneParam: {
...(isChina ? { feTraceId: crypto.randomUUID() } : {}),
...consoleConfig.cornerstoneParam,
},
},
V: "1.0",
}),
});
const usageResponse = await ctx.fetch(USAGE_URL, {
const usageResponse = await ctx.fetch(consoleConfig.usageUrl, {
method: "POST",
headers,
body,
@@ -157,16 +210,18 @@ async function fetchAlibabaTokenPlanUsage(
signal: params.signal,
});
if (!usageResponse.ok) {
ctx.logger?.warn("QwenCloud usage fetch failed", { provider: PROVIDER, status: usageResponse.status });
ctx.logger?.warn("Alibaba Token Plan usage fetch failed", {
provider: PROVIDER,
status: usageResponse.status,
});
return null;
}
const payload: unknown = await usageResponse.json();
if (!isRecord(payload) || payload.successResponse === false || !isRecord(payload.data)) {
ctx.logger?.warn("QwenCloud usage response invalid", { provider: PROVIDER });
ctx.logger?.warn("Alibaba Token Plan usage response invalid", { provider: PROVIDER });
return null;
}
const responseData = unwrapGatewayData(payload.data);
const accountId = accountIdFromUserData(userPayload.data);
const limits = [
buildLimit(
"5h",
@@ -190,10 +245,10 @@ async function fetchAlibabaTokenPlanUsage(
provider: PROVIDER,
fetchedAt: Date.now(),
limits,
metadata: { source: "qwencloud-console", ...(accountId ? { accountId } : {}) },
metadata: { source: isChina ? "bailian-console" : "qwencloud-console", ...(accountId ? { accountId } : {}) },
};
} catch (error) {
ctx.logger?.warn("QwenCloud usage request failed", {
ctx.logger?.warn("Alibaba Token Plan usage request failed", {
provider: PROVIDER,
error: error instanceof Error ? error.name : "unknown",
});
@@ -5,7 +5,10 @@ import {
alibabaTokenPlanRankingStrategy,
alibabaTokenPlanUsageProvider,
} from "@oh-my-pi/pi-ai/usage/alibaba-token-plan";
import { serializeAlibabaTokenPlanCredential } from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
import {
ALIBABA_TOKEN_PLAN_CN_BASE_URL,
serializeAlibabaTokenPlanCredential,
} from "@oh-my-pi/pi-catalog/wire/alibaba-token-plan";
function params(apiKey: string): UsageFetchParams {
return {
@@ -102,6 +105,84 @@ describe("QwenCloud Token Plan opt-in usage", () => {
expect(windows.secondary?.id).toBe("credits:7d");
});
test("fetches China quota through the Beijing console gateway", async () => {
const requests: { url: string; init?: RequestInit }[] = [];
const fetchMock: FetchImpl = (input, init) => {
requests.push({ url: String(input), init });
if (requests.length === 1) {
return Promise.resolve(
new Response('<script>window.ALIYUN_CONSOLE_CONFIG = { SEC_TOKEN: "cn-sec-token" };</script>'),
);
}
return Promise.resolve(
Response.json({
code: "200",
data: {
DataV2: {
data: {
data: {
per1WeekPercentage: 0.7913113,
per1WeekResetTime: 1_786_716_480_000,
},
},
},
},
successResponse: true,
}),
);
};
const cookie = "login_aliyunid_csrf=cn-csrf; aliyun_lang=zh";
const credential = serializeAlibabaTokenPlanCredential("sk-sp-beijing", cookie, ALIBABA_TOKEN_PLAN_CN_BASE_URL);
const report = await alibabaTokenPlanUsageProvider.fetchUsage(params(credential), { fetch: fetchMock });
expect(requests).toHaveLength(2);
expect(requests[0]?.url).toBe("https://bailian.console.aliyun.com/cn-beijing?tab=plan");
expect(new Headers(requests[0]?.init?.headers).get("Cookie")).toBe(cookie);
expect(requests[1]?.url).toBe(
"https://bailian-cs.console.aliyun.com/data/api.json?action=BroadScopeAspnGateway&product=sfm_bailian&api=zeldaHttp.apikeyMgr.%2Ftokenplan%2Fpersonal%2Fapi%2Fv2%2Fusage",
);
const usageHeaders = new Headers(requests[1]?.init?.headers);
expect(usageHeaders.get("Origin")).toBe("https://bailian.console.aliyun.com");
expect(usageHeaders.get("Referer")).toBe("https://bailian.console.aliyun.com/cn-beijing?tab=plan");
const body = new URLSearchParams(String(requests[1]?.init?.body));
expect(body.get("action")).toBe("BroadScopeAspnGateway");
expect(body.get("region")).toBe("cn-beijing");
expect(body.get("sec_token")).toBe("cn-sec-token");
const gatewayParams: unknown = JSON.parse(body.get("params") ?? "null");
expect(gatewayParams).toMatchObject({
Api: "zeldaHttp.apikeyMgr./tokenplan/personal/api/v2/usage",
Data: {
cornerstoneParam: {
feTraceId: expect.any(String),
feURL: "https://bailian.console.aliyun.com/cn-beijing?tab=plan#/efm/subscription/token-plan/personal",
protocol: "V2",
console: "ONE_CONSOLE",
productCode: "p_efm",
switchAgent: 12608464,
switchUserType: 3,
domain: "bailian.console.aliyun.com",
consoleSite: "BAILIAN_ALIYUN",
userNickName: "",
userPrincipalName: "",
xsp_lang: "zh-CN",
},
},
V: "1.0",
});
expect(report).toMatchObject({
provider: "alibaba-token-plan",
limits: [
{
id: "credits:7d",
window: { id: "7d", durationMs: 604_800_000, resetsAt: 1_786_716_480_000 },
amount: { usedFraction: 0.7913113, unit: "percent" },
},
],
});
expect(report?.limits).toHaveLength(1);
});
test("does not claim quota support for API-key-only credentials", async () => {
let fetched = false;
const fetchMock: FetchImpl = () => {
+9 -1
View File
@@ -36,10 +36,17 @@ describe("QwenCloud Token Plan login", () => {
test("China (Beijing) region validates against and routes inference to cn-beijing", async () => {
const authRequests: { url: string; instructions?: string }[] = [];
let requestedUrl = "";
let cookiePrompt = "";
const prompts = ["2", "sk-sp-beijing"];
const credential = await loginAlibabaTokenPlan({
onAuth: request => authRequests.push(request),
onPrompt: async prompt => (prompt.allowEmpty ? "" : (prompts.shift() ?? "")),
onPrompt: async prompt => {
if (prompt.allowEmpty) {
cookiePrompt = prompt.message;
return "";
}
return prompts.shift() ?? "";
},
fetch: input => {
requestedUrl = String(input);
return Promise.resolve(Response.json({ data: [{ id: "qwen3.7-plus" }] }));
@@ -48,6 +55,7 @@ describe("QwenCloud Token Plan login", () => {
expect(requestedUrl).toBe("https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1/models");
expect(authRequests[0]?.url).toBe("https://www.aliyun.com/benefit/scene/tokenplan");
expect(cookiePrompt).toContain("bailian-cs.console.aliyun.com/data/api.json");
expect(JSON.parse(credential)).toEqual({
token: "sk-sp-beijing",
baseUrl: "https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1",
+36
View File
@@ -153,6 +153,42 @@ describe("auth-broker wire surface", () => {
}
});
test("GET /v1/usage outlives the base timeout for a serialized account batch", async () => {
vi.useFakeTimers();
const response = Promise.withResolvers<Response>();
let usageSignal: AbortSignal | undefined;
const fetchImpl: typeof fetch = Object.assign(
async (_input: string | URL | Request, init?: RequestInit) => {
const signal = init?.signal;
if (signal) usageSignal = signal;
return response.promise;
},
{ preconnect: fetch.preconnect },
);
const client = new AuthBrokerClient({
url: "http://broker.invalid",
token,
timeoutMs: 10_000,
maxRetries: 0,
fetchImpl,
});
try {
const usage = client.fetchUsage({ maxAccountsPerProvider: 3 });
await Promise.resolve();
const baseTimeout = AbortSignal.timeout(10_000);
vi.advanceTimersByTime(10_001);
await Promise.resolve();
expect(baseTimeout.aborted).toBe(true);
expect(usageSignal?.aborted).toBe(false);
const generatedAt = Date.now();
response.resolve(Response.json({ generatedAt, reports: [] }));
expect(await usage).toEqual({ generatedAt, reports: [] });
} finally {
vi.useRealTimers();
}
});
test("GET /v1/snapshot returns generation headers and 304 for unchanged long-poll", async () => {
const res = await fetch(`${handle!.url}/v1/snapshot`, {
headers: { Authorization: `Bearer ${token}` },
+195 -3
View File
@@ -15,7 +15,7 @@ import {
} from "@oh-my-pi/pi-ai/auth-broker";
import { snapshotResponseSchema } from "@oh-my-pi/pi-ai/auth-broker/wire-schemas";
import * as oauthUtils from "@oh-my-pi/pi-ai/registry/oauth";
import type { UsageLimit, UsageReport } from "@oh-my-pi/pi-ai/usage";
import type { UsageLimit, UsageProvider, UsageReport } from "@oh-my-pi/pi-ai/usage";
import * as claudeUsage from "@oh-my-pi/pi-ai/usage/claude";
import { removeWithRetries } from "../../utils/src/temp";
@@ -34,8 +34,10 @@ describe("RemoteAuthCredentialStore + AuthStorage integration", () => {
let serverStorage: AuthStorage | undefined;
let handle: AuthBrokerServerHandle | undefined;
const token = "remote-bearer";
let testUsageProviders: Map<string, UsageProvider> | undefined;
beforeEach(async () => {
testUsageProviders = undefined;
for (const key of ANTHROPIC_ENV) {
savedEnv[key] = process.env[key];
delete process.env[key];
@@ -50,7 +52,12 @@ describe("RemoteAuthCredentialStore + AuthStorage integration", () => {
email: "a@example.com",
});
serverStorage = new AuthStorage(serverStore, {
usageProviderResolver: provider => (provider === "anthropic" ? claudeUsage.claudeUsageProvider : undefined),
usageProviderResolver: provider =>
testUsageProviders
? testUsageProviders.get(provider)
: provider === "anthropic"
? claudeUsage.claudeUsageProvider
: undefined,
});
await serverStorage.reload();
handle = startAuthBroker({
@@ -62,6 +69,7 @@ describe("RemoteAuthCredentialStore + AuthStorage integration", () => {
});
afterEach(async () => {
testUsageProviders = undefined;
vi.restoreAllMocks();
await handle?.close();
serverStorage?.close();
@@ -1110,7 +1118,7 @@ describe("RemoteAuthCredentialStore + AuthStorage integration", () => {
test("broker invalidation drops server-side last-good usage reports", async () => {
const credential = serverStore!.listAuthCredentials("anthropic")[0];
if (!credential || credential.credential.type !== "oauth") throw new Error("expected OAuth credential");
if (credential?.credential.type !== "oauth") throw new Error("expected OAuth credential");
serverStore!.updateAuthCredential(credential.id, {
...credential.credential,
expires: Date.now() + 3_600_000,
@@ -1155,6 +1163,190 @@ describe("RemoteAuthCredentialStore + AuthStorage integration", () => {
}
});
test("broker returns an upgraded plan through a delayed serialized Codex refresh", async () => {
const accountIds = ["account-free", "account-upgraded", "account-other"];
const refreshStarted = new Map<string, PromiseWithResolvers<void>>();
const refreshReleases = new Map<string, PromiseWithResolvers<void>>();
for (const accountId of accountIds) {
refreshStarted.set(accountId, Promise.withResolvers<void>());
refreshReleases.set(accountId, Promise.withResolvers<void>());
}
const startedAccounts: string[] = [];
const usageProvider: UsageProvider = {
id: "openai-codex",
supports: params => params.provider === "openai-codex" && params.credential.type === "oauth",
async fetchUsage(params) {
const accountId = params.credential.accountId;
if (!accountId) return null;
const started = refreshStarted.get(accountId);
const release = refreshReleases.get(accountId);
if (!started || !release) throw new Error(`unexpected account ${accountId}`);
startedAccounts.push(accountId);
started.resolve();
await release.promise;
return {
provider: "openai-codex",
fetchedAt: Date.now(),
limits: [
{
id: "openai-codex:7d",
label: "7 days",
scope: { provider: "openai-codex", windowId: "7d" },
amount: { used: 0, limit: 100, unit: "percent" },
status: "ok",
},
],
metadata: {
accountId,
email: `${accountId.slice("account-".length)}@example.com`,
planType: accountId === "account-upgraded" ? "pro" : "free",
},
};
},
};
testUsageProviders = new Map([["openai-codex", usageProvider]]);
for (const accountId of accountIds) {
serverStore!.upsertAuthCredentialForProvider("openai-codex", {
type: "oauth",
access: `access-${accountId}`,
refresh: `refresh-${accountId}`,
expires: Date.now() + 3_600_000,
accountId,
email: `${accountId.slice("account-".length)}@example.com`,
});
}
await serverStorage!.reload();
const brokerClient = new AuthBrokerClient({ url: handle!.url, token, timeoutMs: 10_000 });
const initialResult = await brokerClient.fetchSnapshot();
if (initialResult.status !== 200) throw new Error("expected snapshot");
const remoteStore = new RemoteAuthCredentialStore({
client: brokerClient,
initialSnapshot: initialResult.snapshot,
});
const clientStorage = new AuthStorage(remoteStore);
await clientStorage.reload();
try {
await clientStorage.invalidateUsageCache();
const refresh = clientStorage.fetchUsageReports();
const freeStarted = refreshStarted.get("account-free");
const freeRelease = refreshReleases.get("account-free");
if (!freeStarted || !freeRelease) throw new Error("missing free-account refresh gates");
await freeStarted.promise;
expect(startedAccounts).toEqual(["account-free"]);
freeRelease.resolve();
const upgradedStarted = refreshStarted.get("account-upgraded");
const upgradedRelease = refreshReleases.get("account-upgraded");
if (!upgradedStarted || !upgradedRelease) throw new Error("missing upgraded-account refresh gates");
await upgradedStarted.promise;
expect(startedAccounts).toEqual(["account-free", "account-upgraded"]);
upgradedRelease.resolve();
const otherStarted = refreshStarted.get("account-other");
const otherRelease = refreshReleases.get("account-other");
if (!otherStarted || !otherRelease) throw new Error("missing other-account refresh gates");
await otherStarted.promise;
expect(startedAccounts).toEqual(["account-free", "account-upgraded", "account-other"]);
otherRelease.resolve();
const reports = await refresh;
expect(reports).toHaveLength(3);
expect(reports?.find(report => report.metadata?.accountId === "account-upgraded")?.metadata?.planType).toBe(
"pro",
);
} finally {
for (const release of refreshReleases.values()) release.resolve();
clientStorage.close();
}
});
test("sizes broker usage timeout from the unfiltered account-pool snapshot", async () => {
vi.useFakeTimers();
const now = Date.now();
const credentials: SnapshotResponse["credentials"] = [];
for (let index = 0; index < 6; index += 1) {
const accountId = `account-${index}`;
credentials.push({
id: index + 1,
provider: "openai-codex",
credential: {
type: "oauth",
access: `access-${index}`,
refresh: REMOTE_REFRESH_SENTINEL,
expires: now + 120_000,
accountId,
email: `${index}@example.com`,
},
identityKey: `email:${index}@example.com`,
rotatesInMs: null,
});
}
const visible = credentials[0];
if (!visible?.identityKey) throw new Error("expected visible account identity");
const initialSnapshot: SnapshotResponse = {
generation: 1,
generatedAt: now,
serverNowMs: now,
refresher: { enabled: false, intervalMs: 0, skewMs: 0, nextSweepInMs: Number.MAX_SAFE_INTEGER },
credentials,
};
const response = Promise.withResolvers<Response>();
const snapshotResponse = Promise.withResolvers<Response>();
let usageSignal: AbortSignal | undefined;
const fetchImpl: typeof fetch = Object.assign(
async (input: string | URL | Request, init?: RequestInit) => {
const pathname = new URL(String(input)).pathname;
if (pathname === "/v1/snapshot") return snapshotResponse.promise;
if (pathname !== "/v1/usage") throw new Error(`unexpected path ${pathname}`);
const signal = init?.signal;
if (signal) usageSignal = signal;
return response.promise;
},
{ preconnect: fetch.preconnect },
);
const brokerClient = new AuthBrokerClient({
url: "http://broker.invalid",
token: "unused",
timeoutMs: 10_000,
maxRetries: 0,
fetchImpl,
});
const remoteStore = new RemoteAuthCredentialStore({
client: brokerClient,
streamSnapshots: false,
accountPool: new Map([["openai-codex", new Set([visible.identityKey])]]),
initialSnapshot,
});
try {
const usage = remoteStore.fetchUsageReports();
await Promise.resolve();
const oneAccountBudget = AbortSignal.timeout(20_000);
vi.advanceTimersByTime(20_001);
await Promise.resolve();
expect(oneAccountBudget.aborted).toBe(true);
expect(usageSignal?.aborted).toBe(false);
response.resolve(
Response.json({
generatedAt: Date.now(),
reports: [
{
provider: "openai-codex",
fetchedAt: Date.now(),
limits: [],
metadata: { accountId: "account-0", email: "0@example.com" },
},
],
}),
);
expect(await usage).toHaveLength(1);
} finally {
remoteStore.close();
snapshotResponse.resolve(new Response(null, { status: 304, headers: { ETag: '"1"' } }));
vi.useRealTimers();
}
});
test("account pool exposes only qualified usage reports for visible OAuth identities", async () => {
const brokerClient = new AuthBrokerClient({ url: "http://127.0.0.1:9", token: "unused" });
const now = Date.now();
+10
View File
@@ -5,6 +5,16 @@
### Added
- Added support for GLM-5.3 on the z.AI provider. GLM-5.3 introduces a uniform wire-exact `low`/`high`/`max` reasoning-effort ladder on every host (replacing GLM-5.2's host-specific dialects), makes thinking mandatory (`thinking.type` must always be `enabled`; disabling is no longer supported), and defaults to `max` effort. The model is pinned to 1M context and set as the z.AI provider default.
## [17.3.4] - 2026-08-14
### Added
- Added wire constants for Codex V2 remote-compaction feature negotiation.
### Fixed
- Fixed raw `COPILOT_GITHUB_TOKEN` credentials skipping plan-specific endpoint discovery, which routed GitHub Copilot Business model requests to the personal endpoint and returned HTTP 403. The GitHub Copilot model cache is now scoped per credential, so switching the token no longer serves another account's stale endpoint for the cache TTL ([#8507](https://github.com/can1357/oh-my-pi/issues/8507)).
- Fixed the OpenRouter `deepseek/deepseek-v4-pro-0813` route silently clamping the reasoning effort to `high`: the dated SKU advertises (and accepts) the wire-exact `low`/`high`/`max` ladder, so its effort override no longer collapses to `high`-only. The undated `deepseek/deepseek-v4-pro` OpenRouter route stays `high`-only. ([#8517](https://github.com/can1357/oh-my-pi/issues/8517))
## [17.3.2] - 2026-08-13
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "17.3.3",
"version": "17.3.4",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+10 -3
View File
@@ -383,13 +383,20 @@ function getModelDefinedEfforts<TApi extends Api>(
// on every first-party/aggregator host — the direct API, aggregators, and
// Ollama Cloud alike (medium/xhigh fold into high, max is a real wire
// tier). See https://api-docs.deepseek.com/api/create-chat-completion.
// OpenRouter's non-Flash V4 route still exposes only high; the older
// reasoners (V3.x, R1, deepseek-reasoner) top out at high/max.
// OpenRouter's non-Flash V4 route exposes only high, except the dated
// `deepseek-v4-pro-0813` SKU: its /models metadata advertises (and the
// route accepts) the full low/high/max ladder like every other host.
// The older reasoners (V3.x, R1, deepseek-reasoner) top out at high/max.
if (isDeepseekV4FlashModelId(spec.id)) {
return LOW_HIGH_MAX_REASONING_EFFORTS;
}
if (bareModelId(spec.id).toLowerCase().includes("deepseek-v4")) {
return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : LOW_HIGH_MAX_REASONING_EFFORTS;
if (!isOpenRouterThinkingFormat(compat)) {
return LOW_HIGH_MAX_REASONING_EFFORTS;
}
return bareModelId(spec.id).toLowerCase() === "deepseek-v4-pro-0813"
? LOW_HIGH_MAX_REASONING_EFFORTS
: HIGH_ONLY_REASONING_EFFORTS;
}
return isOpenRouterThinkingFormat(compat) ? HIGH_ONLY_REASONING_EFFORTS : HIGH_MAX_REASONING_EFFORTS;
}
@@ -1,3 +1,5 @@
import { PERSONAL_GITHUB_COPILOT_BASE_URL } from "../wire/github-copilot";
export interface ModelCacheProviderIdOptions {
apiKey?: string;
baseUrl?: string;
@@ -56,6 +58,18 @@ export function resolveModelCacheProviderId(providerId: string, options: ModelCa
const scope = `${options.apiKey ?? ""}\u0000${discoveryBaseUrl}`;
return `${providerId}:models-v1:${Bun.hash(scope).toString(36)}`;
}
case "github-copilot": {
// Copilot model specs bake in the plan-specific endpoint (personal vs
// Business/Enterprise) resolved from the credential. Discovery writes an
// authoritative cache, so `online-if-uncached` serves it for the full
// TTL without re-probing. Keying the namespace on the credential means
// switching `COPILOT_GITHUB_TOKEN` to a different account misses the
// prior endpoint's cache and re-runs discovery instead of hitting the
// stale host and 403ing (PR #8510 review).
const baseUrl = options.baseUrl ?? PERSONAL_GITHUB_COPILOT_BASE_URL;
const scope = `${options.apiKey ?? ""}\u0000${baseUrl}`;
return `github-copilot:models-v1:${Bun.hash(scope).toString(36)}`;
}
case "openrouter":
return "openrouter:pseudo-api";
case "vllm": {
@@ -1,6 +1,7 @@
import { USER_AGENT } from "@oh-my-pi/pi-utils";
import * as logger from "@oh-my-pi/pi-utils/logger";
import {
DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS,
fetchOpenAICompatibleModels,
type OpenAICompatibleModelMapperContext,
type OpenAICompatibleModelRecord,
@@ -25,6 +26,7 @@ import { ALIBABA_TOKEN_PLAN_BASE_URL, parseAlibabaTokenPlanCredential } from "..
import { coreWeaveProjectHeaders } from "../wire/coreweave";
import {
COPILOT_API_HEADERS,
discoverGitHubCopilotApiEndpoint,
getGitHubCopilotBaseUrl,
isPersonalGitHubCopilotBaseUrl,
parseGitHubCopilotApiKey,
@@ -5191,6 +5193,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
const resolveReference = createReferenceResolver(getProviderReferences);
return {
providerId: "github-copilot",
cacheProviderId: resolveModelCacheProviderId("github-copilot", { apiKey: rawApiKey, baseUrl }),
dropCachedModelIdsOnStaticMismatch: COPILOT_CACHE_INVALIDATED_MODEL_IDS,
// COPILOT_API_HEADERS are compile-time constants (User-Agent + API
// version), not credentials. The cache omits all request headers for
@@ -5201,11 +5204,17 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
restorableHeaderFallback: { ...COPILOT_API_HEADERS },
...(apiKey && {
fetchDynamicModels: async () => {
const fetchImpl = discoveryFetch(config?.fetch);
const requestBaseUrl = isPersonalGitHubCopilotBaseUrl(baseUrl)
? ((await withCatalogDiscoveryTimeout(DEFAULT_OPENAI_COMPATIBLE_DISCOVERY_TIMEOUT_MS, signal =>
discoverGitHubCopilotApiEndpoint(apiKey, fetchImpl, signal),
)) ?? baseUrl)
: baseUrl;
const longContextVariants: ModelSpec<Api>[] = [];
const models = await fetchOpenAICompatibleModels<Api>({
api: "openai-completions",
provider: "github-copilot",
baseUrl,
baseUrl: requestBaseUrl,
apiKey,
headers: COPILOT_API_HEADERS,
mapModel: (
@@ -5251,7 +5260,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
const input: ModelSpec<Api>["input"] =
supportsVision === true
? ["text", "image"]
: supportsVision === false || !isPersonalGitHubCopilotBaseUrl(baseUrl)
: supportsVision === false || !isPersonalGitHubCopilotBaseUrl(requestBaseUrl)
? ["text"]
: (reference?.input ?? defaults.input);
// With COPILOT_API_HEADERS the served window is the long-context
@@ -5272,7 +5281,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
...reference,
api,
provider: "github-copilot",
baseUrl,
baseUrl: requestBaseUrl,
name,
input,
contextWindow: defaultTierWindow,
@@ -5294,7 +5303,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
: {
...defaults,
api,
baseUrl,
baseUrl: requestBaseUrl,
name,
input,
contextWindow: defaultTierWindow,
@@ -5340,7 +5349,7 @@ export function githubCopilotModelManagerOptions(config?: GithubCopilotModelMana
}
return base;
},
fetch: config?.fetch,
fetch: fetchImpl,
});
if (models === null) {
return null;
+3
View File
@@ -11,6 +11,8 @@ export const CODEX_CLIENT_VERSION = "0.144.1";
export const OPENAI_HEADERS = {
BETA: "OpenAI-Beta",
/** Codex feature-negotiation header; values identify opt-in wire protocols. */
CODEX_BETA_FEATURES: "x-codex-beta-features",
ACCOUNT_ID: "chatgpt-account-id",
ORIGINATOR: "originator",
VERSION: "version",
@@ -32,6 +34,7 @@ export const OPENAI_HEADERS = {
export const OPENAI_HEADER_VALUES = {
BETA_RESPONSES: "responses=experimental",
BETA_RESPONSES_WEBSOCKETS_V2: "responses_websockets=2026-02-06",
REMOTE_COMPACTION_V2: "remote_compaction_v2",
ORIGINATOR_CODEX: "pi",
} as const;
@@ -1,3 +1,6 @@
import type { FetchImpl } from "../types";
import { isRecord } from "../utils";
/**
* GitHub Copilot wire metadata: API-key envelope parsing and endpoint
* derivation shared by catalog discovery and the pi-ai OAuth flow. The device
@@ -72,6 +75,35 @@ export function normalizeGitHubCopilotApiEndpoint(input: string | undefined): st
return undefined;
}
}
/**
* Resolve the plan-specific Copilot API endpoint advertised for a GitHub token.
* Login and raw environment-token discovery share this best-effort probe. Pass
* a `signal` to bound it against the same discovery deadline as `/models`; a
* stalled probe otherwise blocks discovery indefinitely.
*/
export async function discoverGitHubCopilotApiEndpoint(
token: string,
fetchImpl: FetchImpl,
signal?: AbortSignal,
): Promise<string | undefined> {
try {
const response = await fetchImpl("https://api.github.com/copilot_internal/user", {
headers: {
Accept: "application/json",
Authorization: `token ${token}`,
...OPENCODE_HEADERS,
},
signal,
});
if (!response.ok) return undefined;
const data: unknown = await response.json();
if (!isRecord(data) || !isRecord(data.endpoints)) return undefined;
const endpoint = data.endpoints.api;
return typeof endpoint === "string" ? normalizeGitHubCopilotApiEndpoint(endpoint) : undefined;
} catch {
return undefined;
}
}
export function parseGitHubCopilotApiKey(apiKeyRaw: string): ParsedGitHubCopilotApiKey {
try {
@@ -42,6 +42,13 @@ async function discoverCopilotModels(
const requestApiVersions: Array<string | undefined> = [];
const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
const url = typeof input === "string" ? input : input.toString();
if (url === "https://api.github.com/copilot_internal/user") {
expect(getHeaderValue(init?.headers, "Authorization")).toBe(`token ${expectedAuthorizationToken}`);
// The probe must be bounded by the shared discovery deadline so a
// stalled endpoint cannot hang discovery (PR #8510 review).
expect(init?.signal).toBeInstanceOf(AbortSignal);
return Response.json({ endpoints: { api: expectedBaseUrl } });
}
expect(url).toBe(`${expectedBaseUrl}/models`);
expect(init?.method).toBe("GET");
expect(getHeaderValue(init?.headers, "Authorization")).toBe(`Bearer ${expectedAuthorizationToken}`);
@@ -72,13 +79,77 @@ function cachedCopilotCompletionModel(id: string, name: string): ModelSpec<"open
}
describe("github copilot model limits mapping", () => {
it("uses configured base URL for discovery", async () => {
it("discovers the plan endpoint for a raw environment token before model discovery", async () => {
const token = "ghu_valid_business_token";
const { fetchMock } = await discoverCopilotModels(
{ data: [] },
"copilot-test-key",
"https://api.githubcopilot.com",
token,
"https://api.business.githubcopilot.com",
token,
);
expect(fetchMock).toHaveBeenCalledTimes(1);
expect(fetchMock).toHaveBeenCalledTimes(2);
});
it("falls back to the personal endpoint when the raw-token probe fails", async () => {
const token = "ghu_valid_business_token";
const fetchMock = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
const url = typeof input === "string" ? input : input.toString();
if (url === "https://api.github.com/copilot_internal/user") {
expect(init?.signal).toBeInstanceOf(AbortSignal);
throw new DOMException("The operation timed out.", "TimeoutError");
}
expect(url).toBe("https://api.githubcopilot.com/models");
return Response.json({ data: [] });
});
const models = await githubCopilotModelManagerOptions({ apiKey: token, fetch: fetchMock }).fetchDynamicModels?.();
expect(models).toEqual([]);
expect(fetchMock).toHaveBeenCalledTimes(2);
});
it("does not reuse another token's authoritative cache after COPILOT_GITHUB_TOKEN switches", async () => {
const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-copilot-token-switch-"));
const cacheDbPath = path.join(tempDir, "models.db");
try {
const personalFetch = vi.fn(async (input: string | URL | Request) => {
const url = typeof input === "string" ? input : input.toString();
if (url === "https://api.github.com/copilot_internal/user") {
return Response.json({ endpoints: { api: "https://api.githubcopilot.com" } });
}
if (url === "https://api.githubcopilot.com/models") {
return Response.json({ data: [{ id: "gpt-5.5", name: "GPT-5.5" }] });
}
throw new Error(`unexpected personal request: ${url}`);
});
const personalManager = createModelManager({
...githubCopilotModelManagerOptions({ apiKey: "ghu_personal_token", fetch: personalFetch }),
cacheDbPath,
});
// Personal token discovery writes a fresh authoritative cache.
await personalManager.refresh("online");
const businessSeen: string[] = [];
const businessFetch = vi.fn(async (input: string | URL | Request) => {
const url = typeof input === "string" ? input : input.toString();
businessSeen.push(url);
if (url === "https://api.github.com/copilot_internal/user") {
return Response.json({ endpoints: { api: "https://api.business.githubcopilot.com" } });
}
if (url === "https://api.business.githubcopilot.com/models") {
return Response.json({ data: [{ id: "gpt-5.5", name: "GPT-5.5" }] });
}
throw new Error(`unexpected business request: ${url}`);
});
const businessManager = createModelManager({
...githubCopilotModelManagerOptions({ apiKey: "ghu_business_token", fetch: businessFetch }),
cacheDbPath,
});
// Default online-if-uncached must not satisfy the switched token from the
// prior token's fresh authoritative personal-endpoint cache.
const { models } = await businessManager.refresh("online-if-uncached");
expect(businessSeen).toContain("https://api.github.com/copilot_internal/user");
expect(businessSeen).toContain("https://api.business.githubcopilot.com/models");
expect(models.some(model => model.baseUrl === "https://api.business.githubcopilot.com")).toBe(true);
} finally {
await fs.rm(tempDir, { recursive: true, force: true });
}
});
it("unwraps structured OAuth keys for discovery and routes enterprise discovery to the enterprise host", async () => {
@@ -355,7 +426,7 @@ describe("github copilot model limits mapping", () => {
const { models } = await manager.refresh("online-if-uncached");
const model = models.find(candidate => candidate.id === migration.id);
expect(fetchMock).toHaveBeenCalledTimes(1);
expect(fetchMock).toHaveBeenCalledTimes(2);
expect(model?.api).toBe("openai-responses");
} finally {
await fs.rm(tempDir, { recursive: true, force: true });
@@ -390,7 +461,7 @@ describe("github copilot model limits mapping", () => {
});
const { models } = await manager.refresh("online-if-uncached");
expect(fetchMock).toHaveBeenCalledTimes(1);
expect(fetchMock).toHaveBeenCalledTimes(2);
// The bundled catalog now ships a responses-route grok-4.5, so the id
// resurfaces from the bundle after the failed refresh. The migration
// contract is that the stale cached COMPLETIONS route never comes
@@ -314,6 +314,34 @@ describe("model thinking derivation", () => {
expect(getSupportedEfforts(v32)).toEqual([Effort.High, Effort.Max]);
});
it("grants the low/high/max ladder to OpenRouter deepseek-v4-pro-0813 but not the undated route (issue #8517)", () => {
// OpenRouter's /models advertises reasoning.supported_efforts
// [low, high, max] for the dated SKU; the discovered ladder is baked
// into thinking.efforts.
const discovered = { mode: "effort" as const, efforts: [Effort.Low, Effort.High, Effort.Max] };
const dated = createModel({
id: "deepseek/deepseek-v4-pro-0813",
api: "openrouter",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
thinking: discovered,
});
const bare = createModel({
id: "deepseek/deepseek-v4-pro",
api: "openrouter",
provider: "openrouter",
baseUrl: "https://openrouter.ai/api/v1",
thinking: discovered,
});
// The dated SKU keeps its advertised ladder; :max no longer clamps.
expect(getSupportedEfforts(dated)).toEqual([Effort.Low, Effort.High, Effort.Max]);
expect(clampThinkingLevelForModel(dated, Effort.Max)).toBe(Effort.Max);
// The undated OpenRouter route stays high-only.
expect(getSupportedEfforts(bare)).toEqual([Effort.High]);
expect(clampThinkingLevelForModel(bare, Effort.Max)).toBe(Effort.High);
});
it("encodes the Gemini 3 Pro effort gap and mandatory reasoning in metadata", () => {
const model = createModel({
id: "gemini-3-pro-preview",
+13
View File
@@ -2,6 +2,19 @@
## [Unreleased]
## [17.3.4] - 2026-08-14
### Changed
- Replaced the MuPDF-WASM PDF document backend with `pdf-inspector` through `@oh-my-pi/pi-natives`, preserving cached text conversion and PDF line selectors while reporting pages that need OCR.
- Restored `read <pdf>:` and `read <pdf>:<image>.png` page rendering by automatically capturing PDF pages through the headless Chromium browser tool.
### Fixed
- Fixed Streamable HTTP MCP sessions being invalidated by opening the optional GET SSE stream before sending `notifications/initialized`, which prevented Figma Dev Mode MCP from connecting ([#8514](https://github.com/can1357/oh-my-pi/issues/8514)).
- Fixed the `/hotkeys` table describing Ctrl+D (`app.exit`) as "Exit (when editor is empty)" when it actually exits unconditionally and saves the current prompt as a resumable draft ([#8530](https://github.com/can1357/oh-my-pi/issues/8530)).
- Fixed Ctrl+G external editors failing to launch on Windows because Bun re-quoted the embedded `cmd.exe /c` command line ([#8544](https://github.com/can1357/oh-my-pi/issues/8544)).
## [17.3.3] - 2026-08-14
### Fixed
+1 -4
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-coding-agent",
"version": "17.3.3",
"version": "17.3.4",
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
"homepage": "https://omp.sh",
"author": "Can Boluk",
@@ -41,8 +41,6 @@
"format-prompts": "bun scripts/format-prompts.ts",
"gen:tool-views": "bun --cwd=../collab-web run gen:tool-views",
"gen:bundle": "bun scripts/bundle-dist.ts",
"gen:mupdf": "bun scripts/embed-mupdf-wasm.ts --generate",
"gen:mupdf:reset": "bun scripts/embed-mupdf-wasm.ts --reset",
"gen:native": "bun --cwd=../natives run gen:native",
"gen:native:reset": "bun --cwd=../natives run gen:native:reset",
"prepack": "bun run gen:tool-views && bun run gen:bundle",
@@ -73,7 +71,6 @@
"@opentelemetry/sdk-metrics": "catalog:",
"@opentelemetry/sdk-trace-base": "catalog:",
"@opentelemetry/sdk-trace-node": "catalog:",
"mupdf": "catalog:",
"puppeteer-core": "catalog:"
},
"optionalDependencies": {
@@ -88,7 +88,6 @@ async function main(): Promise<void> {
["bun", "--cwd=../natives", "run", "gen:native"],
crossBuild ? { ...Bun.env, TARGET_PLATFORM: crossBuild.platform, TARGET_ARCH: crossBuild.arch } : Bun.env,
);
await runCommand(["bun", "run", "gen:mupdf"]);
try {
await compileCodingAgent({
repoRoot,
@@ -104,7 +103,6 @@ async function main(): Promise<void> {
await runCommand(["codesign", "--force", "--sign", "-", outputPath]);
}
} finally {
await runCommand(["bun", "run", "gen:mupdf:reset"]);
await runCommand(["bun", "--cwd=../natives", "run", "gen:native:reset"]);
}
} finally {
@@ -15,7 +15,6 @@ const legacyHtmlExportAssetPattern = /^(?:template-[^.]+\.(?:css|html|js)|tool-v
// `omp-legacy-pi-modules` exists only in compiled binaries via the build plugin;
// the npm bundle never executes that `isCompiledBinary()` branch.
const ALWAYS_EXTERNAL = [
"mupdf",
"@oh-my-pi/pi-natives",
"@huggingface/transformers",
"fastembed",
@@ -1,67 +0,0 @@
#!/usr/bin/env bun
// Embeds mupdf's `mupdf-wasm.wasm` into the compiled single-file binary.
//
// mupdf loads its wasm by reading the `mupdf-wasm.wasm` sibling of its own
// module via `new URL(..., import.meta.url)` + `readFileSync`. A `bun --compile`
// binary has no node_modules, so that read fails (`ENOENT .../mupdf-wasm.wasm`),
// and marking mupdf `--external` instead makes `bun --compile` eagerly fail to
// resolve the package at startup (the static `import * as mupdf` lives in a lazy
// chunk but is hoisted). So the binary build bundles mupdf and embeds the wasm
// bytes here, handing them to the WASM module as `$libmupdf_wasm_Module.wasmBinary`
// (see src/utils/markit.ts).
//
// `--generate` copies the wasm next to src/utils/mupdf-wasm-embed.ts and rewrites
// that module to import it via `with { type: "file" }`; `--reset` restores the
// checked-in placeholder and removes the copy. The npm `dist/cli.js` bundle never
// runs this — it keeps mupdf external and loads the wasm from node_modules.
import * as fs from "node:fs/promises";
import { createRequire } from "node:module";
import * as path from "node:path";
const utilsDir = path.join(import.meta.dir, "..", "src", "utils");
const helperPath = path.join(utilsDir, "mupdf-wasm-embed.ts");
const wasmCopyPath = path.join(utilsDir, "mupdf-wasm.wasm");
const placeholder = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
//
// Compiled single-file binaries cannot let mupdf resolve its \`mupdf-wasm.wasm\`
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
// wasm bytes via \`with { type: "file" }\` and copies the wasm next to it. Source
// checkouts, \`bun test\`, and the npm \`dist/cli.js\` bundle keep mupdf external and
// load the wasm from node_modules, so this placeholder returns undefined and the
// build resets back to it afterward.
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
\treturn undefined;
}
`;
const generated = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit or commit.
import { readFileSync } from "node:fs";
import wasmPath from "./mupdf-wasm.wasm" with { type: "file" };
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
\treturn readFileSync(wasmPath);
}
`;
if (process.argv.includes("--reset")) {
await Bun.write(helperPath, placeholder);
try {
await fs.unlink(wasmCopyPath);
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err;
}
process.exit(0);
}
const wasmSource = path.join(path.dirname(createRequire(import.meta.url).resolve("mupdf")), "mupdf-wasm.wasm");
const wasmFile = Bun.file(wasmSource);
if (!(await wasmFile.exists())) {
throw new Error(`mupdf wasm not found at ${wasmSource}; run \`bun install\` first.`);
}
await Bun.write(wasmCopyPath, wasmFile);
await Bun.write(helperPath, generated);
console.log(`Embedded mupdf wasm (${wasmFile.size} bytes) into ${path.relative(process.cwd(), wasmCopyPath)}`);
@@ -10,6 +10,7 @@ import chalk from "@oh-my-pi/pi-utils/chalk";
import { Settings } from "../config/settings";
import { extractUriScheme } from "../internal-urls/parse";
import { InternalUrlRouter } from "../internal-urls/router";
import { closeDaemonClients } from "../launch/client";
import { discoverAndLoadMCPTools } from "../mcp/loader";
import { MCPManager } from "../mcp/manager";
import { discoverAuthStorage } from "../session/auth-broker-config";
@@ -93,6 +94,7 @@ export async function runReadCommand(cmd: ReadCommandArgs): Promise<void> {
if (MCPManager.instance() === mcpManager) MCPManager.setInstance(undefined);
}
authStorage?.close();
await closeDaemonClients();
}
if (failed) process.exit(1);
+8 -8
View File
@@ -1,15 +1,15 @@
This directory contains an in-house document-to-markdown engine adapted from
Portions of this in-house document-to-markdown engine are adapted from
markit-ai (https://github.com/Michaelliv/markit), used under the MIT License.
This attribution covers the shared registry/types and the DOCX, PPTX, XLSX,
and EPUB converters. The PDF converter is implemented separately and is not
derived from markit-ai.
Copyright (c) 2026 Michael Liv
Only the converters for the document formats omp supports are ported (pdf,
docx, pptx, xlsx, epub); the CLI, plugin/provider, and unused converters
(html, image, audio, plain-text, rss, github, wikipedia, csv, json, yaml,
ipynb, iwork, zip, xml) were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and
`.rtf` are routed by the read/fetch tools but have no converter — they surface
a conversion error, exactly as upstream markit did. Logic is ported faithfully
so conversion output matches the upstream package.
The CLI, plugin/provider, and unused converters (html, image, audio,
plain-text, rss, github, wikipedia, csv, json, yaml, ipynb, iwork, zip, xml)
were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and `.rtf` have no converter
and surface a conversion error.
MIT License
@@ -1,103 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Multi-column layout detection and text box reordering.
*
* Many PDFs (legal documents, datasheets, academic papers) use two-column
* layouts. Without column detection, text boxes are ordered by Y position
* only, interleaving left and right column content.
*
* Algorithm:
* 1. Collect left edges of all text boxes on the page
* 2. Find the largest horizontal gap between consecutive left edges
* 3. If gap > MIN_GAP_RATIO of the text width and both sides have
* enough boxes → multi-column detected
* 4. Assign each text box to a column based on its center X
* 5. Return columns in reading order (left-to-right, top-to-bottom)
*
* This only detects the column structure. The caller is responsible for
* processing each column's text boxes independently (table detection,
* rendering, etc.).
*/
import type { TextBox } from "./types";
export interface ColumnLayout {
/** Number of columns detected (1 = single column, 2+ = multi-column). */
columnCount: number;
/** Text boxes grouped by column, in reading order (left to right). */
columns: TextBox[][];
/** X positions of column boundaries (between columns). */
boundaries: number[];
}
/**
* Minimum gap as a fraction of the total text width to consider a column
* boundary. A two-column layout typically has ~50% gap; we use a lower
* threshold to catch asymmetric columns.
*/
const MIN_GAP_RATIO = 0.15;
/** Minimum number of text boxes on each side of the gap. */
const MIN_BOXES_PER_COLUMN = 4;
/** Minimum gap in absolute points to avoid splitting on small whitespace. */
const MIN_GAP_PTS = 40;
/**
* Detect column layout and return text boxes grouped by column.
*
* For single-column pages, returns all boxes in one group.
* For multi-column pages, returns boxes split by column in reading order.
*/
export function detectColumns(textBoxes: TextBox[]): ColumnLayout {
if (textBoxes.length < MIN_BOXES_PER_COLUMN * 2) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Collect unique left edges (rounded to avoid float noise)
const lefts = [...new Set(textBoxes.map(tb => Math.round(tb.bounds.left)))].sort((a, b) => a - b);
if (lefts.length < 2) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
const textXMin = lefts[0];
const textXMax = Math.max(...textBoxes.map(tb => Math.round(tb.bounds.right)));
const textWidth = textXMax - textXMin;
if (textWidth <= 0) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Find the largest gap between consecutive left-edge positions
let maxGap = 0;
let gapLeft = 0;
let gapRight = 0;
for (let i = 1; i < lefts.length; i++) {
const gap = lefts[i] - lefts[i - 1];
if (gap > maxGap) {
maxGap = gap;
gapLeft = lefts[i - 1];
gapRight = lefts[i];
}
}
const gapRatio = maxGap / textWidth;
if (gapRatio < MIN_GAP_RATIO || maxGap < MIN_GAP_PTS) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Split point is the midpoint of the gap
const splitX = (gapLeft + gapRight) / 2;
// Assign boxes to columns based on center X
const leftCol: TextBox[] = [];
const rightCol: TextBox[] = [];
for (const tb of textBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
if (cx < splitX) {
leftCol.push(tb);
} else {
rightCol.push(tb);
}
}
// Validate both columns have enough content
if (leftCol.length < MIN_BOXES_PER_COLUMN || rightCol.length < MIN_BOXES_PER_COLUMN) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
return {
columnCount: 2,
columns: [leftCol, rightCol],
boundaries: [splitX],
};
}
@@ -1,598 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* PDF content extraction using mupdf.
*
* Extracts text boxes (with position, font size, bold) and vector line
* segments (table borders) from each page. Uses mupdf's native WASM
* engine for fast parsing, and reads raw content streams for vector graphics.
*
* Coordinate system: PDF native (origin = bottom-left, Y increases upward).
*/
import type * as mupdf from "mupdf";
import type { ImageRegion, PageContent, Segment, TextBox } from "./types";
// mupdf instantiates its WASM module via a top-level await. A static
// `import * as mupdf` would pull that await into this module's init, which makes
// the whole bundled markit chunk's `__esm` init async — and bun's compiled
// bundler fails to await that init transitively through the `../markit` barrel,
// exposing the converter classes before their module-level consts initialize
// (e.g. `EXTENSIONS` reads as undefined). Importing mupdf lazily keeps the chunk
// init synchronous and also keeps the ~10MB wasm off non-PDF conversions.
let mupdfModule: typeof mupdf | undefined;
async function loadMupdf(): Promise<typeof mupdf> {
if (!mupdfModule) {
mupdfModule = await import("mupdf");
}
return mupdfModule;
}
/** mupdf structured-text JSON bounding box (top-left origin). */
interface StextBBox {
x: number;
y: number;
w: number;
h: number;
}
/** Font metadata attached to a structured-text line. */
interface StextFont {
size?: number;
weight?: string;
name?: string;
}
/** A line within a text block in mupdf structured-text JSON. */
interface StextLine {
text?: string;
font?: StextFont;
bbox: StextBBox;
}
/** A block (text or image) in mupdf structured-text JSON. */
interface StextBlock {
type: string;
bbox: StextBBox;
lines: StextLine[];
}
/** Parsed mupdf structured-text JSON for a page. */
interface StructuredTextJSON {
blocks: StextBlock[];
}
/** A raw text fragment before merging into word/phrase boxes. */
interface RawTextItem {
text: string;
x: number;
y: number;
width: number;
height: number;
fontSize: number;
isBold: boolean;
}
// ---------------------------------------------------------------------------
// Text extraction
// ---------------------------------------------------------------------------
/** Y tolerance for merging text fragments on the same visual line. */
const SAME_LINE_Y_TOLERANCE = 2;
/** Max horizontal gap (pts) to merge adjacent fragments into one text box. */
const MAX_MERGE_GAP = 14;
/**
* Merge horizontally adjacent raw text items on the same visual line into
* word/phrase-level text boxes.
*/
function mergeIntoWords(raws: RawTextItem[]): RawTextItem[] {
if (raws.length === 0) return [];
// Sort by Y descending (top-first in bottom-left coords), then X ascending
const sorted = [...raws].sort((a, b) => {
const dy = b.y - a.y;
return Math.abs(dy) > SAME_LINE_Y_TOLERANCE ? dy : a.x - b.x;
});
const merged: RawTextItem[] = [];
let cur = { ...sorted[0] };
for (let i = 1; i < sorted.length; i++) {
const next = sorted[i];
const sameY = Math.abs(next.y - cur.y) <= SAME_LINE_Y_TOLERANCE;
const close = next.x <= cur.x + cur.width + MAX_MERGE_GAP;
if (sameY && close) {
const gap = next.x - (cur.x + cur.width);
const sep = gap > 1 ? " " : "";
cur.text += sep + next.text;
cur.width = next.x + next.width - cur.x;
cur.height = Math.max(cur.height, next.height);
cur.fontSize = Math.max(cur.fontSize, next.fontSize);
cur.isBold = cur.isBold || next.isBold;
} else {
merged.push(cur);
cur = { ...next };
}
}
merged.push(cur);
return merged;
}
/**
* Extract text boxes from a mupdf page using structured text output.
*
* mupdf's structured text JSON uses top-left origin; we convert to
* bottom-left (standard PDF coordinates) using the page height.
*/
function extractTextBoxes(
page: mupdf.Page,
pageNumber: number,
pageHeight: number,
stext?: StructuredTextJSON,
): TextBox[] {
if (!stext) {
stext = JSON.parse(page.toStructuredText("preserve-whitespace").asJSON()) as StructuredTextJSON;
}
const raws: RawTextItem[] = [];
for (const block of stext.blocks) {
if (block.type !== "text") continue;
for (const line of block.lines) {
const text = line.text?.trim();
if (!text) continue;
const fontSize = line.font?.size ?? 0;
const weight = line.font?.weight ?? "normal";
const fontName = line.font?.name ?? "";
const isBold = weight === "bold" || /bold/i.test(fontName) || /Black|Heavy/i.test(fontName);
// mupdf bbox: {x, y, w, h} in top-left coords
// Convert to bottom-left: pdfY = pageHeight - (bbox.y + bbox.h)
const bboxY = line.bbox.y;
const bboxH = line.bbox.h;
const pdfY = pageHeight - (bboxY + bboxH);
raws.push({
text,
x: line.bbox.x,
y: pdfY,
width: line.bbox.w,
height: bboxH,
fontSize,
isBold,
});
}
}
const words = mergeIntoWords(raws);
return words
.map((w, i) => ({
id: `p${pageNumber}-t${i}`,
text: w.text.trim(),
pageNumber,
fontSize: w.fontSize,
isBold: w.isBold,
bounds: {
left: w.x,
right: w.x + w.width,
bottom: w.y,
top: w.y + w.height,
},
}))
.filter(b => b.text.length > 0);
}
// ---------------------------------------------------------------------------
// Vector segment extraction from raw content stream
// ---------------------------------------------------------------------------
/** Minimum aspect ratio for a filled rect to be considered a line. */
const LINE_ASPECT_THRESHOLD = 6;
/** Minimum length (pts) for a segment to count. */
const MIN_LENGTH = 2;
/** Maximum thickness (pts) for a border line (filters out filled areas). */
const MAX_THICKNESS = 3;
/**
* Convert a thin filled rectangle to a horizontal or vertical segment.
* Returns null if the rect doesn't look like a border line.
*/
function thinRectToSegment(id: string, x: number, y: number, w: number, h: number): Segment | null {
const aw = Math.abs(w);
const ah = Math.abs(h);
if (aw > ah * LINE_ASPECT_THRESHOLD && aw >= MIN_LENGTH && ah <= MAX_THICKNESS) {
// Horizontal line
const cy = y + ah / 2;
return { id, x1: x, y1: cy, x2: x + aw, y2: cy };
}
if (ah > aw * LINE_ASPECT_THRESHOLD && ah >= MIN_LENGTH && aw <= MAX_THICKNESS) {
// Vertical line
const cx = x + aw / 2;
return { id, x1: cx, y1: y, x2: cx, y2: y + ah };
}
return null;
}
/**
* Emit 4 edge segments from a stroked rectangle.
*/
function pushStrokedRectEdges(segments: Segment[], id: string, x: number, y: number, w: number, h: number): void {
const aw = Math.abs(w);
const ah = Math.abs(h);
const base = id;
if (aw >= MIN_LENGTH) {
segments.push({ id: `${base}-b`, x1: x, y1: y, x2: x + aw, y2: y });
segments.push({
id: `${base}-t`,
x1: x,
y1: y + ah,
x2: x + aw,
y2: y + ah,
});
}
if (ah >= MIN_LENGTH) {
segments.push({ id: `${base}-l`, x1: x, y1: y, x2: x, y2: y + ah });
segments.push({
id: `${base}-r`,
x1: x + aw,
y1: y,
x2: x + aw,
y2: y + ah,
});
}
}
const CTM_IDENTITY = [1, 0, 0, 1, 0, 0];
/** Concatenate two affine matrices: result = parent × child. */
function ctmConcat(p: number[], c: number[]): number[] {
return [
p[0] * c[0] + p[2] * c[1],
p[1] * c[0] + p[3] * c[1],
p[0] * c[2] + p[2] * c[3],
p[1] * c[2] + p[3] * c[3],
p[0] * c[4] + p[2] * c[5] + p[4],
p[1] * c[4] + p[3] * c[5] + p[5],
];
}
function ctmApply(m: number[], x: number, y: number): [number, number] {
return [m[0] * x + m[2] * y + m[4], m[1] * x + m[3] * y + m[5]];
}
// ---------------------------------------------------------------------------
// Content stream parsing
// ---------------------------------------------------------------------------
/**
* Parse a PDF content stream and extract line segments from thin filled
* rectangles (re+f), stroked rectangles (re+S), and explicit lines (m/l+S).
* Tracks the CTM via q/Q/cm operators so coordinates are in page space.
*/
function extractSegmentsFromContentStream(raw: string, pageNumber: number): Segment[] {
const segments: Segment[] = [];
const tokens = tokenizeContentStream(raw);
let idx = 0;
let strokeWidth = 1.0;
// Graphics state stack (q/Q): saves CTM + strokeWidth
let ctm = [...CTM_IDENTITY];
const stateStack: Array<{ ctm: number[]; strokeWidth: number }> = [];
// State for path building (in user coordinates, pre-CTM)
let curX = 0;
let curY = 0;
let pathStartX = 0;
let pathStartY = 0;
const pendingRects: Array<{ x: number; y: number; w: number; h: number }> = [];
const pendingLines: Array<{ x1: number; y1: number; x2: number; y2: number }> = [];
function flushPath(mode: "fill" | "stroke"): void {
const sid = () => `p${pageNumber}-s${segments.length}`;
if (mode === "fill") {
for (const r of pendingRects) {
// Transform the rect corners through CTM, then check if it's a thin line
const [x0, y0] = ctmApply(ctm, r.x, r.y);
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
const seg = thinRectToSegment(
sid(),
Math.min(x0, x1),
Math.min(y0, y1),
Math.abs(x1 - x0),
Math.abs(y1 - y0),
);
if (seg) segments.push(seg);
}
} else if (mode === "stroke" && strokeWidth <= MAX_THICKNESS) {
for (const r of pendingRects) {
const [x0, y0] = ctmApply(ctm, r.x, r.y);
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
pushStrokedRectEdges(
segments,
sid(),
Math.min(x0, x1),
Math.min(y0, y1),
Math.abs(x1 - x0),
Math.abs(y1 - y0),
);
}
for (const l of pendingLines) {
const [lx1, ly1] = ctmApply(ctm, l.x1, l.y1);
const [lx2, ly2] = ctmApply(ctm, l.x2, l.y2);
const dx = Math.abs(lx2 - lx1);
const dy = Math.abs(ly2 - ly1);
// Only keep H/V lines
if ((dx >= MIN_LENGTH && dy < 1) || (dy >= MIN_LENGTH && dx < 1)) {
segments.push({ id: sid(), x1: lx1, y1: ly1, x2: lx2, y2: ly2 });
}
}
}
pendingRects.length = 0;
pendingLines.length = 0;
}
while (idx < tokens.length) {
const t = tokens[idx];
if (t === "q") {
stateStack.push({ ctm: [...ctm], strokeWidth });
} else if (t === "Q") {
const saved = stateStack.pop();
if (saved) {
ctm = saved.ctm;
strokeWidth = saved.strokeWidth;
}
} else if (t === "cm" && idx >= 6) {
const a = Number(tokens[idx - 6]);
const b = Number(tokens[idx - 5]);
const c = Number(tokens[idx - 4]);
const d = Number(tokens[idx - 3]);
const e = Number(tokens[idx - 2]);
const f = Number(tokens[idx - 1]);
ctm = ctmConcat(ctm, [a, b, c, d, e, f]);
} else if (t === "w" && idx >= 1) {
strokeWidth = Number(tokens[idx - 1]) || strokeWidth;
} else if (t === "re" && idx >= 4) {
const x = Number(tokens[idx - 4]);
const y = Number(tokens[idx - 3]);
const w = Number(tokens[idx - 2]);
const h = Number(tokens[idx - 1]);
if (Number.isFinite(x + y + w + h)) {
pendingRects.push({ x, y, w, h });
}
} else if (t === "m" && idx >= 2) {
curX = Number(tokens[idx - 2]);
curY = Number(tokens[idx - 1]);
pathStartX = curX;
pathStartY = curY;
} else if (t === "l" && idx >= 2) {
const x2 = Number(tokens[idx - 2]);
const y2 = Number(tokens[idx - 1]);
pendingLines.push({ x1: curX, y1: curY, x2, y2 });
curX = x2;
curY = y2;
} else if (t === "h") {
// closePath: line back to start
if (curX !== pathStartX || curY !== pathStartY) {
pendingLines.push({
x1: curX,
y1: curY,
x2: pathStartX,
y2: pathStartY,
});
}
curX = pathStartX;
curY = pathStartY;
} else if (t === "f" || t === "F" || t === "f*") {
flushPath("fill");
} else if (t === "S" || t === "s") {
if (t === "s") {
// closeStroke: implicit closePath
if (curX !== pathStartX || curY !== pathStartY) {
pendingLines.push({
x1: curX,
y1: curY,
x2: pathStartX,
y2: pathStartY,
});
}
}
flushPath("stroke");
} else if (t === "B" || t === "B*" || t === "b" || t === "b*") {
// fill + stroke combined
flushPath("fill");
flushPath("stroke");
} else if (t === "n") {
// end path without painting — discard
pendingRects.length = 0;
pendingLines.length = 0;
}
idx++;
}
return segments;
}
/**
* Fast tokenizer for PDF content streams.
* Splits on whitespace, skipping comments, string literals, and inline image payloads.
*/
function tokenizeContentStream(raw: string): string[] {
const tokens: string[] = [];
const len = raw.length;
let i = 0;
let inInlineImage = false;
while (i < len) {
const ch = raw.charCodeAt(i);
// Skip whitespace
if (ch <= 32) {
i++;
continue;
}
// Skip comments
if (ch === 37 /* % */) {
while (i < len && raw.charCodeAt(i) !== 10) i++;
continue;
}
// Skip string literals (...)
if (ch === 40 /* ( */) {
let depth = 1;
i++;
while (i < len && depth > 0) {
const c = raw.charCodeAt(i);
if (c === 92 /* \ */) {
i++;
} else if (c === 40) {
depth++;
} else if (c === 41) {
depth--;
}
i++;
}
continue;
}
// Skip hex strings <...>
if (ch === 60 /* < */ && i + 1 < len && raw.charCodeAt(i + 1) !== 60) {
i++;
while (i < len && raw.charCodeAt(i) !== 62) i++;
i++; // skip >
continue;
}
// Skip dict delimiters << >>
if (ch === 60 && i + 1 < len && raw.charCodeAt(i + 1) === 60) {
i += 2;
continue;
}
if (ch === 62 && i + 1 < len && raw.charCodeAt(i + 1) === 62) {
i += 2;
continue;
}
// Skip stray closing delimiters from malformed streams. They cannot start
// a token, so leaving i unchanged would spin forever.
if (ch === 41 || ch === 62) {
i++;
continue;
}
// Regular token: read until whitespace or delimiter
const start = i;
while (i < len) {
const c = raw.charCodeAt(i);
if (c <= 32 || c === 40 || c === 41 || c === 60 || c === 62 || c === 37) break;
i++;
}
if (i > start) {
const token = raw.substring(start, i);
tokens.push(token);
if (token === "BI") {
inInlineImage = true;
} else if (token === "ID" && inInlineImage) {
while (i < len && raw.charCodeAt(i) <= 32) i++;
while (i < len) {
const c = raw.charCodeAt(i);
const prev = i === 0 ? 32 : raw.charCodeAt(i - 1);
const next = i + 2 >= len ? 32 : raw.charCodeAt(i + 2);
if (c === 69 && raw.charCodeAt(i + 1) === 73 && prev <= 32 && next <= 32) {
i += 2;
break;
}
i++;
}
inInlineImage = false;
}
}
}
return tokens;
}
// ---------------------------------------------------------------------------
// Image region detection
// ---------------------------------------------------------------------------
/** Minimum area (pts²) for an image to be considered a diagram, not an icon. */
const MIN_IMAGE_AREA = 5000;
function extractImageRegions(stext: StructuredTextJSON, pageNumber: number, pageHeight: number): ImageRegion[] {
const regions: ImageRegion[] = [];
for (const block of stext.blocks) {
if (block.type !== "image") continue;
const { x, y, w, h } = block.bbox;
if (w * h < MIN_IMAGE_AREA) continue; // skip tiny icons
// Convert Y from mupdf (top-left) to PDF (bottom-left) for ordering
const pdfTopY = pageHeight - y;
regions.push({
id: `p${pageNumber}-img${regions.length}`,
pageNumber,
bbox: { x, y, w, h },
topY: pdfTopY,
});
}
return regions;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/**
* Render an image region from a PDF page as a PNG buffer.
* Uses mupdf's DrawDevice to render just the cropped area at 2x resolution.
*/
export async function renderImageRegion(input: Uint8Array, region: ImageRegion): Promise<Uint8Array> {
const m = await loadMupdf();
const doc = m.Document.openDocument(input, "application/pdf");
const page = doc.loadPage(region.pageNumber - 1);
const pad = 10;
const bx = region.bbox.x - pad;
const by = region.bbox.y - pad;
const bw = region.bbox.w + 2 * pad;
const bh = region.bbox.h + 2 * pad;
const scale = 2;
const pw = Math.round(bw * scale);
const ph = Math.round(bh * scale);
const pix = new m.Pixmap(m.ColorSpace.DeviceRGB, [0, 0, pw, ph], false);
pix.clear(255);
const matrix: mupdf.Matrix = [scale, 0, 0, scale, -bx * scale, -by * scale];
const dl = page.toDisplayList();
const dev = new m.DrawDevice(matrix, pix);
dl.run(dev, m.Matrix.identity);
dev.close();
return pix.asPNG();
}
/**
* Extract text boxes and vector segments from all pages of a PDF buffer.
*/
export async function extractPages(input: Uint8Array): Promise<PageContent[]> {
const m = await loadMupdf();
const doc = m.Document.openDocument(input, "application/pdf");
const pages: PageContent[] = [];
for (let i = 0; i < doc.countPages(); i++) {
const pageNumber = i + 1;
const page = doc.loadPage(i);
const bounds = page.getBounds();
const pageHeight = bounds[3] - bounds[1];
// Single structured text pass with both flags
const stext = JSON.parse(
page.toStructuredText("preserve-whitespace,preserve-images").asJSON(),
) as StructuredTextJSON;
// Extract text boxes and image regions from the same parse
const textBoxes = extractTextBoxes(page, pageNumber, pageHeight, stext);
const images = extractImageRegions(stext, pageNumber, pageHeight);
// Extract vector segments from raw content stream
let segments: Segment[] = [];
try {
const pageObj = (page as mupdf.PDFPage).getObject();
const contents = pageObj.get("Contents");
if (contents) {
let rawBytes: Uint8Array;
if (contents.isArray()) {
// Multiple content streams — concatenate
const parts: Uint8Array[] = [];
const len = contents.length ?? 0;
for (let j = 0; j < len; j++) {
const stream = contents.get(j);
if (stream?.readStream) {
parts.push(stream.readStream().asUint8Array());
}
}
const totalLen = parts.reduce((s, p) => s + p.length, 0);
rawBytes = new Uint8Array(totalLen);
let offset = 0;
for (const part of parts) {
rawBytes.set(part, offset);
offset += part.length;
}
} else {
rawBytes = contents.readStream().asUint8Array();
}
const raw = new TextDecoder().decode(rawBytes);
segments = extractSegmentsFromContentStream(raw, pageNumber);
}
} catch {
// Content stream extraction failed — proceed with text only
}
pages.push({ pageNumber, textBoxes, segments, images });
}
return pages;
}
@@ -1,780 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Table grid detection from vector segments and text boxes.
*
* Ported from @oharato/pdf2md-ts with TypeScript types and without
* CJK-specific borderless table heuristics. The core algorithm:
*
* 1. Classify segments as horizontal or vertical lines
* 2. Group horizontal Y-lines into table groups (split by vertical gaps)
* 3. For each group:
* a. Full grid (H+V lines): build cells from grid intersections,
* place text via raycasting
* b. H-line only (no V lines): infer columns from text X positions
* 4. Prune empty rows/cols
*
* Coordinate system: PDF native (bottom-left origin, Y increases upward).
*/
import type { Segment, TableCell, TableGrid, TextBox } from "./types";
export interface GridResult {
grids: TableGrid[];
consumedIds: string[];
}
type RayDirection = "up" | "down" | "left" | "right";
interface Ray {
direction: RayDirection;
segmentId: string | null;
distance: number;
}
interface Interval {
min: number;
max: number;
}
function castRaysForTextBox(textBox: TextBox, segments: Segment[]): Ray[] {
const cx = (textBox.bounds.left + textBox.bounds.right) / 2;
const cy = (textBox.bounds.top + textBox.bounds.bottom) / 2;
let up: Ray = { direction: "up", segmentId: null, distance: Infinity };
let down: Ray = { direction: "down", segmentId: null, distance: Infinity };
let left: Ray = { direction: "left", segmentId: null, distance: Infinity };
let right: Ray = {
direction: "right",
segmentId: null,
distance: Infinity,
};
for (const seg of segments) {
const isH = Math.abs(seg.y1 - seg.y2) < 0.5;
const isV = Math.abs(seg.x1 - seg.x2) < 0.5;
if (isH) {
const minX = Math.min(seg.x1, seg.x2);
const maxX = Math.max(seg.x1, seg.x2);
if (cx >= minX && cx <= maxX) {
const d = seg.y1 - cy;
if (d >= 0 && d < up.distance) up = { direction: "up", segmentId: seg.id, distance: d };
const dd = cy - seg.y1;
if (dd >= 0 && dd < down.distance) down = { direction: "down", segmentId: seg.id, distance: dd };
}
}
if (isV) {
const minY = Math.min(seg.y1, seg.y2);
const maxY = Math.max(seg.y1, seg.y2);
if (cy >= minY && cy <= maxY) {
const d = cx - seg.x1;
if (d >= 0 && d < left.distance) left = { direction: "left", segmentId: seg.id, distance: d };
const rd = seg.x1 - cx;
if (rd >= 0 && rd < right.distance) right = { direction: "right", segmentId: seg.id, distance: rd };
}
}
}
return [up, down, left, right];
}
// ---------------------------------------------------------------------------
// Utility
// ---------------------------------------------------------------------------
const AXIS_EPSILON = 0.8;
const PAGE_MARGIN = 20;
function uniqueSorted(values: number[]): number[] {
const sorted = [...values].sort((a, b) => a - b);
const result: number[] = [];
for (const v of sorted) {
if (result.length === 0 || Math.abs(result[result.length - 1] - v) > 1) result.push(v);
}
return result;
}
// ---------------------------------------------------------------------------
// Y-line group splitting
// ---------------------------------------------------------------------------
function chainCoversRange(intervals: Interval[], lowerY: number, upperY: number, eps: number): boolean {
const sorted = [...intervals].sort((a, b) => a.min - b.min);
let covered = lowerY;
for (const iv of sorted) {
if (iv.min > covered + eps) break;
if (iv.max > covered) covered = iv.max;
if (covered >= upperY - eps) return true;
}
return false;
}
function countBridgingVLineCols(upperY: number, lowerY: number, verticals: Segment[]): number {
const eps = 1.5;
const byX = new Map<number, Interval[]>();
for (const seg of verticals) {
const rx = Math.round(seg.x1);
if (!byX.has(rx)) byX.set(rx, []);
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
}
let count = 0;
for (const intervals of byX.values()) {
if (chainCoversRange(intervals, lowerY, upperY, eps)) count++;
}
return count;
}
function bridgingXSet(upperY: number, lowerY: number, verticals: Segment[]): Set<number> {
const eps = 1.5;
const xs = new Set<number>();
const byX = new Map<number, Interval[]>();
for (const seg of verticals) {
const rx = Math.round(seg.x1);
if (!byX.has(rx)) byX.set(rx, []);
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
}
for (const [rx, intervals] of byX) {
if (chainCoversRange(intervals, lowerY, upperY, eps)) xs.add(rx);
}
return xs;
}
const MIN_RICH_BRIDGING_COLS = 3;
function splitYLinesIntoGroups(yLines: number[], verticals: Segment[]): number[][] {
if (yLines.length === 0) return [];
const eps = 1.5;
const allX = verticals.map(s => Math.round(s.x1));
const globalXMin = allX.length > 0 ? Math.min(...allX) : 0;
const globalXMax = allX.length > 0 ? Math.max(...allX) : 0;
const groups: number[][] = [];
let currentGroup = [yLines[0]];
let prevBridgingCols = -1;
for (let i = 1; i < yLines.length; i++) {
const upperY = yLines[i - 1];
const lowerY = yLines[i];
const cols = countBridgingVLineCols(upperY, lowerY, verticals);
if (cols === 0) {
groups.push(currentGroup);
currentGroup = [yLines[i]];
prevBridgingCols = -1;
continue;
}
if (prevBridgingCols >= MIN_RICH_BRIDGING_COLS && cols < MIN_RICH_BRIDGING_COLS) {
const bxs = bridgingXSet(upperY, lowerY, verticals);
const isOuterFrameOnly = [...bxs].every(
x => Math.abs(x - globalXMin) <= eps || Math.abs(x - globalXMax) <= eps,
);
if (!isOuterFrameOnly) {
groups.push(currentGroup);
currentGroup = [yLines[i - 1], yLines[i]];
prevBridgingCols = cols;
continue;
}
}
currentGroup.push(yLines[i]);
prevBridgingCols = cols;
}
groups.push(currentGroup);
return groups;
}
// ---------------------------------------------------------------------------
// Sub-row Y-cluster expansion
// ---------------------------------------------------------------------------
const Y_CLUSTER_GAP = 10;
const MIN_COLS_IN_TOP_CLUSTER = 2;
function assignToYCluster(y: number, clusters: number[]): number {
let closest = 0;
let closestDist = Math.abs(y - clusters[0]);
for (let k = 1; k < clusters.length; k++) {
const d = Math.abs(y - clusters[k]);
if (d < closestDist) {
closestDist = d;
closest = k;
}
}
return closest;
}
function expandSubRowsByYClusters(
originalRows: number,
cols: number,
cells: TableCell[],
cellBoxes: Map<TableCell, TextBox[]>,
): number {
let addedRows = 0;
for (let origRow = 0; origRow < originalRows; origRow++) {
const currentRow = origRow + addedRows;
const rowCellInfos: Array<{ cell: TableCell; col: number; boxes: TextBox[] }> = [];
for (let col = 0; col < cols; col++) {
const cell = cells.find(c => c.row === currentRow && c.col === col);
if (!cell) continue;
const boxes = cellBoxes.get(cell);
if (boxes && boxes.length > 0) rowCellInfos.push({ cell, col, boxes });
}
if (rowCellInfos.length === 0) continue;
const allMidYs = rowCellInfos.flatMap(({ boxes }) => boxes.map(b => (b.bounds.top + b.bounds.bottom) / 2));
const sortedY = [...new Set(allMidYs.map(y => Math.round(y * 10) / 10))].sort((a, b) => b - a);
const clusters = [sortedY[0]];
for (let i = 1; i < sortedY.length; i++) {
if (clusters[clusters.length - 1] - sortedY[i] > Y_CLUSTER_GAP) {
clusters.push(sortedY[i]);
}
}
if (clusters.length < 2) continue;
const colsInTopCluster = new Set<number>();
const totalNonEmptyCols = new Set<number>();
for (const { col, boxes } of rowCellInfos) {
totalNonEmptyCols.add(col);
if (boxes.some(b => assignToYCluster((b.bounds.top + b.bounds.bottom) / 2, clusters) === 0)) {
colsInTopCluster.add(col);
}
}
if (colsInTopCluster.size < MIN_COLS_IN_TOP_CLUSTER) continue;
if (colsInTopCluster.size >= totalNonEmptyCols.size) continue;
const sparseColsHaveMultipleBoxes = rowCellInfos.some(
({ col, boxes }) => !colsInTopCluster.has(col) && boxes.length > 1,
);
if (!sparseColsHaveMultipleBoxes) continue;
const numSubRows = clusters.length;
const numNewRows = numSubRows - 1;
for (const cell of cells) {
if (cell.row > currentRow) cell.row += numNewRows;
}
for (let subRow = 1; subRow < numSubRows; subRow++) {
for (let col = 0; col < cols; col++) {
cells.push({
row: currentRow + subRow,
col,
text: "",
rowSpan: 1,
colSpan: 1,
});
}
}
for (const { cell: origCell, col, boxes } of rowCellInfos) {
const subRowBoxGroups: TextBox[][] = Array.from({ length: numSubRows }, () => []);
for (const box of boxes) {
const cy = (box.bounds.top + box.bounds.bottom) / 2;
subRowBoxGroups[assignToYCluster(cy, clusters)].push(box);
}
cellBoxes.set(origCell, subRowBoxGroups[0]);
if (subRowBoxGroups[0].length === 0) cellBoxes.delete(origCell);
for (let subRow = 1; subRow < numSubRows; subRow++) {
if (subRowBoxGroups[subRow].length > 0) {
const newCell = cells.find(c => c.row === currentRow + subRow && c.col === col);
if (newCell) cellBoxes.set(newCell, subRowBoxGroups[subRow]);
}
}
}
addedRows += numNewRows;
}
return originalRows + addedRows;
}
// ---------------------------------------------------------------------------
// Cross-column text box splitting
// ---------------------------------------------------------------------------
/**
* Find which column a horizontal position falls into.
* Returns -1 if outside the grid.
*/
function findCol(x: number, xLines: number[]): number {
for (let i = 0; i < xLines.length - 1; i++) {
if (x >= xLines[i] && x <= xLines[i + 1]) return i;
}
return -1;
}
/**
* When a text box spans across one or more vertical column boundaries,
* split it into multiple virtual text boxes — one per column — with the
* text divided proportionally by width.
*
* We split at word boundaries closest to the proportional split point
* so we don't chop words in half.
*/
function splitCrossColumnBoxes(textBoxes: TextBox[], xLines: number[]): TextBox[] {
const result: TextBox[] = [];
const MARGIN = 5; // allow small overlap before considering it cross-column
for (const tb of textBoxes) {
const leftCol = findCol(tb.bounds.left + MARGIN, xLines);
const rightCol = findCol(tb.bounds.right - MARGIN, xLines);
// Not spanning columns, or outside grid — keep as-is
if (leftCol < 0 || rightCol < 0 || leftCol === rightCol) {
result.push(tb);
continue;
}
// Text box spans from leftCol to rightCol — split it
const totalWidth = tb.bounds.right - tb.bounds.left;
if (totalWidth <= 0) {
result.push(tb);
continue;
}
const words = tb.text.split(/\s+/);
if (words.length <= 1) {
// Single word spanning columns — just assign to whichever col has more overlap
result.push(tb);
continue;
}
// For each column boundary crossing, find the best word-boundary split
let remainingWords = [...words];
let currentLeft = tb.bounds.left;
for (let col = leftCol; col <= rightCol && remainingWords.length > 0; col++) {
const colRight = col < xLines.length - 1 ? xLines[col + 1] : tb.bounds.right;
const segmentRight = Math.min(colRight, tb.bounds.right);
if (col === rightCol) {
// Last column — take all remaining words
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: remainingWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: tb.bounds.right,
},
});
remainingWords = [];
} else {
// Find how many words fit in this column segment proportionally
const segmentWidth = segmentRight - currentLeft;
const fractionOfTotal = segmentWidth / totalWidth;
const approxChars = Math.round(fractionOfTotal * tb.text.length);
// Walk words to find the split closest to the proportional point
let charCount = 0;
let splitIdx = 0;
for (let w = 0; w < remainingWords.length; w++) {
const nextCount = charCount + remainingWords[w].length + (w > 0 ? 1 : 0);
if (nextCount > approxChars && splitIdx > 0) break;
charCount = nextCount;
splitIdx = w + 1;
}
if (splitIdx === 0) splitIdx = 1; // take at least one word
if (splitIdx >= remainingWords.length) {
// All remaining words fit here
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: remainingWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: segmentRight,
},
});
remainingWords = [];
} else {
const partWords = remainingWords.slice(0, splitIdx);
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: partWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: segmentRight,
},
});
remainingWords = remainingWords.slice(splitIdx);
currentLeft = segmentRight;
}
}
}
}
return result;
}
// ---------------------------------------------------------------------------
// Full grid table (H + V lines)
// ---------------------------------------------------------------------------
function buildCells(rows: number, cols: number): TableCell[] {
const cells: TableCell[] = [];
for (let row = 0; row < rows; row++) {
for (let col = 0; col < cols; col++) {
cells.push({ row, col, text: "", rowSpan: 1, colSpan: 1 });
}
}
return cells;
}
function buildTableGrid(
pageNumber: number,
yLines: number[],
xLines: number[],
filteredSegments: Segment[],
textBoxes: TextBox[],
): { grid: TableGrid; consumedIds: string[] } {
let rows = yLines.length - 1;
const cols = xLines.length - 1;
const cells = buildCells(rows, cols);
const consumedIds: string[] = [];
const yMin = yLines[yLines.length - 1];
const yMax = yLines[0];
const xMin = xLines[0];
const xMax = xLines[xLines.length - 1];
// Split text boxes that span multiple columns before placement
const splitBoxes = splitCrossColumnBoxes(textBoxes, xLines);
// Track which split piece IDs get placed in cells, so we can consume
// the original (unsplit) text box IDs too.
const placedSplitIds = new Set<string>();
// Look for header text boxes just above the grid.
// Use the ORIGINAL (unsplit) text boxes for header detection so that
// wide paragraph text isn't falsely split into column-sized header chunks.
// Reject boxes wider than 1.5 columns — those are paragraph text, not headers.
const avgColWidth = (xMax - xMin) / cols;
const maxHeaderBoxWidth = avgColWidth * 1.5;
const headerBoxes = textBoxes.filter(tb => {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const boxWidth = tb.bounds.right - tb.bounds.left;
return cy > yMax && cy <= yMax + 20 && cx >= xMin && cx <= xMax && boxWidth <= maxHeaderBoxWidth;
});
if (headerBoxes.length > 0) {
rows += 1;
for (const cell of cells) cell.row += 1;
for (let col = 0; col < cols; col++) {
cells.push({ row: 0, col, text: "", rowSpan: 1, colSpan: 1 });
}
for (const tb of headerBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col >= 0 && col < cols) {
const cell = cells.find(c => c.row === 0 && c.col === col);
if (cell) {
cell.text = cell.text.length === 0 ? tb.text : `${cell.text} ${tb.text}`;
consumedIds.push(tb.id);
}
}
}
}
const cellBoxes = new Map<TableCell, TextBox[]>();
for (const tb of splitBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
if (cy < yMin || cy > yMax || cx < xMin || cx > xMax) continue;
const rays = castRaysForTextBox(tb, filteredSegments);
const rayConfidence = rays.filter(r => r.segmentId !== null).length;
let row = yLines.findIndex((lineY, idx) => {
const next = yLines[idx + 1];
return next !== undefined && cy <= lineY && cy >= next;
});
if (row < 0 || row >= (headerBoxes.length > 0 ? rows - 1 : rows)) continue;
if (headerBoxes.length > 0) row += 1;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col < 0 || col >= cols) continue;
if (rayConfidence === 0) continue;
const cell = cells.find(c => c.row === row && c.col === col);
if (!cell) continue;
if (!cellBoxes.has(cell)) cellBoxes.set(cell, []);
cellBoxes.get(cell)?.push(tb);
consumedIds.push(tb.id);
if (tb.id.includes("-split")) placedSplitIds.add(tb.id);
}
rows = expandSubRowsByYClusters(rows, cols, cells, cellBoxes);
// Merge text boxes within each cell into cell text
for (const [cell, boxes] of cellBoxes.entries()) {
boxes.sort((a, b) => b.bounds.top - a.bounds.top);
const lines: string[] = [];
let currentLine: string[] = [];
let currentY = boxes[0].bounds.top;
for (const box of boxes) {
if (Math.abs(box.bounds.top - currentY) > 5) {
lines.push(currentLine.join(" "));
currentLine = [box.text];
currentY = box.bounds.top;
} else {
currentLine.push(box.text);
}
}
if (currentLine.length > 0) lines.push(currentLine.join(" "));
cell.text = lines.join("<br>");
}
const grid = pruneEmptyRowsAndCols({
pageNumber,
rows,
cols,
cells,
warnings: [],
topY: yLines[0],
isBorderless: false,
});
// Also consume the original (unsplit) text box IDs when any of their
// split pieces were placed in a cell.
for (const splitId of placedSplitIds) {
const origId = splitId.replace(/-split\d+$/, "");
if (!consumedIds.includes(origId)) {
consumedIds.push(origId);
}
}
return { grid, consumedIds };
}
// ---------------------------------------------------------------------------
// H-line-only table (inferred columns)
// ---------------------------------------------------------------------------
const COL_GAP_THRESHOLD = 20;
const HONLY_ROW_GAP = 30;
const HONLY_ROW_TOLERANCE = 8;
const MIN_TABLE_HEIGHT = 24;
const MIN_LEFT_SPREAD = 50;
function inferXLinesFromBoxes(textBoxes: TextBox[], xMin: number, xMax: number): number[] {
const centers = textBoxes.map(tb => (tb.bounds.left + tb.bounds.right) / 2).sort((a, b) => a - b);
if (centers.length === 0) return [xMin, xMax];
const boundaries = [xMin];
for (let i = 1; i < centers.length; i++) {
if (centers[i] - centers[i - 1] >= COL_GAP_THRESHOLD) {
boundaries.push((centers[i - 1] + centers[i]) / 2);
}
}
boundaries.push(xMax);
return boundaries;
}
function buildHLineOnlyTable(
pageNumber: number,
yLines: number[],
xMin: number,
xMax: number,
textBoxes: TextBox[],
alreadyConsumed: Set<string>,
): { grid: TableGrid; consumedIds: string[] } | null {
const yMax = yLines[0];
const yMin = yLines[yLines.length - 1];
const candidates = textBoxes.filter(tb => !alreadyConsumed.has(tb.id));
const BOX_LEFT_TOLERANCE = 30;
const inRange = candidates.filter(tb => {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
return (
tb.bounds.left >= xMin - BOX_LEFT_TOLERANCE &&
tb.bounds.right <= xMax + BOX_LEFT_TOLERANCE &&
cy >= yMin &&
cy <= yMax
);
});
// Extend downward below yMin
const belowYMin = candidates
.filter(tb => {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
return cx >= xMin && cx <= xMax && cy < yMin;
})
.sort((a, b) => (b.bounds.top + b.bounds.bottom) / 2 - (a.bounds.top + a.bounds.bottom) / 2);
const extensionBoxes: TextBox[] = [];
let lastY = yMin;
for (const tb of belowYMin) {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
if (lastY - cy > HONLY_ROW_GAP) break;
extensionBoxes.push(tb);
lastY = cy;
}
const allBoxes = [...inRange, ...extensionBoxes];
if (allBoxes.length === 0) return null;
const leftEdges = allBoxes.map(tb => tb.bounds.left);
if (Math.max(...leftEdges) - Math.min(...leftEdges) < MIN_LEFT_SPREAD) return null;
const xLines = inferXLinesFromBoxes(allBoxes, xMin, xMax);
if (xLines.length < 2) return null;
const cols = xLines.length - 1;
// Build visual rows
const visualRows: Array<{ midY: number; boxes: TextBox[] }> = [];
const sortedBoxes = [...allBoxes].sort((a, b) => {
const ya = (a.bounds.top + a.bounds.bottom) / 2;
const yb = (b.bounds.top + b.bounds.bottom) / 2;
if (Math.abs(ya - yb) > 0.5) return yb - ya;
return a.bounds.left - b.bounds.left;
});
for (const box of sortedBoxes) {
const cy = (box.bounds.top + box.bounds.bottom) / 2;
const last = visualRows[visualRows.length - 1];
if (last && Math.abs(last.midY - cy) <= HONLY_ROW_TOLERANCE) {
last.boxes.push(box);
} else {
visualRows.push({ midY: cy, boxes: [box] });
}
}
if (visualRows.length === 0) return null;
const cells: TableCell[] = [];
const consumedIds: string[] = [];
for (let rowIdx = 0; rowIdx < visualRows.length; rowIdx++) {
const vrow = visualRows[rowIdx];
const colBoxes = new Map<number, TextBox[]>();
for (const box of vrow.boxes) {
const cx = (box.bounds.left + box.bounds.right) / 2;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col >= 0 && col < cols) {
if (!colBoxes.has(col)) colBoxes.set(col, []);
colBoxes.get(col)?.push(box);
}
}
for (let c = 0; c < cols; c++) {
const cbs = (colBoxes.get(c) ?? []).sort((a, b) => a.bounds.left - b.bounds.left);
cells.push({
row: rowIdx,
col: c,
text: cbs.map(b => b.text).join(" "),
rowSpan: 1,
colSpan: 1,
});
consumedIds.push(...cbs.map(b => b.id));
}
}
const contentTopY = visualRows.length > 0 ? visualRows[0].midY : yMax;
const grid = pruneEmptyRowsAndCols({
pageNumber,
rows: visualRows.length,
cols,
cells,
warnings: [],
topY: contentTopY,
isBorderless: false,
});
return { grid, consumedIds };
}
// ---------------------------------------------------------------------------
// Pruning
// ---------------------------------------------------------------------------
function pruneEmptyRowsAndCols(table: TableGrid): TableGrid {
const occupiedRows = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.row));
const occupiedCols = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.col));
if (occupiedRows.size === 0) return table;
const rowMap = new Map<number, number>();
let newRow = 0;
for (let r = 0; r < table.rows; r++) {
if (occupiedRows.has(r)) rowMap.set(r, newRow++);
}
const colMap = new Map<number, number>();
let newCol = 0;
for (let c = 0; c < table.cols; c++) {
if (occupiedCols.has(c)) colMap.set(c, newCol++);
}
const prunedCells = table.cells
.filter(c => occupiedRows.has(c.row) && occupiedCols.has(c.col))
.map(c => ({
...c,
row: rowMap.get(c.row) ?? c.row,
col: colMap.get(c.col) ?? c.col,
}));
return { ...table, rows: newRow, cols: newCol, cells: prunedCells };
}
// ---------------------------------------------------------------------------
// Diagram vs table discrimination
// ---------------------------------------------------------------------------
/** Maximum column count for a plausible data table. */
const MAX_TABLE_COLS = 25;
/**
* Returns true if a grid looks like a vector diagram rather than a data table.
*
* Heuristics (any match → diagram):
* 1. Column count > 25 (diagrams create many X-lines from box edges)
* 2. Fill ratio < 25% (most cells empty — scattered boxes)
* 3. Fill < 50% AND duplicate text ratio > 30% (repeating labels in a
* diagram layout, e.g. "Hash", "Transaction" appearing in each column)
* 4. Fill < 50% AND cols >= 6 (moderate sparseness with wide grid)
*/
function isDiagram(grid: TableGrid): boolean {
const totalCells = grid.rows * grid.cols;
if (totalCells === 0) return true;
const filled = grid.cells.filter(c => c.text.trim().length > 0);
const fillRatio = filled.length / totalCells;
// Very high column count
if (grid.cols > MAX_TABLE_COLS) return true;
// Very sparse
if (fillRatio < 0.25) return true;
// Compute duplicate text ratio among non-trivial cells.
// Exclude short values (≤3 chars) like "—", "V", "YES", "NO" which
// naturally repeat in real data tables.
const substantive = filled.filter(c => c.text.trim().length > 3);
const uniqueTexts = new Set(substantive.map(c => c.text.trim())).size;
const dupRatio = substantive.length > 2 ? 1 - uniqueTexts / substantive.length : 0;
// Sparse + highly duplicated substantive text → repeating diagram
if (fillRatio < 0.5 && dupRatio > 0.3) return true;
// High duplication + wide grid → repeating diagram even at moderate fill
if (dupRatio > 0.4 && grid.cols >= 6) return true;
// Sparse + wide grid with no substantive text to judge
if (fillRatio < 0.4 && grid.cols >= 6) return true;
return false;
}
/**
* Detect all table grids on a single page from its text boxes and segments.
*/
export function resolveTableGrids(pageNumber: number, textBoxes: TextBox[], segments: Segment[]): GridResult {
const vertical = segments.filter(s => Math.abs(s.x1 - s.x2) <= AXIS_EPSILON);
const horizontal = segments.filter(s => Math.abs(s.y1 - s.y2) <= AXIS_EPSILON);
// Filter segments to the text's visible area
const textYValues = textBoxes.flatMap(t => [t.bounds.bottom, t.bounds.top]);
const textYMin = textYValues.length > 0 ? Math.min(...textYValues) - PAGE_MARGIN : -Infinity;
const textYMax = textYValues.length > 0 ? Math.max(...textYValues) + PAGE_MARGIN : Infinity;
const textXValues = textBoxes.flatMap(t => [t.bounds.left, t.bounds.right]);
const textXMin = textXValues.length > 0 ? Math.min(...textXValues) - 100 : -Infinity;
const textXMax = textXValues.length > 0 ? Math.max(...textXValues) + 100 : Infinity;
const filteredH = horizontal.filter(
s => s.y1 >= textYMin && s.y1 <= textYMax && s.x1 <= textXMax && s.x2 >= textXMin,
);
const hMaxX2 = filteredH.length > 0 ? Math.max(...filteredH.map(s => s.x2)) : textXMax;
const vSegXMax = Math.max(textXMax, hMaxX2 + PAGE_MARGIN);
const filteredV = vertical.filter(s => {
const segMin = Math.min(s.y1, s.y2);
const segMax = Math.max(s.y1, s.y2);
return segMax >= textYMin && segMin <= textYMax && s.x1 >= textXMin && s.x1 <= vSegXMax;
});
const allYLines = uniqueSorted(filteredH.flatMap(s => [s.y1, s.y2])).sort((a, b) => b - a);
if (allYLines.length < 2) {
return { grids: [], consumedIds: [] };
}
const filteredSegments = [...filteredH, ...filteredV];
const yGroups = splitYLinesIntoGroups(allYLines, filteredV);
const grids: TableGrid[] = [];
const gridConsumedIds: string[][] = [];
// Flat set for the alreadyConsumed check in H-line-only tables
const allConsumedIds: string[] = [];
for (const yLines of yGroups) {
if (yLines.length < 2) continue;
const yMin = yLines[yLines.length - 1];
const yMax = yLines[0];
const groupVerticals = filteredV.filter(s => {
const segMin = Math.min(s.y1, s.y2);
const segMax = Math.max(s.y1, s.y2);
return segMin < yMax - 1.5 && segMax > yMin + 1.5;
});
const groupXLines = uniqueSorted(groupVerticals.flatMap(s => [s.x1, s.x2]));
if (groupXLines.length < 2) {
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
const groupHoriz = filteredH.filter(s => s.y1 >= yMin - 1.5 && s.y1 <= yMax + 1.5);
if (groupHoriz.length === 0) continue;
const hxMin = Math.min(...groupHoriz.map(s => s.x1));
const hxMax = Math.max(...groupHoriz.map(s => s.x2));
const result = buildHLineOnlyTable(pageNumber, yLines, hxMin, hxMax, textBoxes, new Set(allConsumedIds));
if (result) {
grids.push(result.grid);
gridConsumedIds.push(result.consumedIds);
allConsumedIds.push(...result.consumedIds);
}
continue;
}
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
const result = buildTableGrid(pageNumber, yLines, groupXLines, filteredSegments, textBoxes);
grids.push(result.grid);
gridConsumedIds.push(result.consumedIds);
allConsumedIds.push(...result.consumedIds);
}
// Filter out grids that look like vector diagrams, not data tables.
// Their consumed text box IDs are released so the text becomes free text.
const filteredGrids: TableGrid[] = [];
const filteredConsumedIds: string[] = [];
for (let i = 0; i < grids.length; i++) {
if (isDiagram(grids[i])) continue;
filteredGrids.push(grids[i]);
filteredConsumedIds.push(...gridConsumedIds[i]);
}
return { grids: filteredGrids, consumedIds: filteredConsumedIds };
}
@@ -1,106 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Running header/footer detection and removal.
*
* Many PDFs have repeated text at the top or bottom of every page:
* document titles, chapter names, page numbers, copyright notices.
* These pollute the markdown output as false headings or noise.
*
* Algorithm:
* 1. For each page, bucket text boxes by Y position (top/bottom zones)
* 2. Collect the text content at each zone across all pages
* 3. Text appearing on >20% of pages OR 8+ consecutive pages is a
* running header/footer
* 4. Remove matching text boxes before further processing
*/
import type { PageContent } from "./types";
/** Minimum number of pages to enable header/footer detection. */
const MIN_PAGES = 5;
/** Minimum Y position for top zone (from bottom of page in PDF coords). */
const TOP_ZONE_MIN_Y = 700;
/** Maximum Y position for bottom zone. */
const BOTTOM_ZONE_MAX_Y = 80;
/**
* Minimum consecutive pages a text must appear on to be considered a
* running header/footer. Catches both document-wide headers (appearing
* on every page) and chapter-specific headers (appearing on 4+ consecutive
* pages within a chapter).
*/
const MIN_CONSECUTIVE_PAGES = 8;
/**
* Detect and remove running headers and footers from all pages.
* Mutates the pages array in place, removing header/footer text boxes.
*
* Uses two strategies:
* 1. Global frequency: text appearing on > 20% of all pages
* 2. Consecutive runs: text appearing on 8+ consecutive pages
*/
export function stripHeadersFooters(pages: PageContent[]): void {
if (pages.length < MIN_PAGES) return;
// Step 1: Build per-page zone text sets
const pageZoneTexts: Set<string>[] = [];
for (const page of pages) {
const zoneTexts = new Set<string>();
for (const tb of page.textBoxes) {
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
if (midY >= TOP_ZONE_MIN_Y || midY <= BOTTOM_ZONE_MAX_Y) {
const key = tb.text.trim().replace(/\s+/g, " ");
if (key.length > 0) zoneTexts.add(key);
}
}
pageZoneTexts.push(zoneTexts);
}
// Step 2: Count global frequency AND longest consecutive run for each text
const globalCount = new Map<string, number>();
const maxConsecutive = new Map<string, number>();
// Collect all unique zone texts
const allTexts = new Set<string>();
for (const zts of pageZoneTexts) {
for (const t of zts) allTexts.add(t);
}
for (const text of allTexts) {
let total = 0;
let consecutive = 0;
let maxRun = 0;
for (const zts of pageZoneTexts) {
if (zts.has(text)) {
total++;
consecutive++;
if (consecutive > maxRun) maxRun = consecutive;
} else {
consecutive = 0;
}
}
globalCount.set(text, total);
maxConsecutive.set(text, maxRun);
}
// Step 3: Identify running headers/footers
const globalThreshold = Math.max(3, Math.floor(pages.length * 0.2));
const repeatedTexts = new Set<string>();
for (const text of allTexts) {
const gc = globalCount.get(text) ?? 0;
const mc = maxConsecutive.get(text) ?? 0;
// Global: appears on 20%+ of pages
if (gc >= globalThreshold) {
repeatedTexts.add(text);
continue;
}
// Consecutive: appears on 8+ consecutive pages (chapter-level headers)
if (mc >= MIN_CONSECUTIVE_PAGES) {
repeatedTexts.add(text);
}
}
if (repeatedTexts.size === 0) return;
// Step 4: Remove matching text boxes from each page
for (const page of pages) {
page.textBoxes = page.textBoxes.filter(tb => {
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
if (midY < TOP_ZONE_MIN_Y && midY > BOTTOM_ZONE_MAX_Y) return true;
const normalized = tb.text.trim().replace(/\s+/g, " ");
return !repeatedTexts.has(normalized);
});
}
}
@@ -1,48 +1,10 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* PDF to Markdown converter.
*
* Uses mupdf (native WASM) for fast PDF parsing and a custom pipeline for
* table detection via vector line extraction + raycasting.
*
* Pipeline:
* 1. Extract text boxes + vector segments + image regions per page (mupdf)
* 2. Detect column layout (single vs multi-column)
* 3. Per column: detect table grids from segments (grid detection + raycasting)
* 4. Render diagrams as PNG files (if output directory provided)
* 5. Render tables as markdown tables, free text as paragraphs/headings
*/
import * as path from "node:path";
import { pdfToMarkdown } from "@oh-my-pi/pi-natives";
import type { ConversionResult, Converter, StreamInfo } from "../../types";
import { detectColumns } from "./columns";
import { extractPages, renderImageRegion } from "./extract";
import { resolveTableGrids } from "./grid";
import { stripHeadersFooters } from "./headers";
import { renderPageContent } from "./render";
import type { Segment, TextBox } from "./types";
const EXTENSIONS = [".pdf"];
const MIMETYPES = ["application/pdf", "application/x-pdf"];
type ImageBlock = { topY: number; markdown: string };
/**
* Process a set of text boxes (one column or full page): run table detection,
* separate free text, and render to markdown.
*/
function processColumn(
pageNumber: number,
textBoxes: TextBox[],
segments: Segment[],
imageBlocks: ImageBlock[],
): string {
const { grids, consumedIds } = resolveTableGrids(pageNumber, textBoxes, segments);
const consumedSet = new Set(consumedIds);
const freeTextBoxes = textBoxes.filter(tb => !consumedSet.has(tb.id));
return renderPageContent(freeTextBoxes, grids, imageBlocks, textBoxes);
}
/** Converts PDF buffers to Markdown through the native `pdf-inspector` bridge. */
export class PdfConverter implements Converter {
name = "pdf";
@@ -56,91 +18,17 @@ export class PdfConverter implements Converter {
return false;
}
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
const pdfBytes = new Uint8Array(input);
const pages = await extractPages(pdfBytes);
// Remove running headers/footers before processing.
stripHeadersFooters(pages);
const imageDir = streamInfo.imageDir;
async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
const result = await pdfToMarkdown(input);
const notice =
result.pagesNeedingOcr.length > 0
? `Text extraction is incomplete for PDF pages ${result.pagesNeedingOcr.join(", ")}. Use the browser tool to render those pages or OCR them.`
: undefined;
const pageMarkdowns: string[] = [];
for (const page of pages) {
// Build image blocks for this page.
const imageBlocks: ImageBlock[] = [];
if (imageDir && page.images.length > 0) {
for (const img of page.images) {
const filename = `${img.id}.png`;
const filepath = path.join(imageDir, filename);
try {
const png = await renderImageRegion(pdfBytes, img);
await Bun.write(filepath, png);
imageBlocks.push({ topY: img.topY, markdown: `![${img.id}](${filepath})` });
} catch {
// Image rendering failed — skip.
}
}
} else if (page.images.length > 0) {
for (const img of page.images) {
imageBlocks.push({
topY: img.topY,
markdown: `<!-- image: ${img.id} (page ${img.pageNumber}, ${img.bbox.w}x${img.bbox.h}pt) -->`,
});
}
}
// Detect column layout.
// If the page has vertical segments (tables), suppress column detection
// when one detected column is very narrow — that's a table's first column,
// not a page layout column.
const layout = detectColumns(page.textBoxes);
if (layout.columnCount > 1 && page.segments.some(s => Math.abs(s.x1 - s.x2) <= 0.8)) {
const pageXMin = Math.min(...page.textBoxes.map(tb => tb.bounds.left));
const pageXMax = Math.max(...page.textBoxes.map(tb => tb.bounds.right));
const pageWidth = pageXMax - pageXMin;
const minColFraction = 0.3;
const tooNarrow = layout.columns.some(col => {
const colXMin = Math.min(...col.map(tb => tb.bounds.left));
const colXMax = Math.max(...col.map(tb => tb.bounds.right));
return (colXMax - colXMin) / pageWidth < minColFraction;
});
if (tooNarrow) {
layout.columnCount = 1;
layout.columns = [page.textBoxes];
layout.boundaries = [];
}
}
if (layout.columnCount === 1) {
// Single column — process normally.
const md = processColumn(page.pageNumber, page.textBoxes, page.segments, imageBlocks);
if (md.length > 0) pageMarkdowns.push(md);
} else {
// Multi-column — process each column independently, then join.
const columnMarkdowns: string[] = [];
for (const colBoxes of layout.columns) {
// Filter segments to those within this column's X range.
const colXMin = Math.min(...colBoxes.map(tb => tb.bounds.left));
const colXMax = Math.max(...colBoxes.map(tb => tb.bounds.right));
const margin = 10;
const colSegments = page.segments.filter(seg => {
const segXMin = Math.min(seg.x1, seg.x2);
const segXMax = Math.max(seg.x1, seg.x2);
return segXMax >= colXMin - margin && segXMin <= colXMax + margin;
});
// Images go with the first column only (no X info to split by).
const md = processColumn(
page.pageNumber,
colBoxes,
colSegments,
columnMarkdowns.length === 0 ? imageBlocks : [],
);
if (md.length > 0) columnMarkdowns.push(md);
}
const joined = columnMarkdowns.join("\n\n");
if (joined.length > 0) pageMarkdowns.push(joined);
}
}
return { markdown: pageMarkdowns.join("\n\n") };
const conversion: ConversionResult = {
markdown: notice ? [result.markdown, notice].filter(Boolean).join("\n\n") : result.markdown,
};
if (result.title !== undefined) conversion.title = result.title;
return conversion;
}
}
@@ -1,501 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Markdown rendering for PDF pages.
*
* Converts table grids and free text boxes into markdown, handling:
* - Table grid → markdown table (`| col | col |`)
* - Free text → paragraphs with heading detection (by font size)
* - Content ordering (top-to-bottom via Y coordinate)
* - Paragraph wrap merging (lines broken across PDF line boundaries)
* - Page number removal
*
* Ported from @oharato/pdf2md-ts, stripped of CJK/TDnet-specific logic.
*/
import type { ContentBlock, TableGrid, TextBox } from "./types";
/** A free-text line grouped from horizontally adjacent text boxes. */
interface RenderLine {
text: string;
topY: number;
fontSize: number;
isBold: boolean;
isTabular: boolean;
}
/** A content block carrying the Y of its last wrapped line during merging. */
type WrapBlock = ContentBlock & { lastTopY: number };
// ---------------------------------------------------------------------------
// Utility
// ---------------------------------------------------------------------------
/** Convert full-width ASCII characters (A→A, !→! etc.) to normal ASCII. */
function normalizeFullWidthAscii(text: string): string {
return text.replace(/[!-~]/g, ch => String.fromCharCode(ch.charCodeAt(0) - 0xfee0));
}
function escapePipes(text: string): string {
return normalizeFullWidthAscii(text).replaceAll("|", "\\|").replaceAll("\n", "<br>");
}
/** Parse a markdown pipe-delimited row into cell strings. */
function parsePipeRow(line: string): string[] {
const trimmed = line.trim();
if (!trimmed.startsWith("|") || !trimmed.endsWith("|")) return [];
return trimmed
.slice(1, -1)
.split("|")
.map(cell => cell.trim());
}
// ---------------------------------------------------------------------------
// Table rendering
// ---------------------------------------------------------------------------
/**
* Render a TableGrid as a markdown table.
*/
export function renderTableToMarkdown(table: TableGrid): string {
if (table.rows === 0 || table.cols === 0) return "";
const matrix = Array.from({ length: table.rows }, () => Array.from({ length: table.cols }, () => ""));
for (const cell of table.cells) {
if (cell.row < table.rows && cell.col < table.cols) {
matrix[cell.row][cell.col] = escapePipes(cell.text.trim());
}
}
const normalized = normalizeShiftedSparseColumns(matrix);
const promoted = promoteSubHeaderPrefixes(normalized);
const header = `| ${promoted[0].join(" | ")} |`;
const divider = `| ${Array.from({ length: promoted[0].length }, () => "---").join(" | ")} |`;
const body = promoted
.slice(1)
.map(row => `| ${row.join(" | ")} |`)
.join("\n");
return [header, divider, body].filter(l => l.length > 0).join("\n");
}
/**
* Fix tables with ≥5 columns where sparse single-value columns are
* misaligned. Shifts those values to the adjacent dense column and
* removes the now-empty sparse columns.
*/
function normalizeShiftedSparseColumns(matrix: string[][]): string[][] {
if (matrix.length === 0 || matrix[0].length < 5) return matrix;
const _rows = matrix.length;
const cols = matrix[0].length;
const counts = Array.from({ length: cols }, (_, c) =>
matrix.reduce((n, row) => n + (row[c].trim().length > 0 ? 1 : 0), 0),
);
const denseCols = new Set(
counts
.map((count, col) => ({ count, col }))
.filter(({ col, count }) => col === 0 || count >= 2)
.map(({ col }) => col),
);
const sparseCols = counts
.map((count, col) => ({ count, col }))
.filter(({ col, count }) => col > 0 && col < cols - 1 && count === 1)
.map(({ col }) => col);
if (sparseCols.length < 2 || denseCols.size < 4) return matrix;
const moves: Array<{ from: number; to: number; row: number }> = [];
for (const from of sparseCols) {
const row = matrix.findIndex(r => r[from].trim().length > 0);
const to = from + 1;
if (row < 0) return matrix;
if (!denseCols.has(to)) return matrix;
if (matrix[row][to].trim().length > 0) return matrix;
moves.push({ from, to, row });
}
const copy = matrix.map(row => [...row]);
for (const { from, to, row } of moves) {
copy[row][to] = copy[row][to].trim().length > 0 ? `${copy[row][to]} ${copy[row][from]}` : copy[row][from];
copy[row][from] = "";
}
const keepCols = Array.from({ length: cols }, (_, c) => c).filter(c => copy.some(row => row[c].trim().length > 0));
if (keepCols.length === cols) return copy;
return copy.map(row => keepCols.map(c => row[c]));
}
/**
* When a data row has ≥2 parenthesized qualifiers in non-first columns
* (and the first column is empty), promote them into the header row.
*/
function promoteSubHeaderPrefixes(matrix: string[][]): string[][] {
if (matrix.length < 2) return matrix;
const PAREN_RE = /^\([^)]{1,40}\)$/;
const result = matrix.map(row => [...row]);
const cols = matrix[0].length;
const rowsToRemove = new Set<number>();
for (let r = 1; r < result.length; r++) {
if (rowsToRemove.has(r)) continue;
const promotable: Array<{ col: number; prefix: string; isFullCell: boolean }> = [];
for (let col = 1; col < cols; col++) {
const cell = (result[r][col] ?? "").trim();
if (!cell) continue;
const parts = cell.split("<br>");
if (parts.length === 1 && PAREN_RE.test(cell)) {
promotable.push({ col, prefix: cell, isFullCell: true });
} else if (parts.length >= 2 && PAREN_RE.test(parts[0].trim())) {
promotable.push({
col,
prefix: parts[0].trim(),
isFullCell: false,
});
}
}
if (promotable.length < 2) continue;
if (promotable.some(p => p.isFullCell) && result[r][0].trim().length > 0) continue;
for (const { col, prefix, isFullCell } of promotable) {
result[0][col] = result[0][col].trim() ? `${result[0][col]} ${prefix}` : prefix;
if (isFullCell) {
result[r][col] = "";
} else {
const parts = result[r][col].split("<br>");
result[r][col] = parts.slice(1).join("<br>");
}
}
if (result[r].every(cell => cell.trim().length === 0)) {
rowsToRemove.add(r);
}
}
return result.filter((_, r) => !rowsToRemove.has(r));
}
// ---------------------------------------------------------------------------
// Free text rendering
// ---------------------------------------------------------------------------
/** Y tolerance for grouping text boxes onto the same visual line. */
const TEXT_LINE_Y_TOLERANCE = 3;
/** Minimum X gap between adjacent boxes to mark line as tabular. */
const TABULAR_X_GAP = 30;
/**
* Minimum font size (pts) to consider when computing the modal body font.
* Tiny labels from diagrams, footnote markers, and superscripts are excluded
* so they don't skew the modal toward small sizes.
*/
const MIN_BODY_FONT_SIZE = 7;
/**
* Compute the most frequent font size among text boxes, ignoring very small
* text that likely comes from diagrams, footnotes, or superscripts.
*/
function modalFontSize(textBoxes: TextBox[]): number {
const counts = new Map<number, number>();
for (const tb of textBoxes) {
const size = Math.round((tb.fontSize ?? 0) * 10) / 10;
if (size < MIN_BODY_FONT_SIZE) continue;
counts.set(size, (counts.get(size) ?? 0) + 1);
}
let modal = 0;
let maxCount = 0;
for (const [size, count] of counts) {
if (count > maxCount) {
maxCount = count;
modal = size;
}
}
return modal;
}
/** Group free text boxes into horizontal lines, sorted top-to-bottom. */
function groupFreeTextIntoLines(textBoxes: TextBox[]): RenderLine[] {
if (textBoxes.length === 0) return [];
const sorted = [...textBoxes].sort((a, b) => {
const ya = (a.bounds.top + a.bounds.bottom) / 2;
const yb = (b.bounds.top + b.bounds.bottom) / 2;
const dy = yb - ya;
if (Math.abs(dy) > TEXT_LINE_Y_TOLERANCE) return dy;
return a.bounds.left - b.bounds.left;
});
const lines: RenderLine[] = [];
let curParts = [sorted[0].text];
let curBoxes = [sorted[0]];
let curY = (sorted[0].bounds.top + sorted[0].bounds.bottom) / 2;
let curTopY = curY;
let curFontSize = sorted[0].fontSize;
let curIsBold = sorted[0].isBold;
const finishLine = () => {
let isTabular = false;
for (let j = 1; j < curBoxes.length; j++) {
if (curBoxes[j].bounds.left - curBoxes[j - 1].bounds.right > TABULAR_X_GAP) {
isTabular = true;
break;
}
}
lines.push({
text: curParts.join(" "),
topY: curTopY,
fontSize: curFontSize,
isBold: curIsBold,
isTabular,
});
};
for (let i = 1; i < sorted.length; i++) {
const box = sorted[i];
const cy = (box.bounds.top + box.bounds.bottom) / 2;
if (Math.abs(cy - curY) <= TEXT_LINE_Y_TOLERANCE) {
curParts.push(box.text);
curBoxes.push(box);
curFontSize = Math.max(curFontSize, box.fontSize);
curIsBold = curIsBold || box.isBold;
} else {
finishLine();
curParts = [box.text];
curBoxes = [box];
curY = cy;
curTopY = cy;
curFontSize = box.fontSize;
curIsBold = box.isBold;
}
}
finishLine();
return lines;
}
/** Determine markdown heading prefix based on font size relative to body. */
function headingPrefix(fontSize: number, bodyFontSize: number, isBold: boolean): string {
if (bodyFontSize <= 0) return "";
const ratio = fontSize / bodyFontSize;
// Large headings (>2x body size)
if (ratio >= 2.0) return "# ";
// Medium headings (~1.5x body size)
if (ratio >= 1.4) return "## ";
// Small headings (bold and slightly larger)
if (ratio >= 1.1 && isBold) return "### ";
return "";
}
// ---------------------------------------------------------------------------
// Block merging
// ---------------------------------------------------------------------------
/** Merge consecutive blocks with the same heading prefix (wrapped headings). */
function mergeConsecutiveHeadings(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
if (blocks.length === 0) return [];
const HEADING_RE = /^(#{1,6} )/;
const maxGap = Math.max(bodyFS * 3, 30);
const merged: ContentBlock[] = [];
let cur: ContentBlock = { ...blocks[0] };
for (let i = 1; i < blocks.length; i++) {
const next = blocks[i];
const curMatch = cur.content.match(HEADING_RE);
const nextMatch = next.content.match(HEADING_RE);
const gap = cur.topY - next.topY;
if (curMatch && nextMatch && curMatch[1] === nextMatch[1] && gap <= maxGap) {
cur = {
topY: cur.topY,
content: `${cur.content} ${next.content.slice(nextMatch[1].length)}`,
isTabular: cur.isTabular || next.isTabular,
};
} else {
merged.push(cur);
cur = { ...next };
}
}
merged.push(cur);
return merged;
}
/**
* Merge consecutive plain-text blocks that are wrapped lines of the same paragraph.
*/
function mergeParagraphWraps(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
if (blocks.length === 0 || bodyFS <= 0) return blocks;
const HEADING_RE = /^#{1,6} /;
const SENTENCE_END_RE = /[.!?…)\]]\s*$/;
const maxGap = bodyFS * 2.0;
const MIN_WRAP_LENGTH = 25;
const merged: ContentBlock[] = [];
let cur: WrapBlock = { ...blocks[0], lastTopY: blocks[0].topY };
for (let i = 1; i < blocks.length; i++) {
const next = blocks[i];
const curIsBody = !HEADING_RE.test(cur.content) && !cur.content.startsWith("|");
const nextIsBody = !HEADING_RE.test(next.content) && !next.content.startsWith("|");
const gap = cur.lastTopY - next.topY;
const isWrap =
curIsBody &&
nextIsBody &&
!cur.isTabular &&
!next.isTabular &&
gap > 0 &&
gap <= maxGap &&
cur.content.length > MIN_WRAP_LENGTH &&
!SENTENCE_END_RE.test(cur.content);
if (isWrap) {
cur = {
topY: cur.topY,
lastTopY: next.topY,
content: `${cur.content.trimEnd()} ${next.content.trimStart()}`,
isTabular: false,
};
} else {
merged.push({ topY: cur.topY, content: cur.content });
cur = { ...next, lastTopY: next.topY };
}
}
merged.push({ topY: cur.topY, content: cur.content });
return merged;
}
/** Remove page number blocks near the bottom of the page. */
function removePageNumbers(blocks: ContentBlock[]): ContentBlock[] {
const PAGE_NUM_RE = /^(?:#{1,6}\s*)?\d+\s*$/;
const BOTTOM_Y = 120;
return blocks.filter((block, idx) => {
const isBottom = idx >= blocks.length - 3;
const isLowY = block.topY <= BOTTOM_Y;
const isPageNum = PAGE_NUM_RE.test(block.content.trim());
return !(isBottom && isLowY && isPageNum);
});
}
// ---------------------------------------------------------------------------
// Detached first-column table reconstruction
// ---------------------------------------------------------------------------
/**
* Fix tables where the first column was emitted as free text blocks
* around a markdown table containing only the right-side columns.
*
* Detects: a plain-text header line with (N+1) tokens above an N-column
* markdown table, plus short label lines whose count matches the table's
* logical row count. Reconstructs into a proper (N+1)-column table.
*/
function normalizeDetachedFirstColumnTables(blocks: ContentBlock[]): ContentBlock[] {
const HEADING_RE = /^#{1,6}\s/;
const isTableBlock = (text: string) => text.trimStart().startsWith("|");
const isPlainBlock = (text: string) => !HEADING_RE.test(text) && !isTableBlock(text);
const isShortLabel = (text: string) => {
const t = text.trim();
return t.length > 0 && t.length <= 40;
};
const splitTokens = (text: string) =>
text
.trim()
.split(/[ \t]+/)
.filter(Boolean);
const replacements = new Map<number, string>();
const remove = new Set<number>();
for (let tableIdx = 0; tableIdx < blocks.length; tableIdx++) {
if (remove.has(tableIdx)) continue;
const tableBlock = blocks[tableIdx];
if (!isTableBlock(tableBlock.content)) continue;
const tableLines = tableBlock.content
.split("\n")
.map(line => line.trim())
.filter(line => line.startsWith("|"));
const dataRows = tableLines
.filter(line => !/^\|\s*[-: ]+\|/.test(line))
.map(parsePipeRow)
.filter(row => row.length > 0);
if (dataRows.length === 0) continue;
const cols = dataRows[0].length;
if (cols < 2 || dataRows.some(row => row.length !== cols)) continue;
// Expand by <br> count to get logical row count
const logicalRows: string[][] = [];
for (const row of dataRows) {
const splitCells = row.map(cell => cell.split("<br>").map(p => p.trim()));
const rowSpan = Math.max(...splitCells.map(parts => parts.length));
for (let k = 0; k < rowSpan; k++) {
logicalRows.push(splitCells.map(parts => parts[k] ?? ""));
}
}
if (logicalRows.length < 2) continue;
// Find header with (cols + 1) non-numeric tokens
let headerIdx = -1;
let headerTokens: string[] = [];
for (let i = Math.max(0, tableIdx - 4); i <= tableIdx - 1; i++) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text)) continue;
const tokens = splitTokens(text);
if (tokens.length === cols + 1 && tokens.every(tok => !/[0-9]/.test(tok))) {
headerIdx = i;
headerTokens = tokens;
}
}
if (headerIdx < 0) continue;
// Collect short label lines above/below table
const aboveLabels: Array<{ idx: number; text: string }> = [];
for (let i = tableIdx - 1; i > headerIdx; i--) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text) || !isShortLabel(text)) break;
aboveLabels.push({ idx: i, text });
}
aboveLabels.reverse();
const belowLabels: Array<{ idx: number; text: string }> = [];
for (let i = tableIdx + 1; i < blocks.length; i++) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text) || !isShortLabel(text)) break;
belowLabels.push({ idx: i, text });
}
const labels = [...aboveLabels, ...belowLabels];
if (labels.length !== logicalRows.length) continue;
// Reconstruct the full table
const normalizedLines: string[] = [];
normalizedLines.push(`| ${headerTokens.join(" | ")} |`);
normalizedLines.push(`| ${Array.from({ length: cols + 1 }, () => "---").join(" | ")} |`);
for (let r = 0; r < logicalRows.length; r++) {
normalizedLines.push(`| ${labels[r].text} | ${logicalRows[r].join(" | ")} |`);
}
replacements.set(tableIdx, normalizedLines.join("\n"));
remove.add(headerIdx);
for (const label of labels) remove.add(label.idx);
}
if (replacements.size === 0 && remove.size === 0) return blocks;
const out: ContentBlock[] = [];
for (let i = 0; i < blocks.length; i++) {
if (remove.has(i)) continue;
const replaced = replacements.get(i);
if (replaced) {
out.push({ topY: blocks[i].topY, content: replaced });
} else {
out.push(blocks[i]);
}
}
return out;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/**
* Render one page's content: free text and tables interleaved top-to-bottom.
*/
export function renderPageContent(
freeTextBoxes: TextBox[],
tables: TableGrid[],
imageBlocks: Array<{ topY: number; markdown: string }> = [],
allTextBoxes?: TextBox[],
): string {
const blocks: ContentBlock[] = [];
// Use ALL text boxes (before table/diagram filtering) for modal font size,
// so that diagram labels released as free text don't skew the body size.
const bodyFS = modalFontSize(allTextBoxes ?? freeTextBoxes);
// Free text lines
for (const line of groupFreeTextIntoLines(freeTextBoxes)) {
const prefix = headingPrefix(line.fontSize, bodyFS, line.isBold);
blocks.push({
topY: line.topY,
content: prefix + line.text,
isTabular: prefix === "" && line.isTabular,
});
}
// Tables
for (const table of tables) {
const md = renderTableToMarkdown(table);
if (md.length > 0) {
blocks.push({ topY: table.topY, content: md });
}
}
// Images
for (const img of imageBlocks) {
blocks.push({ topY: img.topY, content: img.markdown });
}
// Sort top-to-bottom (higher Y = higher on page = comes first)
blocks.sort((a, b) => b.topY - a.topY);
const cleaned = removePageNumbers(blocks);
const headingsMerged = mergeConsecutiveHeadings(cleaned, bodyFS);
const merged = mergeParagraphWraps(headingsMerged, bodyFS);
const normalized = normalizeDetachedFirstColumnTables(merged);
return normalized
.map(b => b.content)
.join("\n\n")
.trim();
}
@@ -1,84 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/** Bounding box in PDF coordinate space (origin = bottom-left). */
export type Bounds = {
left: number;
right: number;
/** Higher value = higher on the page. */
top: number;
bottom: number;
};
/** A text fragment with position and font metadata. */
export type TextBox = {
id: string;
text: string;
bounds: Bounds;
pageNumber: number;
/** Dominant font size in points. */
fontSize: number;
/** True if rendered bold (font name or rendering mode). */
isBold: boolean;
};
/** A horizontal or vertical line segment extracted from vector graphics. */
export type Segment = {
id: string;
x1: number;
y1: number;
x2: number;
y2: number;
};
/** A single cell in a resolved table grid. */
export type TableCell = {
row: number;
col: number;
text: string;
rowSpan: number;
colSpan: number;
};
/** A resolved table grid ready for markdown rendering. */
export type TableGrid = {
pageNumber: number;
rows: number;
cols: number;
cells: TableCell[];
warnings: string[];
/** Top Y coordinate (PDF space: larger = higher on page). */
topY: number;
/** True for tables detected without vector borders. */
isBorderless: boolean;
};
/** An image/diagram region detected on a page. */
export type ImageRegion = {
id: string;
pageNumber: number;
/** Bounding box in mupdf coordinates (top-left origin). */
bbox: {
x: number;
y: number;
w: number;
h: number;
};
/** Y position in PDF coordinates (bottom-left) for ordering. */
topY: number;
};
/** Result of extracting content from a single PDF page. */
export type PageContent = {
pageNumber: number;
textBoxes: TextBox[];
segments: Segment[];
images: ImageRegion[];
};
/** A block of rendered content (text paragraph or table). */
export type ContentBlock = {
topY: number;
content: string;
/** True if this line has wide gaps between text boxes (column headers). */
isTabular?: boolean;
};
+8 -9
View File
@@ -92,7 +92,7 @@ async function initializeConnection(
transport: MCPTransport,
options?: {
signal?: AbortSignal;
/** Called after the initialize response (which sets the session ID) but before notifications/initialized. */
/** Called after notifications/initialized succeeds. */
onInitialized?: () => void | Promise<void>;
},
): Promise<MCPInitializeResult> {
@@ -119,14 +119,13 @@ async function initializeConnection(
// initialize; transports that don't need it ignore this.
transport.setProtocolVersion?.(result.protocolVersion);
// Hook point: the transport now has the session ID from the initialize response.
// For HTTP, this is the moment to open the SSE stream so server-to-client requests
// triggered by notifications/initialized (e.g. roots/list) can be delivered.
await options?.onInitialized?.();
// Send initialized notification
// Send initialized before opening the optional GET SSE stream. Servers may
// reject or terminate sessions that receive session traffic before this
// notification; POST response streams already carry messages during setup.
await transport.notify("notifications/initialized");
await options?.onInitialized?.();
return result;
}
@@ -162,8 +161,8 @@ export async function connectToServer(
const initResult = await initializeConnection(transport, {
signal: options?.signal,
async onInitialized() {
// Open the SSE stream before sending initialized, so server-to-client
// requests triggered by on_initialized (e.g. roots/list) are delivered.
// Open the optional GET SSE stream only after the initialized
// notification makes the session ready for further traffic.
if ("startSSEListener" in transport! && typeof transport!.startSSEListener === "function") {
await (transport as { startSSEListener(): Promise<void> }).startSSEListener();
}
@@ -353,6 +353,13 @@ export class RpcClient {
// failures are reaped by the readyPromise catch below; established
// workers are reaped here so pending requests cannot hang indefinitely.
if (!readySettled) {
// Stdout can close before the exit reaper finishes draining stderr.
// child.exited settles only after the stderr tail is complete (for
// nonzero exits), so give it a bounded head start: the exit watcher
// below was registered first and rejects with the real stderr text
// instead of an empty "Stderr:" (flaked under full-suite load).
await Promise.race([child.exited.catch(() => {}), Bun.sleep(250)]);
if (readySettled) return;
readySettled = true;
readyReject(new Error(`Agent output stream ended before ready. Stderr: ${child.peekStderr()}`));
return;
@@ -39,7 +39,7 @@ export function buildHotkeysMarkdown(bindings: HotkeysMarkdownBindings): string
"| `Tab` | Path completion / accept autocomplete |",
`| \`${appKey(bindings, "app.interrupt")}\` | Cancel autocomplete / interrupt active work |`,
`| \`${appKey(bindings, "app.clear")}\` | Clear editor (first) / exit (second) |`,
`| \`${appKey(bindings, "app.exit")}\` | Exit (when editor is empty) |`,
`| \`${appKey(bindings, "app.exit")}\` | Exit (saves current prompt as draft) |`,
`| \`${appKey(bindings, "app.suspend")}\` | Suspend to background |`,
`| \`${appKey(bindings, "app.display.reset")}\` | Reset terminal display |`,
`| \`${appKey(bindings, "app.thinking.cycle")}\` | Cycle thinking level |`,
@@ -1,250 +0,0 @@
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { isEexist, isEnotempty, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
import type { ToolSession } from "../sdk";
import { loadImageInput, MAX_IMAGE_INPUT_BYTES, webpExclusionForModel } from "../utils/image-loading";
import { convertFileWithMarkit } from "../utils/markit";
import type { ReadToolDetails } from "./read";
import { prependSuffixResolutionNotice } from "./read-format";
import { isNotFoundError } from "./read-path-resolution";
import { formatBytes } from "./render-utils";
import { ToolError } from "./tool-errors";
import { toolResult } from "./tool-result";
const MAX_IMAGE_SIZE = MAX_IMAGE_INPUT_BYTES;
const PDF_IMAGE_PLACEHOLDER_RE = /<!--\s*image:\s*([^\s<>]+)(.*?)-->/g;
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
const PDF_IMAGE_MEMBER_EXTENSION_RE = /\.png$/i;
const PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH = 96;
interface PdfImageSnapshot {
directory: string;
filePath: string;
digest: string;
}
interface PdfImageExtraction {
controller: AbortController;
promise: Promise<string>;
settled: boolean;
waiters: number;
}
const pdfImageExtractions = new Map<string, PdfImageExtraction>();
function pdfImageMemberPath(pdfPath: string, imageId: string): string {
const member = PDF_IMAGE_MEMBER_EXTENSION_RE.test(imageId) ? imageId : `${imageId}.png`;
return `${pdfPath}:${member}`;
}
export function rewritePdfImagePlaceholders(markdown: string, pdfPath: string): string {
return markdown.replace(PDF_IMAGE_PLACEHOLDER_RE, (_match: string, imageId: string, metadataText: string) => {
const metadata = metadataText.trim();
const suffix = metadata.length > 0 ? ` (${metadata})` : "";
return `Image ${imageId}${suffix}: read \`${pdfImageMemberPath(pdfPath, imageId)}\``;
});
}
export function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; member: string } | null {
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
if (!match) return null;
const pdfPath = match[1];
const member = match[2];
if (pdfPath === undefined || member === undefined) return null;
if (member.length !== 0 && !PDF_IMAGE_MEMBER_EXTENSION_RE.test(member)) return null;
return { pdfPath, member };
}
function pdfImageCacheDir(session: ToolSession, absolutePdfPath: string, contentDigest: string): string {
const artifactsDir = session.getArtifactsDir?.();
let root = artifactsDir ?? undefined;
if (root === undefined) {
const sessionFile = session.getSessionFile();
root = sessionFile?.endsWith(".jsonl") ? sessionFile.slice(0, -6) : path.join(os.tmpdir(), "omp-read-pdf-images");
}
const basename = path
.basename(absolutePdfPath)
.replace(/[^A-Za-z0-9._-]/g, "_")
.slice(0, PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH);
const pathDigest = Bun.hash(absolutePdfPath).toString(36);
return path.join(root, "read-pdf-images", `${basename}-${pathDigest}-${contentDigest}`);
}
async function snapshotPdfSource(absolutePdfPath: string, signal?: AbortSignal): Promise<PdfImageSnapshot> {
const directory = await fs.mkdtemp(path.join(os.tmpdir(), "omp-read-pdf-"));
try {
const bytes = await untilAborted(signal, () => Bun.file(absolutePdfPath).bytes());
signal?.throwIfAborted();
const digest = new Bun.CryptoHasher("sha256").update(bytes).digest("hex");
const filePath = path.join(directory, "source.pdf");
await Bun.write(filePath, bytes);
signal?.throwIfAborted();
return { directory, filePath, digest };
} catch (error) {
await fs.rm(directory, { recursive: true, force: true });
throw error;
}
}
async function listPdfImageMembers(imageDir: string): Promise<string[]> {
try {
const entries = await fs.readdir(imageDir, { withFileTypes: true });
const members: string[] = [];
for (const entry of entries) {
if (entry.isFile() && PDF_IMAGE_MEMBER_EXTENSION_RE.test(entry.name)) members.push(entry.name);
}
return members.sort();
} catch (error) {
if (isNotFoundError(error)) return [];
throw error;
}
}
async function extractPdfImages(snapshot: PdfImageSnapshot, imageDir: string, signal: AbortSignal): Promise<string> {
const markerPath = path.join(imageDir, ".extracted");
try {
await fs.stat(markerPath);
return imageDir;
} catch (error) {
if (!isNotFoundError(error)) throw error;
}
await fs.mkdir(path.dirname(imageDir), { recursive: true });
const stagingDir = await fs.mkdtemp(`${imageDir}.tmp-`);
let published = false;
try {
const result = await convertFileWithMarkit(snapshot.filePath, signal, { imageDir: stagingDir });
if (!result.ok) {
throw new ToolError(`Cannot extract images from PDF: ${result.error ?? "conversion failed"}`);
}
await Bun.write(path.join(stagingDir, ".extracted"), "ok");
try {
await fs.rename(stagingDir, imageDir);
published = true;
} catch (error) {
if (!isEexist(error) && !isEnotempty(error)) throw error;
try {
await fs.stat(markerPath);
} catch (markerError) {
if (isNotFoundError(markerError)) throw error;
throw markerError;
}
}
return imageDir;
} finally {
if (!published) await fs.rm(stagingDir, { recursive: true, force: true });
}
}
function createPdfImageExtraction(snapshot: PdfImageSnapshot, imageDir: string): PdfImageExtraction {
const controller = new AbortController();
const promise = extractPdfImages(snapshot, imageDir, controller.signal).finally(() =>
fs.rm(snapshot.directory, { recursive: true, force: true }),
);
const extraction: PdfImageExtraction = { controller, promise, settled: false, waiters: 0 };
const settle = () => {
extraction.settled = true;
if (pdfImageExtractions.get(imageDir) === extraction) pdfImageExtractions.delete(imageDir);
};
void promise.then(settle, settle);
return extraction;
}
async function waitForPdfImageExtraction(
extraction: PdfImageExtraction,
signal: AbortSignal | undefined,
): Promise<string> {
extraction.waiters++;
try {
return await untilAborted(signal, extraction.promise);
} finally {
extraction.waiters--;
if (extraction.waiters === 0 && !extraction.settled) {
extraction.controller.abort();
try {
await extraction.promise;
} catch {}
}
}
}
async function ensurePdfImageCache(
session: ToolSession,
absolutePdfPath: string,
signal?: AbortSignal,
): Promise<string> {
const snapshot = await snapshotPdfSource(absolutePdfPath, signal);
const imageDir = pdfImageCacheDir(session, absolutePdfPath, snapshot.digest);
const existing = pdfImageExtractions.get(imageDir);
if (existing && !existing.settled && !existing.controller.signal.aborted) {
await fs.rm(snapshot.directory, { recursive: true, force: true });
return waitForPdfImageExtraction(existing, signal);
}
const extraction = createPdfImageExtraction(snapshot, imageDir);
pdfImageExtractions.set(imageDir, extraction);
return waitForPdfImageExtraction(extraction, signal);
}
export async function readPdfImageMember(
session: ToolSession,
autoResizeImages: boolean,
absolutePdfPath: string,
pdfDisplayPath: string,
member: string,
suffixResolution: { from: string; to: string } | undefined,
signal?: AbortSignal,
): Promise<AgentToolResult<ReadToolDetails>> {
const imageDir = await ensurePdfImageCache(session, absolutePdfPath, signal);
const members = await listPdfImageMembers(imageDir);
if (member.length === 0) {
const text =
members.length === 0
? "No extractable PDF image members found."
: `Extractable PDF image members:\n${members
.map(imageMember => `- read \`${pdfDisplayPath}:${imageMember}\``)
.join("\n")}`;
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
.text(prependSuffixResolutionNotice(text, suffixResolution))
.sourcePath(absolutePdfPath)
.done();
}
if (!members.includes(member)) {
const available = members.length === 0 ? "(none)" : members.join(", ");
throw new ToolError(`PDF image member '${member}' not found. Available members: ${available}`);
}
const imagePath = path.join(imageDir, member);
const imageStat = await Bun.file(imagePath).stat();
if (imageStat.size > MAX_IMAGE_SIZE) {
const sizeStr = formatBytes(imageStat.size);
const maxStr = formatBytes(MAX_IMAGE_SIZE);
throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`);
}
const metadata = await readImageMetadata(imagePath);
const mimeType = metadata?.mimeType;
if (!mimeType) throw new ToolError(`PDF image member '${member}' is not a supported image.`);
const imageInput = await loadImageInput({
path: `${pdfDisplayPath}:${member}`,
cwd: session.cwd,
autoResize: autoResizeImages,
maxBytes: MAX_IMAGE_SIZE,
resolvedPath: imagePath,
detectedMimeType: mimeType,
excludeWebP: webpExclusionForModel(session.getActiveModel?.()),
});
if (!imageInput) {
throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`);
}
const textNote = prependSuffixResolutionNotice(imageInput.textNote, suffixResolution);
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
.content([
{ type: "text", text: textNote },
{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType },
])
.sourcePath(imageInput.resolvedPath)
.done();
}
+135
View File
@@ -0,0 +1,135 @@
import { pathToFileURL } from "node:url";
import { untilAborted } from "@oh-my-pi/pi-utils";
import type { ToolSession } from "../sdk";
import type { BrowserHandle } from "./browser/registry";
import type { ScreenshotResult } from "./browser/tab-protocol";
import { ToolAbortError, ToolError } from "./tool-errors";
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
const PDF_PAGE_MEMBER_RE = /^(?:p|page[-_]?)(\d+)(?:[-_].*)?\.png$/i;
const PDF_RENDER_TIMEOUT_MS = 30_000;
// Chromium's PDF plugin paints in an out-of-process frame after navigation has
// completed. Wait for document dimensions, then cross compositor boundaries
// before capturing; otherwise the screenshot can contain only the viewer shell.
const PDF_SCREENSHOT_CODE = `
let viewerFrame;
await wait(async () => {
for (const frame of page.frames()) {
try {
const loaded = await frame.evaluate(() => {
const viewer = document.querySelector("pdf-viewer");
const toolbar = viewer?.shadowRoot?.querySelector("viewer-toolbar");
const pageLength = toolbar
?.shadowRoot?.querySelector("viewer-page-selector")
?.shadowRoot?.querySelector("#pagelength")
?.textContent;
if (Number(pageLength) > 0 && !toolbar?.hasAttribute("loading_")) return true;
const plugin = document.querySelector('embed[type="application/x-google-chrome-pdf"]');
const sizer = document.querySelector("#sizer");
return plugin !== null && sizer !== null && sizer.clientWidth > 0 && sizer.clientHeight > 0;
});
if (loaded) {
viewerFrame = frame;
return true;
}
} catch {}
}
return false;
});
await page.screenshot({ type: "png" });
await viewerFrame.evaluate(() => {
const { promise, resolve } = Promise.withResolvers();
requestAnimationFrame(() =>
requestAnimationFrame(() =>
requestAnimationFrame(() => requestAnimationFrame(resolve)),
),
);
return promise;
});
return await tab.screenshot({ fullPage: true, silent: true });
`;
/** A legacy PDF image-member path interpreted as a page screenshot request. */
export interface PdfImageReadTarget {
/** PDF path before the member delimiter. */
pdfPath: string;
/** Original member text after the delimiter. */
member: string;
/** One-indexed page inferred from names such as `p2-img0.png`; defaults to page 1. */
page: number;
}
/** Parse a former PDF image-member path as a Chromium page screenshot request. */
export function splitPdfImageReadPath(readPath: string): PdfImageReadTarget | null {
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
const pdfPath = match?.[1];
const member = match?.[2];
if (!pdfPath || member === undefined) return null;
const pageText = PDF_PAGE_MEMBER_RE.exec(member)?.[1];
const parsedPage = pageText === undefined ? 1 : Number(pageText);
const page = Number.isSafeInteger(parsedPage) && parsedPage > 0 ? parsedPage : 1;
return { pdfPath, member, page };
}
/** Render one PDF page through the browser tool's shared headless Chromium. */
export async function renderPdfPageScreenshot(
session: ToolSession,
absolutePdfPath: string,
page: number,
signal?: AbortSignal,
): Promise<ScreenshotResult> {
const [{ acquireBrowser, holdBrowser, releaseBrowser }, { acquireTab, releaseTab, runInTab }] = await Promise.all([
import("./browser/registry"),
import("./browser/tab-supervisor"),
]);
const timeoutSignal = AbortSignal.timeout(PDF_RENDER_TIMEOUT_MS);
const renderSignal = signal ? AbortSignal.any([signal, timeoutSignal]) : timeoutSignal;
const tabName = `read-pdf-${Bun.randomUUIDv7()}`;
const url = pathToFileURL(absolutePdfPath);
url.hash = `page=${page}&toolbar=0&navpanes=0&view=Fit`;
let browserLease = false;
let tabOpened = false;
let browser: BrowserHandle | undefined;
try {
const acquiredBrowser = await untilAborted(renderSignal, () =>
acquireBrowser({ kind: "headless", headless: true }, { cwd: session.cwd, signal: renderSignal }),
);
browser = acquiredBrowser;
holdBrowser(acquiredBrowser);
browserLease = true;
await untilAborted(renderSignal, () =>
acquireTab(tabName, acquiredBrowser, {
url: url.href,
waitUntil: "load",
timeoutMs: PDF_RENDER_TIMEOUT_MS,
signal: renderSignal,
ownerSessionId: session.getSessionId?.() ?? undefined,
}),
);
tabOpened = true;
await releaseBrowser(acquiredBrowser, { kill: false });
browserLease = false;
const result = await runInTab(tabName, {
code: PDF_SCREENSHOT_CODE,
timeoutMs: PDF_RENDER_TIMEOUT_MS,
signal: renderSignal,
session,
});
const screenshot = result.screenshots.at(-1);
if (!screenshot) throw new ToolError(`Chromium did not capture PDF page ${page}.`);
return screenshot;
} catch (error) {
if (signal?.aborted) throw new ToolAbortError();
if (timeoutSignal.aborted) {
throw new ToolError(`Timed out rendering PDF page ${page} in Chromium.`);
}
throw error;
} finally {
if (tabOpened) await releaseTab(tabName, { kill: false });
if (browserLease && browser) await releaseBrowser(browser, { kill: false });
}
}
+63 -37
View File
@@ -100,7 +100,7 @@ import {
isRemoteMountPath,
type SuffixMatchCache,
} from "./read-path-resolution";
import { readPdfImageMember, rewritePdfImagePlaceholders, splitPdfImageMemberReadPath } from "./read-pdf-images";
import { type PdfImageReadTarget, renderPdfPageScreenshot, splitPdfImageReadPath } from "./read-pdf";
import { isMultiRange, isRawSelector, type ParsedSelector, parseSel, selToOffsetLimit } from "./read-selector";
import { readSqlite, resolveSqliteReadPath } from "./read-sqlite";
import { isProseSummaryPath, renderSummary, routeReadThroughBridge, trySummarize } from "./read-summary";
@@ -398,8 +398,13 @@ type ReadParams = ReadToolInput;
*/
export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
readonly name = "read";
readonly approval = (args: unknown): ToolTier =>
pathTargetsSsh(String((args as { path?: unknown }).path ?? "")) ? "exec" : "read";
readonly approval = (args: unknown): ToolTier => {
let readPath = "";
if (args && typeof args === "object" && "path" in args) readPath = String(args.path ?? "");
if (pathTargetsSsh(readPath)) return "exec";
const target = splitPathAndSel(readPath);
return target.sel === undefined && splitPdfImageReadPath(readPath) ? "exec" : "read";
};
readonly label = "Read";
readonly loadMode = "essential";
description: string;
@@ -546,6 +551,40 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
return toolResult<ReadToolDetails>({ notes, displayReadTargets }).content(content).done();
}
async #readPdfPageScreenshot(options: {
readPath: string;
absolutePdfPath: string;
page: number;
pdfFileSize: number;
suffixResolution?: { from: string; to: string };
signal?: AbortSignal;
}): Promise<AgentToolResult<ReadToolDetails>> {
const { readPath, absolutePdfPath, page, pdfFileSize, suffixResolution, signal } = options;
const screenshot = await renderPdfPageScreenshot(this.session, absolutePdfPath, page, signal);
const screenshotFile = Bun.file(screenshot.dest);
const screenshotMetadata = await readImageMetadata(screenshot.dest);
const loaded = await this.#loadImageContent({
readPath,
absolutePath: screenshot.dest,
mimeType: screenshot.mimeType,
imageMetadata: screenshotMetadata,
fileSize: screenshotFile.size,
});
if (suffixResolution) {
const firstText = loaded.content.find((entry): entry is TextContent => entry.type === "text");
if (firstText) firstText.text = prependSuffixResolutionNotice(firstText.text, suffixResolution);
}
const image = loaded.content.find((entry): entry is ImageContent => entry.type === "image");
const details: ReadToolDetails = {
...loaded.details,
resolvedPath: absolutePdfPath,
contentType: image?.mimeType ?? screenshot.mimeType,
fileSize: pdfFileSize,
suffixResolution,
};
return toolResult(details).content(loaded.content).sourcePath(loaded.sourcePath).done();
}
/**
* Build content blocks for an on-disk image file: an `inspect_image`
* metadata note when inspection is active, otherwise the decoded image
@@ -892,7 +931,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
// Prefer a literal filesystem match over selector interpretation so real
// POSIX filenames containing selector-looking suffixes win over structured
// archive / sqlite / pdf-image dispatch. A selector promoted from local://
// archive / sqlite / unsupported PDF-image dispatch. A selector promoted from local://
// remains separate so it cannot be mistaken for part of the resolved path.
const literalSplit =
promotedSelector === undefined
@@ -903,6 +942,8 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
? readPath.includes(":") && (await probeLiteralPathExists(readPath, this.session.cwd)) !== "missing"
: literalSplit.sel === undefined && splitPathAndSel(readPath).sel !== undefined;
let pdfImageRead: PdfImageReadTarget | null = null;
if (!rawPathIsLiteral) {
const archivePath = await resolveArchiveReadPath(this.session, readPath, suffixCache, signal);
if (archivePath) {
@@ -925,39 +966,14 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
return readSqlite(sqlitePath, signal);
}
const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath);
if (pdfImageMemberPath) {
let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd);
let suffixResolution: { from: string; to: string } | undefined;
try {
const stat = await Bun.file(absolutePdfPath).stat();
if (stat.isDirectory())
throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`);
} catch (error) {
if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error;
const suffixMatch = await findSuffixMatchCached(
this.session,
suffixCache,
pdfImageMemberPath.pdfPath,
signal,
);
if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`);
absolutePdfPath = suffixMatch.absolutePath;
suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath };
}
return readPdfImageMember(
this.session,
this.#autoResizeImages,
absolutePdfPath,
pdfImageMemberPath.pdfPath,
pdfImageMemberPath.member,
suffixResolution,
signal,
);
}
const pdfCandidate = literalSplit.sel === undefined ? splitPdfImageReadPath(readPath) : null;
pdfImageRead =
pdfCandidate && (await probeLiteralPathExists(readPath, this.session.cwd)) === "missing"
? pdfCandidate
: null;
}
const localTarget = literalSplit;
const localTarget = pdfImageRead ? { path: pdfImageRead.pdfPath, sel: undefined } : literalSplit;
const localReadPath = localTarget.path;
const parsed = parseSel(localTarget.sel);
@@ -1036,6 +1052,17 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
return this.#readFileConflicts(absolutePath, suffixResolution, signal);
}
if (pdfImageRead) {
return this.#readPdfPageScreenshot({
readPath,
absolutePdfPath: absolutePath,
page: pdfImageRead.page,
pdfFileSize: fileSize,
suffixResolution,
signal,
});
}
const imageMetadata = await readImageMetadata(absolutePath);
const mimeType = imageMetadata?.mimeType;
const ext = path.extname(absolutePath).toLowerCase();
@@ -1102,8 +1129,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
// Convert document via markit.
const result = await convertFileWithMarkit(absolutePath, signal);
if (result.ok) {
const renderedContent =
ext === ".pdf" ? rewritePdfImagePlaceholders(result.content, resolvedDisplayPath) : result.content;
const renderedContent = result.content;
// Route the converted markdown through the in-memory text builder
// so line-range selectors (`file.pdf:50-100`, `:5-16,40-80`) and
// raw mode apply against the converted output. Without this,
@@ -33,6 +33,27 @@ export interface OpenInEditorOptions {
trimTrailingNewline?: boolean;
}
/** Subprocess argv and Windows quoting mode used to launch an external editor. */
export interface EditorSpawnCommand {
cmd: string[];
windowsVerbatimArguments: boolean;
}
/** Resolves shell argv without letting the host runtime re-quote the editor command. */
export function resolveEditorSpawnCommand(
editorCmd: string,
tmpFile: string,
platform: NodeJS.Platform = process.platform,
): EditorSpawnCommand {
const windows = platform === "win32";
// cmd.exe strips the outer /s /c quote pair; Bun must pass the embedded
// editor/path quotes verbatim instead of applying argv escaping to them.
const cmd = windows
? ["cmd.exe", "/d", "/s", "/c", `"${editorCmd} "${tmpFile}""`]
: [$which("sh") ?? "sh", "-c", `${editorCmd} "$1"`, "sh", tmpFile];
return { cmd, windowsVerbatimArguments: windows };
}
/**
* Opens `content` in the user's external editor and returns the edited text.
* Returns `null` if the editor exits with a non-zero code.
@@ -50,15 +71,13 @@ export async function openInEditor(
try {
await Bun.write(tmpFile, content);
const spawnCommand = resolveEditorSpawnCommand(editorCmd, tmpFile);
const [stdin, stdout, stderr] = options?.stdio ?? ["inherit", "inherit", "inherit"];
const cmd =
process.platform === "win32"
? ["cmd", "/c", `${editorCmd} "${tmpFile}"`]
: [$which("sh") ?? "sh", "-c", `${editorCmd} "$1"`, "sh", tmpFile];
const child = Bun.spawn(cmd, {
const child = Bun.spawn(spawnCommand.cmd, {
stdin,
stdout,
stderr,
windowsVerbatimArguments: spawnCommand.windowsVerbatimArguments,
});
const exitCode = await child.exited;
if (exitCode === 0) {
+6 -44
View File
@@ -1,5 +1,5 @@
import * as path from "node:path";
import { logger, untilAborted } from "@oh-my-pi/pi-utils";
import { untilAborted } from "@oh-my-pi/pi-utils";
import type { ConversionResult, Markit, StreamInfo } from "../markit";
import { ToolAbortError } from "../tools/tool-errors";
import {
@@ -8,7 +8,6 @@ import {
readMarkitConversionCache,
writeMarkitConversionCache,
} from "./markit-cache";
import { loadEmbeddedMupdfWasm } from "./mupdf-wasm-embed";
/**
* File extensions markit can actually convert to markdown — one per registered
@@ -31,53 +30,16 @@ export interface MarkitConversionResult {
export interface MarkitFileConversionOptions {
/**
* Directory the PDF converter writes extracted images/diagrams into. When
* set, each embedded image is rendered to `<id>.png` and referenced by path
* in the markdown; when unset, markit emits an `<!-- image: <id> ... -->`
* placeholder comment instead.
* Directory converters may use for extracted image or diagram files. Since
* those files are conversion side effects, conversions using this option
* bypass the markdown cache.
*/
imageDir?: string;
}
interface MuPdfWasmModuleConfig {
print?: (...values: unknown[]) => void;
printErr?: (...values: unknown[]) => void;
wasmBinary?: Uint8Array;
}
function logMuPdfWasmOutput(stream: "stdout" | "stderr", values: unknown[]): void {
const message = values.length === 1 && typeof values[0] === "string" ? values[0] : values.map(String).join(" ");
logger.debug("mupdf wasm output", { stream, message });
}
// `$libmupdf_wasm_Module` is declared globally (as `any`) by the mupdf package.
// Install print hooks before the WASM module initializes so its stdout/stderr
// route to the file logger instead of corrupting the TUI.
function installMuPdfWasmLogger(): void {
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
moduleConfig.print = (...values: unknown[]) => logMuPdfWasmOutput("stdout", values);
moduleConfig.printErr = (...values: unknown[]) => logMuPdfWasmOutput("stderr", values);
globalThis.$libmupdf_wasm_Module = moduleConfig;
}
// Hand the WASM module its bytes directly when the compiled binary embedded them
// (scripts/embed-mupdf-wasm.ts); a single-file binary has no node_modules for
// mupdf to read `mupdf-wasm.wasm` from. Source/npm builds get undefined here and
// mupdf loads its own wasm. Must run before the mupdf module evaluates.
function installEmbeddedMupdfWasm(): void {
const wasmBinary = loadEmbeddedMupdfWasm();
if (!wasmBinary) return;
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
moduleConfig.wasmBinary = wasmBinary;
globalThis.$libmupdf_wasm_Module = moduleConfig;
}
installMuPdfWasmLogger();
let markit: () => Markit | Promise<Markit> = async () => {
// Lazy: keep the document engine (mammoth/mupdf) off the startup
// import graph — it loads only when a document is first converted.
installEmbeddedMupdfWasm();
// Lazy: keep the document engine off the startup import graph — it loads
// only when a document is first converted.
const promise = import("../markit").then(({ Markit }) => {
const instance = new Markit();
markit = () => instance;
@@ -1,12 +0,0 @@
// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
//
// Compiled single-file binaries cannot let mupdf resolve its `mupdf-wasm.wasm`
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
// wasm bytes via `with { type: "file" }` and copies the wasm next to it. Source
// checkouts, `bun test`, and the npm `dist/cli.js` bundle keep mupdf external and
// load the wasm from node_modules, so this placeholder returns undefined and the
// build resets back to it afterward.
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
return undefined;
}
@@ -2,7 +2,7 @@ import { afterEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as path from "node:path";
import { TempDir } from "@oh-my-pi/pi-utils";
import { getEditorCommand, openInEditor } from "../src/utils/external-editor";
import { getEditorCommand, openInEditor, resolveEditorSpawnCommand } from "../src/utils/external-editor";
interface MutableProcess {
platform: NodeJS.Platform;
@@ -66,6 +66,21 @@ describe("getEditorCommand", () => {
});
describe("openInEditor", () => {
it("passes the cmd.exe command line verbatim on Windows", () => {
const tmpFile = String.raw`C:\Users\Example User\AppData\Local\Temp\omp-editor-123.omp.md`;
expect(resolveEditorSpawnCommand('"C:\\Program Files\\Code.exe" --wait', tmpFile, "win32")).toEqual({
cmd: [
"cmd.exe",
"/d",
"/s",
"/c",
String.raw`""C:\Program Files\Code.exe" --wait "C:\Users\Example User\AppData\Local\Temp\omp-editor-123.omp.md""`,
],
windowsVerbatimArguments: true,
});
});
it.skipIf(process.platform === "win32")("supports quoted editor paths containing spaces", async () => {
const tempDir = TempDir.createSync("@external-editor-");
try {
+8 -1
View File
@@ -30,7 +30,14 @@ const legacyState = {
};
if (Bun.env.MOCK_RPC_EXIT_BEFORE_READY) {
process.stderr.write(Bun.env.MOCK_RPC_EXIT_STDERR ?? "");
const message = Bun.env.MOCK_RPC_EXIT_STDERR ?? "";
if (message) {
// Await the pipe write: exiting immediately can drop unflushed stderr
// bytes, leaving the client's startup error without the failure text.
const { promise, resolve } = Promise.withResolvers<void>();
process.stderr.write(message, () => resolve());
await promise;
}
process.exit(Number(Bun.env.MOCK_RPC_EXIT_BEFORE_READY));
}
@@ -1,4 +1,5 @@
import { afterEach, describe, expect, it } from "bun:test";
import { connectToServer } from "@oh-my-pi/pi-coding-agent/mcp/client";
import { HttpTransport } from "@oh-my-pi/pi-coding-agent/mcp/transports/http";
const encoder = new TextEncoder();
@@ -49,6 +50,58 @@ async function withPendingGuard<T>(promise: Promise<T>, label: string): Promise<
]);
}
describe("MCP Streamable HTTP initialization", () => {
it("sends initialized before opening the optional GET SSE stream", async () => {
const requests: string[] = [];
let initialized = false;
let sessionValid = true;
server = Bun.serve({
port: 0,
async fetch(req) {
if (req.method === "GET") {
requests.push("GET");
if (initialized) return new Response(null, { status: 405 });
sessionValid = false;
return new Response("session is not initialized", { status: 400 });
}
if (req.method === "DELETE") return new Response(null, { status: 204 });
const body = (await req.json()) as { id?: string | number; method: string };
requests.push(body.method);
if (body.method === "initialize") {
const response = {
jsonrpc: "2.0",
id: body.id,
result: {
protocolVersion: "2025-11-25",
capabilities: {},
serverInfo: { name: "session-order", version: "1.0.0" },
},
};
return new Response(`event: message\ndata: ${JSON.stringify(response)}\n\n`, {
headers: {
"Content-Type": "text/event-stream",
"Mcp-Session-Id": "session-order",
},
});
}
if (!sessionValid) return new Response("session terminated", { status: 409 });
initialized = true;
return new Response(null, { status: 202 });
},
});
const connection = await connectToServer("session-order", {
type: "http",
url: `http://127.0.0.1:${server.port}/mcp`,
timeout: GUARD_TIMEOUT_MS,
});
expect(requests).toEqual(["initialize", "notifications/initialized", "GET"]);
await connection.transport.close();
});
});
describe("MCP Streamable HTTP transport timeouts", () => {
it("keeps the request timeout active until a JSON response body is fully read", async () => {
server = Bun.serve({
@@ -8,7 +8,7 @@ import type { OAuthCredentials } from "@oh-my-pi/pi-ai/oauth/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { resolveOllamaModelCacheProviderId } from "@oh-my-pi/pi-catalog/provider-models";
import { resolveModelCacheProviderId, resolveOllamaModelCacheProviderId } from "@oh-my-pi/pi-catalog/provider-models";
import type { ModelSpec, OpenAICompat } from "@oh-my-pi/pi-catalog/types";
import {
applyLlamaCppQwenThinking,
@@ -2562,9 +2562,10 @@ providers:
});
// Emulate a legacy write: the variant has no same-id static header source,
// so it is flagged unrestorable even though its base carries the headers.
writeModelCache("github-copilot", Date.now(), [cachedVariant], true, "", cacheDbPath);
const cacheProviderId = resolveModelCacheProviderId("github-copilot");
writeModelCache(cacheProviderId, Date.now(), [cachedVariant], true, "", cacheDbPath);
const db = new Database(cacheDbPath);
db.run("UPDATE model_cache SET header_restore_version = 0 WHERE provider_id = ?", ["github-copilot"]);
db.run("UPDATE model_cache SET header_restore_version = 0 WHERE provider_id = ?", [cacheProviderId]);
db.close();
const registry = new ModelRegistry(authStorage, modelsJsonPath);
@@ -2585,7 +2586,8 @@ providers:
requestModelId: "gpt-5.6-sol",
headers: { "X-Tenant-Route": "tenant-a" },
});
writeModelCache("github-copilot", Date.now(), [cachedAlias], true, "", cacheDbPath, [bundledBase]);
const cacheProviderId = resolveModelCacheProviderId("github-copilot");
writeModelCache(cacheProviderId, Date.now(), [cachedAlias], true, "", cacheDbPath, [bundledBase]);
const registry = new ModelRegistry(authStorage, modelsJsonPath);
@@ -4,6 +4,7 @@ import * as os from "node:os";
import * as path from "node:path";
import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import { getBundledModels } from "@oh-my-pi/pi-catalog/models";
import { resolveModelCacheProviderId } from "@oh-my-pi/pi-catalog/provider-models";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { removeSyncWithRetries } from "@oh-my-pi/pi-utils";
@@ -29,7 +30,8 @@ describe("startup model cache header restoration (#5780)", () => {
expect(withHeaders.length).toBeGreaterThan(0);
// Prior process: cache the live copilot catalog. v10 never persists headers.
writeModelCache("github-copilot", Date.now(), bundled, true, "fp-test", dbPath, bundled);
const cacheProviderId = resolveModelCacheProviderId("github-copilot");
writeModelCache(cacheProviderId, Date.now(), bundled, true, "fp-test", dbPath, bundled);
const raw = fs.readFileSync(dbPath).toString("latin1");
for (const model of withHeaders) {
for (const value of Object.values(model.headers ?? {})) {
@@ -0,0 +1,35 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import * as piNatives from "@oh-my-pi/pi-natives";
import { PdfConverter } from "../src/markit/converters/pdf";
describe("PdfConverter", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("keeps accepting PDF extensions and MIME types", () => {
const converter = new PdfConverter();
expect(converter.accepts({ extension: ".pdf" })).toBe(true);
expect(converter.accepts({ mimetype: "application/pdf" })).toBe(true);
expect(converter.accepts({ mimetype: "application/pdf; charset=binary" })).toBe(true);
expect(converter.accepts({ mimetype: "application/x-pdf" })).toBe(true);
expect(converter.accepts({ extension: ".txt", mimetype: "text/plain" })).toBe(false);
});
it("returns a browser and OCR notice for an image-only PDF", async () => {
vi.spyOn(piNatives, "pdfToMarkdown").mockResolvedValue({
markdown: "",
pageCount: 3,
pagesNeedingOcr: [1, 3],
hasEncodingIssues: false,
});
const result = await new PdfConverter().convert(Buffer.from("image-only pdf"), { extension: ".pdf" });
expect(result.markdown).toBe(
"Text extraction is incomplete for PDF pages 1, 3. Use the browser tool to render those pages or OCR them.",
);
expect(result.markdown.length).toBeGreaterThan(0);
});
});
@@ -1247,6 +1247,11 @@ describe("createAgentSession defaultInactive tool activation", () => {
const errors: string[] = [];
const unsubscribe = runner.onError(error => {
errors.push(error.error);
// The 10ms budget exists only to reap the stalled first handler
// quickly; handlers run sequentially and the budget is read per
// handler, so restoring it here keeps machine load from timing out
// the genuine recovery registration too (flaked in full-suite runs).
testSetExtensionHandlerTimeoutMs(EXTENSION_HANDLER_TIMEOUT_MS);
});
testSetExtensionHandlerTimeoutMs(10);
@@ -1454,6 +1459,10 @@ describe("createAgentSession defaultInactive tool activation", () => {
releaseStalledRegistration.resolve();
const failure = await detachedFailure.promise;
// Restore the default budget before the recovered registration flush:
// the 10ms budget was only for reaping the stalled activation, and the
// real presentation pass can exceed it under full-suite load.
testSetExtensionHandlerTimeoutMs(EXTENSION_HANDLER_TIMEOUT_MS);
releaseRecoveredRegistration.resolve();
await recoveredActivation.promise;
@@ -1,447 +0,0 @@
/**
* PDF image extraction: markit emits inert `<!-- image: <id> ... -->`
* placeholders for embedded PDF images. The read tool rewrites those into
* browsable `read <pdf>:<id>.png` handles, and serves the actual PNG when that
* handle is read — extracting via markit's `imageDir` into a session-artifact
* cache. These lock the rewrite, the member extraction, member validation, and
* the caching contract.
*/
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
import * as piUtils from "@oh-my-pi/pi-utils";
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
// 1x1 transparent PNG — small enough to pass through image loading untouched.
const TINY_PNG = Buffer.from(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg==",
"base64",
);
function makeSession(testDir: string): ToolSession {
const sessionFile = path.join(testDir, "session.jsonl");
const artifactsDir = sessionFile.slice(0, -6);
return {
cwd: testDir,
hasUI: false,
getSessionFile: () => sessionFile,
getArtifactsDir: () => artifactsDir,
getSessionSpawns: () => null,
settings: Settings.isolated({ "images.autoResize": false }),
} as unknown as ToolSession;
}
/** Spy on markit so PDF "extraction" writes the given members into imageDir. */
function mockExtraction(members: Record<string, Buffer> = { "p11-img0.png": TINY_PNG }) {
return vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_filePath: string, _signal, options) => {
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
for (const name in members) {
fs.writeFileSync(path.join(options.imageDir, name), members[name]!);
}
}
return { ok: true, content: "" };
});
}
function imageBytes(result: AgentToolResult<ReadToolDetails>): Buffer {
const image = result.content.find(content => content.type === "image");
if (image?.type !== "image") throw new Error("Expected an image result");
return Buffer.from(image.data, "base64");
}
function mockBlockedExtraction() {
const entered = Promise.withResolvers<void>();
const release = Promise.withResolvers<void>();
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_sourcePath, signal, options) => {
entered.resolve();
await release.promise;
signal?.throwIfAborted();
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
}
return { ok: true, content: "" };
});
return { entered, release, spy };
}
/**
* Resolves once `count` callers are attached as waiters on the shared PDF
* extraction. Waiter attachment is the only `untilAborted` call that receives
* a promise (source snapshots pass thunks), so counting promise arguments
* observes it. The abort tests need this barrier: aborting a caller while it
* is the sole waiter tears the extraction down and deadlocks against the
* blocked conversion mock.
*/
function extractionWaitersAttached(count: number): Promise<void> {
const attached = Promise.withResolvers<void>();
const original = piUtils.untilAborted;
let seen = 0;
vi.spyOn(piUtils, "untilAborted").mockImplementation((signal, pr) => {
if (typeof pr !== "function" && ++seen === count) attached.resolve();
return original(signal, pr);
});
return attached.promise;
}
describe("read PDF image extraction", () => {
let testDir: string;
let pdfPath: string;
beforeEach(() => {
testDir = path.join(os.tmpdir(), `read-pdf-img-${Snowflake.next()}`);
fs.mkdirSync(testDir, { recursive: true });
pdfPath = path.join(testDir, "doc.pdf");
fs.writeFileSync(pdfPath, "%PDF-stub");
});
afterEach(() => {
vi.restoreAllMocks();
removeSyncWithRetries(testDir);
});
it("rewrites image placeholders into browse handles on a full read", async () => {
const converted = [
"Heading",
"",
"<!-- image: p11-img0 (page 11, 199x124pt) -->",
"",
"<!-- image: p11-img1 (page 11, 199x54pt) -->",
"",
"Footer",
].join("\n");
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: converted });
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: pdfPath });
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).not.toContain("<!-- image:");
expect(text).toContain("read `doc.pdf:p11-img0.png`");
expect(text).toContain("read `doc.pdf:p11-img1.png`");
// Page/size metadata is preserved in the handle text.
expect(text).toContain("page 11, 199x124pt");
});
it("rewrites placeholders inside a line-range view", async () => {
const lines = Array.from({ length: 20 }, (_, i) => `pdf line ${i + 1}`);
lines[9] = "<!-- image: p3-img0 (page 3, 100x50pt) -->"; // line 10
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: lines.join("\n") });
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: `${pdfPath}:8-12` });
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).not.toContain("<!-- image:");
expect(text).toContain("read `doc.pdf:p3-img0.png`");
});
it("extracts a PDF image member as an inline image block", async () => {
const spy = mockExtraction();
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
const image = result.content.find(c => c.type === "image");
expect(image).toBeDefined();
expect(image && "mimeType" in image ? image.mimeType : undefined).toBe("image/png");
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).toContain("Read image file");
// Extraction was driven through markit with an imageDir target.
expect(spy).toHaveBeenCalledTimes(1);
expect(spy.mock.calls[0]?.[2]?.imageDir).toBeTruthy();
});
it("reuses the extraction cache across member reads", async () => {
const spy = mockExtraction();
const tool = new ReadTool(makeSession(testDir));
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
// Second read is served from the `.extracted` cache, not re-converted.
expect(spy).toHaveBeenCalledTimes(1);
});
it("re-extracts image members after same-path PDF replacement", async () => {
const sourceA = Buffer.from("%PDF-source-a");
const sourceB = Buffer.from("%PDF-source-b");
fs.writeFileSync(pdfPath, sourceA);
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(
path.join(options.imageDir, "p11-img0.png"),
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
);
}
return { ok: true, content: "" };
});
const tool = new ReadTool(makeSession(testDir));
const originalStat = fs.statSync(pdfPath);
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
fs.writeFileSync(pdfPath, sourceB);
fs.utimesSync(pdfPath, originalStat.atime, originalStat.mtime);
const second = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
expect(imageBytes(first).subarray(TINY_PNG.length)).toEqual(sourceA);
expect(imageBytes(second).subarray(TINY_PNG.length)).toEqual(sourceB);
expect(spy).toHaveBeenCalledTimes(2);
});
it("converts an immutable snapshot when the source changes during extraction", async () => {
const sourceA = Buffer.from("%PDF-source-a");
const sourceB = Buffer.from("%PDF-source-b");
fs.writeFileSync(pdfPath, sourceA);
const entered = Promise.withResolvers<void>();
const release = Promise.withResolvers<void>();
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
entered.resolve();
await release.promise;
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(
path.join(options.imageDir, "p11-img0.png"),
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
);
}
return { ok: true, content: "" };
});
const pending = new ReadTool(makeSession(testDir)).execute("call", { path: `${pdfPath}:p11-img0.png` });
await entered.promise;
fs.writeFileSync(pdfPath, sourceB);
release.resolve();
const result = await pending;
expect(imageBytes(result).subarray(TINY_PNG.length)).toEqual(sourceA);
});
it("coalesces concurrent cold image extraction", async () => {
const { entered, release, spy } = mockBlockedExtraction();
const tool = new ReadTool(makeSession(testDir));
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
const second = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await entered.promise;
const conversionCount = spy.mock.calls.length;
release.resolve();
const [firstResult, secondResult] = await Promise.all([first, second]);
expect(conversionCount).toBe(1);
expect(imageBytes(firstResult)).toEqual(imageBytes(secondResult));
});
it("keeps shared extraction running when its owner aborts", async () => {
const { entered, release, spy } = mockBlockedExtraction();
const bothAttached = extractionWaitersAttached(2);
const tool = new ReadTool(makeSession(testDir));
const ownerController = new AbortController();
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, ownerController.signal);
await entered.promise;
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await bothAttached;
ownerController.abort();
await expect(owner).rejects.toThrow(/Aborted|Cancelled/);
release.resolve();
const result = await joiner;
expect(result.content.some(content => content.type === "image")).toBe(true);
expect(spy).toHaveBeenCalledTimes(1);
});
it("keeps shared extraction running when a joiner aborts", async () => {
const { entered, release, spy } = mockBlockedExtraction();
const bothAttached = extractionWaitersAttached(2);
const tool = new ReadTool(makeSession(testDir));
const joinerController = new AbortController();
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await entered.promise;
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, joinerController.signal);
await bothAttached;
joinerController.abort();
await expect(joiner).rejects.toThrow(/Aborted|Cancelled/);
release.resolve();
const result = await owner;
expect(result.content.some(content => content.type === "image")).toBe(true);
expect(spy).toHaveBeenCalledTimes(1);
});
it("cleans temporary extraction state when the only caller aborts", async () => {
const entered = Promise.withResolvers<void>();
let snapshotPath: string | undefined;
let stagingDir: string | undefined;
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, signal, options) => {
snapshotPath = sourcePath;
stagingDir = options?.imageDir;
entered.resolve();
const aborted = Promise.withResolvers<void>();
const onAbort = () => aborted.resolve();
if (signal?.aborted) onAbort();
else signal?.addEventListener("abort", onAbort, { once: true });
await aborted.promise;
signal?.removeEventListener("abort", onAbort);
signal?.throwIfAborted();
return { ok: true, content: "" };
});
const controller = new AbortController();
const pending = new ReadTool(makeSession(testDir)).execute(
"call",
{ path: `${pdfPath}:p11-img0.png` },
controller.signal,
);
await entered.promise;
controller.abort();
await expect(pending).rejects.toThrow(/Aborted|Cancelled/);
if (!snapshotPath || !stagingDir) throw new Error("Expected extraction paths");
expect(fs.existsSync(path.dirname(snapshotPath))).toBe(false);
expect(fs.existsSync(stagingDir)).toBe(false);
});
it("does not let a failed generation delete a replacement generation", async () => {
const sourceA = Buffer.from("%PDF-source-a");
const sourceB = Buffer.from("%PDF-source-b");
fs.writeFileSync(pdfPath, sourceA);
const firstEntered = Promise.withResolvers<void>();
const failFirst = Promise.withResolvers<void>();
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
const source = fs.readFileSync(sourcePath);
if (source.equals(sourceA)) {
firstEntered.resolve();
await failFirst.promise;
return { ok: false, content: "", error: "generation A failed" };
}
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), Buffer.concat([TINY_PNG, source]));
}
return { ok: true, content: "" };
});
const tool = new ReadTool(makeSession(testDir));
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await firstEntered.promise;
fs.writeFileSync(pdfPath, sourceB);
const replacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
failFirst.resolve();
await expect(first).rejects.toThrow(/Cannot extract images/);
const cachedReplacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
expect(imageBytes(replacement).subarray(TINY_PNG.length)).toEqual(sourceB);
expect(imageBytes(cachedReplacement)).toEqual(imageBytes(replacement));
expect(spy).toHaveBeenCalledTimes(2);
});
it("isolates equal-content PDFs with the same basename in different directories", async () => {
const otherDir = path.join(testDir, "other");
const otherPdfPath = path.join(otherDir, path.basename(pdfPath));
fs.mkdirSync(otherDir, { recursive: true });
fs.writeFileSync(otherPdfPath, fs.readFileSync(pdfPath));
let conversion = 0;
const spy = vi
.spyOn(markit, "convertFileWithMarkit")
.mockImplementation(async (_sourcePath, _signal, options) => {
conversion++;
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(
path.join(options.imageDir, "p11-img0.png"),
Buffer.concat([TINY_PNG, Buffer.from(String(conversion))]),
);
}
return { ok: true, content: "" };
});
const tool = new ReadTool(makeSession(testDir));
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
const second = await tool.execute("call", { path: `${otherPdfPath}:p11-img0.png` });
expect(imageBytes(first).subarray(TINY_PNG.length).toString()).toBe("1");
expect(imageBytes(second).subarray(TINY_PNG.length).toString()).toBe("2");
expect(spy).toHaveBeenCalledTimes(2);
});
it("supports PDF basenames at the filesystem component limit", async () => {
const longPdfPath = path.join(testDir, `${"a".repeat(250)}.pdf`);
fs.writeFileSync(longPdfPath, "%PDF-stub");
mockExtraction();
const result = await new ReadTool(makeSession(testDir)).execute("call", {
path: `${longPdfPath}:p11-img0.png`,
});
expect(result.content.some(content => content.type === "image")).toBe(true);
});
it("errors with the available members for an unknown member", async () => {
mockExtraction();
const tool = new ReadTool(makeSession(testDir));
await expect(tool.execute("call", { path: `${pdfPath}:does-not-exist.png` })).rejects.toThrow(
/not found.*p11-img0\.png/s,
);
});
it("rejects member traversal attempts", async () => {
mockExtraction();
const tool = new ReadTool(makeSession(testDir));
// `../../escape.png` matches the image-member shape but is not a known
// basename, so it must be refused rather than joined into the cache path.
await expect(tool.execute("call", { path: `${pdfPath}:../../escape.png` })).rejects.toThrow(/not found/);
});
it("lists extractable members for a trailing-colon read", async () => {
mockExtraction({ "p1-img0.png": TINY_PNG, "p2-img0.png": TINY_PNG });
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: `${pdfPath}:` });
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).toContain(`read \`${pdfPath}:p1-img0.png\``);
expect(text).toContain(`read \`${pdfPath}:p2-img0.png\``);
});
it("does not cache a failed conversion", async () => {
let failedSnapshotPath: string | undefined;
let failedImageDir: string | undefined;
const spy = vi.spyOn(markit, "convertFileWithMarkit");
spy.mockImplementationOnce(async (sourcePath, _signal, options) => {
failedSnapshotPath = sourcePath;
failedImageDir = options?.imageDir;
return { ok: false, content: "", error: "boom" };
});
const tool = new ReadTool(makeSession(testDir));
await expect(tool.execute("call", { path: `${pdfPath}:p11-img0.png` })).rejects.toThrow(/Cannot extract images/);
if (!failedSnapshotPath || !failedImageDir) throw new Error("Expected failed extraction paths");
expect(fs.existsSync(path.dirname(failedSnapshotPath))).toBe(false);
expect(fs.existsSync(path.join(failedImageDir, ".extracted"))).toBe(false);
spy.mockImplementationOnce(async (_filePath: string, _signal, options) => {
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
}
return { ok: true, content: "" };
});
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
expect(result.content.some(c => c.type === "image")).toBe(true);
expect(spy).toHaveBeenCalledTimes(2);
});
});
@@ -0,0 +1,95 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
import * as pdfRead from "@oh-my-pi/pi-coding-agent/tools/read-pdf";
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
import { removeWithRetries } from "@oh-my-pi/pi-utils";
const ONE_PX_PNG = Buffer.from(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAIAAACQd1PeAAAADElEQVR4nGP4z8AAAAMBAQDJ/pLvAAAAAElFTkSuQmCC",
"base64",
);
function makeSession(cwd: string): ToolSession {
return {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated({ "images.autoResize": false, "inspect_image.mode": "off" }),
} as ToolSession;
}
function textOf(result: AgentToolResult<ReadToolDetails>): string {
return result.content
.filter(entry => entry.type === "text")
.map(entry => entry.text)
.join("\n");
}
describe("read PDF page screenshots", () => {
let testDir: string;
let pdfPath: string;
let screenshotPath: string;
beforeEach(async () => {
testDir = await fs.mkdtemp(path.join(os.tmpdir(), "read-pdf-page-"));
pdfPath = path.join(testDir, "doc.pdf");
await fs.writeFile(pdfPath, `%PDF-stub-${testDir}`);
screenshotPath = path.join(testDir, "rendered.png");
await fs.writeFile(screenshotPath, ONE_PX_PNG);
});
afterEach(async () => {
vi.restoreAllMocks();
await removeWithRetries(testDir);
});
it("renders former image-member reads through Chromium", async () => {
const render = vi.spyOn(pdfRead, "renderPdfPageScreenshot").mockResolvedValue({
dest: screenshotPath,
mimeType: "image/png",
bytes: ONE_PX_PNG.byteLength,
width: 1,
height: 1,
});
const tool = new ReadTool(makeSession(testDir));
for (const [readPath, page] of [
[`${pdfPath}:`, 1],
[`${pdfPath}:p2-img0.png`, 2],
] as const) {
const result = await tool.execute("read-pdf-image", { path: readPath });
expect(result.content.some(entry => entry.type === "image" && entry.mimeType === "image/png")).toBe(true);
expect(textOf(result)).toContain("Read image file [image/png]");
expect(result.details?.resolvedPath).toBe(pdfPath);
expect(render).toHaveBeenLastCalledWith(expect.anything(), pdfPath, page, undefined);
}
expect(tool.approval({ path: `${pdfPath}:p1-img0.png` })).toBe("exec");
expect(tool.approval({ path: `${pdfPath}:2-2` })).toBe("read");
});
it("preserves a literal filename that looks like a PDF image listing", async () => {
const literalPath = `${pdfPath}:`;
await fs.writeFile(literalPath, "literal colon path wins\n");
const result = await new ReadTool(makeSession(testDir)).execute("read-literal", { path: literalPath });
expect(textOf(result)).toContain("literal colon path wins");
});
it("routes PDF line selectors through normal document conversion", async () => {
const convert = vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({
ok: true,
content: "first line\nselected line\nthird line\n",
});
const result = await new ReadTool(makeSession(testDir)).execute("read-pdf-lines", { path: `${pdfPath}:2-2` });
expect(convert).toHaveBeenCalledTimes(1);
expect(textOf(result)).toContain("selected line");
});
});
@@ -105,7 +105,7 @@ describe("document conversion cache", () => {
it("skips cache for imageDir conversions", async () => {
const convert = vi.spyOn(Markit.prototype, "convert").mockResolvedValue({ markdown: "image body" });
const docPath = path.join(testDir, "image-doc.pdf");
const docPath = path.join(testDir, "image-doc.docx");
await fs.writeFile(docPath, new TextEncoder().encode("image bytes"));
const imageDir = path.join(testDir, "images");
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/hashline",
"version": "17.3.3",
"version": "17.3.4",
"description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+6
View File
@@ -2,6 +2,12 @@
## [Unreleased]
## [17.3.4] - 2026-08-14
### Fixed
- Fixed `recall()` silently dropping `scope='global'` rows whenever a `channelId` filter was active: `buildWhere()` appended a redundant hard `channel_id = ?` clause on top of the `(session_id = ? OR scope = 'global' OR channel_id = ?)` visibility clause, so global rows whose `channel_id` didn't match (e.g. imported rows with `channel_id NULL`) were excluded. Channel isolation is preserved by the visibility clause alone. This made imported/global episodic memory permanently unrecallable through callers that always pass a channel (such as the coding-agent memory backend). ([#8525](https://github.com/can1357/oh-my-pi/issues/8525))
## [17.2.11] - 2026-08-07
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-mnemopi",
"version": "17.3.3",
"version": "17.3.4",
"description": "Local SQLite memory engine for Oh My Pi agents",
"homepage": "https://omp.sh",
"author": "Can Boluk",
-4
View File
@@ -527,10 +527,6 @@ function buildWhere(
clauses.push(`${prefix}author_type = ?`);
params.push(authorType);
}
if (channelId !== null && channelId !== "") {
clauses.push(`${prefix}channel_id = ?`);
params.push(channelId);
}
return { where: clauses.join(" AND "), params };
}
@@ -116,6 +116,28 @@ describe("beam recall free functions", () => {
expect(new Set(results.map(result => result.tier_label))).toEqual(new Set(["working", "episodic"]));
});
it("keeps global episodic rows visible when a channelId filter is active while still isolating other channels", async () => {
const beam = makeBeam();
// Global row with channel_id NULL — the shape produced by importFromDict()
// and by any cross-channel/global memory. Must survive a channelId filter.
insertEpisodic(beam, "em-global", "quokka migration protocol uses base64 snapshots");
// Non-global row owned by a different channel/session — must stay hidden.
beam.db.run(
"INSERT INTO episodic_memory (id, content, source, timestamp, session_id, importance, scope, channel_id, veracity, memory_type) VALUES ('em-other-channel', 'quokka migration protocol for team beta', 'test', ?, 'other-session', 0.5, 'session', 'other-bank', 'unknown', 'general')",
["2026-05-30T12:00:00.000Z"],
);
const results = await recall(beam, "quokka migration protocol", 5, {
queryTime: "2026-05-30T12:00:00.000Z",
channelId: "project-bank",
includeWorking: false,
});
const ids = results.map(result => result.id);
expect(ids).toContain("em-global");
expect(ids).not.toContain("em-other-channel");
});
it("boosts memories near the requested temporal target", async () => {
const beam = makeBeam();
insertEpisodic(beam, "em-old", "incident alpha resolved by rotating credentials", {
+7
View File
@@ -2,6 +2,12 @@
## [Unreleased]
## [17.3.4] - 2026-08-14
### Added
- Added the async `pdfToMarkdown` native API backed by `pdf-inspector`, with page numbering, page-count, OCR-needed-page, and encoding-issue metadata.
### Changed
- Docker images (`Dockerfile`, `scripts/install-tests/*.dockerfile`) build the native addon through the cargo/napi-rs backend (`OMP_NATIVE_BUILD_BACKEND=cargo`) instead of Bazel: a single fixed host target gains nothing from hermetic cross toolchains, and none of those images shipped bazelisk. `OMP_NATIVE_CARGO_PROFILE` picks the profile for that path (images use `ci`, local default stays `local`).
@@ -10,6 +16,7 @@
- Fixed the root Cargo workspace failing to load when a stale directory exists under `crates/` — e.g. a deleted crate whose directory survived `git reset --hard`. `members` no longer globs `crates/pi-*`, so a directory without a `Cargo.toml` can no longer break every cargo and Bazel build.
- Fixed Docker build contexts shipping nested build output: `.dockerignore` patterns are anchored at the context root, so bare `target/` and `dist/` matched neither `go-port/*/target` (~1.4 GB) nor `packages/*/dist` (~600 MB).
- Fixed `deviceCheckGenerateToken` aborting the whole process with `SIGTRAP` when called from a macOS session without GUI/graphic access (SSH, a launchd `LaunchDaemon`, a CI runner, a service account, a sandbox), which made every `openai-codex/*` OAuth model unusable for such accounts. `-[DCDevice isSupported]` synchronously opens an XPC connection to the per-user DeviceCheck metadata daemon, which exists only in an interactive GUI login session; without one the connection setup hits `_xpc_api_misuse` and traps before any completion handler runs, so the promise never rejects. The binding now checks the caller's security session for the `sessionHasGraphicAccess` attribute first and resolves `{ supported: false, error: … }` instead of touching DeviceCheck when it is absent ([#8353](https://github.com/can1357/oh-my-pi/issues/8353)).
## [17.3.1] - 2026-08-13
+6 -1
View File
@@ -10,6 +10,7 @@ Native Rust functionality via N-API.
- **Audio**: Cross-platform low-latency microphone capture and gapless speaker playback
- **WebRTC**: Native Opus media, SDP offer/answer negotiation, and data-channel events for live sessions
- **File locking**: Process-owned cross-process locks with in-memory kernel names on Linux/Windows and `flock(2)` sidecars on other Unix platforms
- **PDF**: In-memory PDF-to-Markdown extraction with OCR-page classification via `pdf-inspector`
General-purpose image processing (decode/resize/encode for files and buffers)
lives in [`Bun.Image`](https://bun.com/docs/runtime/image) on the JS side; this
@@ -19,7 +20,7 @@ that terminal protocol.
## Usage
```typescript
import { grep, find, encodeSixel } from "@oh-my-pi/pi-natives";
import { encodeSixel, grep, pdfToMarkdown } from "@oh-my-pi/pi-natives";
// Grep for a pattern
const results = await grep({
@@ -38,6 +39,10 @@ const files = await find({
// SIXEL encode for a terminal cell box (px)
const sequence = encodeSixel(pngBytes, widthPx, heightPx);
// Extract PDF text and identify pages that still need OCR
const pdf = await pdfToMarkdown(pdfBytes);
console.log(pdf.markdown, pdf.pagesNeedingOcr);
```
## Building
+27 -1
View File
@@ -279,7 +279,7 @@ export declare function __ompInstallTokioRuntime(): void
* `packages/natives/native/index.js` (which derives the name from
* `package.json#version`).
*/
export declare function __piNativesV17_3_3(): void
export declare function __piNativesV17_3_4(): void
/**
* Apply ast-grep rewrite rules to matching files; honors `dryRun` and returns
@@ -1578,6 +1578,32 @@ export interface PatchHunk {
lines: Array<string>
}
/** Markdown and inspection metadata produced from a PDF document. */
export interface PdfMarkdownResult {
/** Extracted document content in Markdown format. */
markdown: string
/** Document title from PDF metadata, when present. */
title?: string
/** Total number of pages in the document. */
pageCount: number
/** One-indexed page numbers whose content requires OCR. */
pagesNeedingOcr: Array<number>
/** Whether the document contains text encoding problems. */
hasEncodingIssues: boolean
}
/**
* Convert an in-memory PDF to Markdown and return its inspection metadata.
*
* Conversion copies the typed array before dispatch so JavaScript mutation
* cannot race the native worker.
*
* # Errors
* Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
* be parsed or converted.
*/
export declare function pdfToMarkdown(input: Uint8Array): Promise<PdfMarkdownResult>
export interface PointerOptions {
button?: string
count?: number
+2 -1
View File
@@ -29,7 +29,7 @@ export const Shell = nativeBindings.Shell;
// functions
export const __ompInstallTokioRuntime = nativeBindings.__ompInstallTokioRuntime;
export const __piNativesV17_3_3 = nativeBindings.__piNativesV17_3_3;
export const __piNativesV17_3_4 = nativeBindings.__piNativesV17_3_4;
export const astEdit = nativeBindings.astEdit;
export const astGrep = nativeBindings.astGrep;
export const astMatch = nativeBindings.astMatch;
@@ -69,6 +69,7 @@ export const matchesLegacySequence = nativeBindings.matchesLegacySequence;
export const mmrRerankIndices = nativeBindings.mmrRerankIndices;
export const parseKey = nativeBindings.parseKey;
export const parseKittySequence = nativeBindings.parseKittySequence;
export const pdfToMarkdown = nativeBindings.pdfToMarkdown;
export const readImageFromClipboard = nativeBindings.readImageFromClipboard;
export const renderSnapcompactPng = nativeBindings.renderSnapcompactPng;
export const search = nativeBindings.search;
+1
View File
@@ -26,6 +26,7 @@ export interface DetectCompiledBinaryInput {
export function detectCompiledBinary(input: DetectCompiledBinaryInput): boolean;
export interface GetAddonFilenamesInput {
tag: string;
arch: string;
-1
View File
@@ -87,7 +87,6 @@ export function detectCompiledBinary({ embeddedAddon, env, importMetaUrl }) {
}
return false;
}
/**
* @param {{ tag: string; arch: string; variant: "modern" | "baseline" | null | undefined }} input
* @returns {string[]}
+2 -2
View File
@@ -1,7 +1,7 @@
{
"name": "@oh-my-pi/pi-natives",
"version": "17.3.3",
"description": "Native Rust bindings for audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API",
"version": "17.3.4",
"description": "Native Rust bindings for PDF conversion, audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API",
"type": "module",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+38
View File
@@ -23,6 +23,7 @@ import {
matchesKey,
PtySession,
parseKey,
pdfToMarkdown,
summarizeCode,
supportsLanguage,
truncateToWidth,
@@ -87,6 +88,30 @@ async function createFifo(fifoPath: string) {
throw new Error(await new Response(process.stderr).text());
}
function textPdf(text: string): Uint8Array {
const stream = `BT /F1 12 Tf 72 720 Td (${text}) Tj ET`;
const objects = [
"<< /Type /Catalog /Pages 2 0 R >>",
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
`<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`,
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>",
];
let document = "%PDF-1.4\n";
const offsets: number[] = [];
for (const [index, object] of objects.entries()) {
offsets.push(document.length);
document += `${index + 1} 0 obj\n${object}\nendobj\n`;
}
const xrefOffset = document.length;
document += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
for (const offset of offsets) {
document += `${offset.toString().padStart(10, "0")} 00000 n \n`;
}
document += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xrefOffset}\n%%EOF\n`;
return Buffer.from(document);
}
describe("pi-natives", () => {
beforeAll(async () => {
await setupFixtures();
@@ -763,6 +788,19 @@ describe("pi-natives", () => {
expect(await Bun.file(markerPath).exists()).toBe(false);
});
});
describe("pdfToMarkdown", () => {
it("isolates blocking conversion from later JavaScript buffer mutation", async () => {
const input = textPdf("Copied PDF bytes");
const conversion = pdfToMarkdown(input);
input.fill(0);
const result = await conversion;
expect(result.pageCount).toBe(1);
expect(result.markdown).toContain("Copied PDF bytes");
});
});
describe("htmlToMarkdown", () => {
it("should convert basic HTML to markdown", async () => {
const html = "<h1>Hello World</h1><p>This is a paragraph.</p>";
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/omptype",
"version": "17.3.3",
"version": "17.3.4",
"description": "ArkType-compatible runtime schema validation with lazy JIT compilation",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/snapcompact",
"version": "17.3.3",
"version": "17.3.4",
"description": "Bitmap-frame context compression for vision-capable LLMs",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/omp-stats",
"version": "17.3.3",
"version": "17.3.4",
"description": "Local observability dashboard for pi AI usage statistics",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+6
View File
@@ -2,6 +2,12 @@
## [Unreleased]
## [17.3.4] - 2026-08-14
### Fixed
- Fixed a terminal Device-Attributes reply leaking into the composer as literal text (e.g. `1;22;…;52c`) when it arrived after the startup capability-probe sentinel FIFO drained, a race made observable by the added latency of an SSH/zmx PTY chain. DA1 replies (`CSI ? … c`) and split private-CSI responses are now consumed for the whole session lifetime, not only while a probe sentinel is outstanding ([#8542](https://github.com/can1357/oh-my-pi/issues/8542)).
## [17.3.3] - 2026-08-14
### Fixed
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-tui",
"version": "17.3.3",
"version": "17.3.4",
"description": "Terminal User Interface library with differential rendering for efficient text-based applications",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+15 -8
View File
@@ -945,10 +945,11 @@ export class ProcessTerminal implements Terminal {
// flush timeout elapses mid-sequence, the prefix `\x1b[?<digits>` arrives as
// one event and the tail `;...<terminator>` arrives as individual character
// events that would otherwise leak into the prompt as keystrokes. See #1238.
if (
this.#privateCsiResponseBuffer ||
(privateCsiPartialPattern.test(sequence) && this.#da1SentinelOwners.length > 0)
) {
// A private CSI (`\x1b[?…`) is a terminal->host report, never a keystroke, so
// reassembly stays armed for the whole session — not just while a probe
// sentinel is outstanding — otherwise a reply that lands after the sentinel
// FIFO drains (slow SSH/PTY links) leaks its tail into the composer (#8542).
if (this.#privateCsiResponseBuffer || privateCsiPartialPattern.test(sequence)) {
if (this.#privateCsiResponseBuffer && sequence.startsWith("\x1b")) {
// New escape arrived mid-reassembly — abandon partial and re-process the new sequence.
this.#privateCsiResponseBuffer = "";
@@ -1038,10 +1039,16 @@ export class ProcessTerminal implements Terminal {
}
// DA1 response: swallow our sentinel reply regardless of whether an
// earlier capability-specific response already succeeded. Other terminal
// probes should never see these replies.
if (da1ResponsePattern.test(sequence) && this.#da1SentinelOwners.length > 0) {
const owner = this.#da1SentinelOwners.shift()!;
// earlier capability-specific response already succeeded. `CSI ? … c` is
// exclusively a terminal->host report, so it is swallowed even with no
// outstanding sentinel — a reply that arrives after the FIFO drains (slow
// SSH/PTY links) must never reach the composer as literal text (#8542).
if (da1ResponsePattern.test(sequence)) {
const owner = this.#da1SentinelOwners.shift();
if (!owner) {
// Late/unowned reply: nothing to resolve, just drop the bytes.
return;
}
switch (owner.kind) {
case "osc11": {
if (this.#osc11Pending) {
+102
View File
@@ -0,0 +1,102 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import { ProcessTerminal } from "@oh-my-pi/pi-tui/terminal";
import { setTerminalHeadless } from "@oh-my-pi/pi-utils";
// #8542: a terminal Device-Attributes reply to omp's startup capability probe
// leaks into the composer as literal text (`1;22;...;52c`) when it arrives
// after the DA1 sentinel FIFO has already drained. The extra SSH+zmx PTY hops
// slow the query->response round-trip enough to make the race observable.
//
// Contract: `CSI ? … c` is exclusively a terminal->host report, never a
// keystroke, so it MUST be consumed for the whole session lifetime and never
// forwarded to the input handler that feeds the composer.
// A meaty multi-parameter DA1 reply, exactly as the reporter observed it.
const DA1_REPLY = "\x1b[?1;22;23;24;28;32;42;52c";
const stdinIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "isTTY");
const stdoutIsTtyDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "isTTY");
const stdinSetRawModeDescriptor = Object.getOwnPropertyDescriptor(process.stdin, "setRawMode");
const stdoutColumnsDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "columns");
const stdoutRowsDescriptor = Object.getOwnPropertyDescriptor(process.stdout, "rows");
function restoreProperty(target: object, key: string, descriptor: PropertyDescriptor | undefined): void {
if (descriptor) {
Object.defineProperty(target, key, descriptor);
return;
}
delete (target as Record<string, unknown>)[key];
}
describe("issue #8542: late DA response must not leak into the composer", () => {
let terminal: ProcessTerminal | undefined;
let previousHeadless = false;
let spies: Array<{ mockRestore(): void }> = [];
const captured: string[] = [];
function setup(): void {
previousHeadless = setTerminalHeadless(false);
Object.defineProperty(process.stdin, "isTTY", { value: true, configurable: true });
Object.defineProperty(process.stdout, "isTTY", { value: true, configurable: true });
Object.defineProperty(process.stdin, "setRawMode", { value: vi.fn(), configurable: true });
Object.defineProperty(process.stdout, "columns", { value: 100, configurable: true });
Object.defineProperty(process.stdout, "rows", { value: 30, configurable: true });
spies = [
vi.spyOn(process.stdin, "resume").mockImplementation(() => process.stdin),
vi.spyOn(process.stdin, "pause").mockImplementation(() => process.stdin),
vi.spyOn(process.stdin, "setEncoding").mockImplementation(() => process.stdin),
vi.spyOn(process.stdout, "write").mockImplementation(() => true),
vi.spyOn(process, "kill").mockImplementation(() => true),
];
captured.length = 0;
terminal = new ProcessTerminal();
terminal.start(
data => captured.push(data),
() => {},
);
}
afterEach(() => {
terminal?.stop();
terminal = undefined;
for (const spy of spies) spy.mockRestore();
spies = [];
restoreProperty(process.stdin, "isTTY", stdinIsTtyDescriptor);
restoreProperty(process.stdout, "isTTY", stdoutIsTtyDescriptor);
restoreProperty(process.stdin, "setRawMode", stdinSetRawModeDescriptor);
restoreProperty(process.stdout, "columns", stdoutColumnsDescriptor);
restoreProperty(process.stdout, "rows", stdoutRowsDescriptor);
setTerminalHeadless(previousHeadless);
});
it("swallows a single-event DA reply that arrives after the sentinel FIFO drains", () => {
setup();
// Complete `CSI ? … c` sequences flow through the StdinBuffer synchronously.
// Over-supply them: the first few resolve the startup probe sentinels, the
// rest model the slow SSH/PTY reply that lands with an empty FIFO. None may
// reach the composer.
for (let i = 0; i < 32; i++) process.stdin.emit("data", DA1_REPLY);
expect(captured.join("")).toBe("");
});
it("reassembles and swallows a split DA reply arriving with an empty FIFO", async () => {
setup();
// Drain the sentinel FIFO first (complete replies, processed synchronously).
for (let i = 0; i < 32; i++) process.stdin.emit("data", "\x1b[?62c");
captured.length = 0;
// The prefix of a slow reply arrives alone; the StdinBuffer holds it as an
// unambiguous private-CSI partial, then flushes it once its real timeout
// (<= PARTIAL_HOLD_MAX_MS = 150ms) elapses mid-sequence. This exercises the
// terminal-level reassembly path that only fires against the wall clock —
// deterministic fake timers cannot drive the StdinBuffer's internal flush
// here, so a genuine delay past the hold bound is required.
process.stdin.emit("data", "\x1b[?1;22;23");
await Bun.sleep(200);
// Tail bytes arrive as ordinary input after the flush.
process.stdin.emit("data", ";24;28;32;42;52c");
expect(captured.join("")).toBe("");
});
});
@@ -625,9 +625,12 @@ describe("ProcessTerminal OSC 11 appearance detection", () => {
process.stdin.emit("data", "\x1b[?1;2c");
expect(received).toEqual([]);
// An eighth stray DA1 has no owner and must reach the input handler — it is
// An eighth stray DA1 has no owner, yet is still swallowed: `CSI ? … c` is
// exclusively a terminal->host report, never a keystroke, so a reply that
// lands after the sentinel FIFO drains (slow SSH/PTY links) must not leak
// into the composer as literal text (#8542).
process.stdin.emit("data", "\x1b[?1;2c");
expect(received).toEqual(["\x1b[?1;2c"]);
expect(received).toEqual([]);
terminal.stop();
});
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-utils",
"version": "17.3.3",
"version": "17.3.4",
"description": "Shared utilities for pi packages",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-wire",
"version": "17.3.3",
"version": "17.3.4",
"description": "Shared wire protocol types for Oh My Pi packages",
"homepage": "https://omp.sh",
"author": "Can Boluk",
-4
View File
@@ -158,24 +158,20 @@ async function generateBundle(): Promise<void> {
if (isDryRun) {
console.log("DRY RUN bun run gen:stats");
console.log("DRY RUN bun --cwd=packages/collab-web run gen:tool-views");
console.log("DRY RUN bun run gen:mupdf");
return;
}
await runCommand(["bun", "run", "gen:stats"], repoRoot);
await runCommand(["bun", "--cwd=packages/collab-web", "run", "gen:tool-views"], repoRoot);
await runCommand(["bun", "run", "gen:mupdf"], repoRoot);
}
async function resetArtifacts(): Promise<void> {
if (isDryRun) {
console.log("DRY RUN bun run gen:native:reset");
console.log("DRY RUN bun run gen:stats:reset");
console.log("DRY RUN bun run gen:mupdf:reset");
return;
}
await runCommand(["bun", "run", "gen:native:reset"], repoRoot);
await runCommand(["bun", "run", "gen:stats:reset"], repoRoot);
await runCommand(["bun", "run", "gen:mupdf:reset"], repoRoot);
}
async function main(): Promise<void> {