feat: replaced custom mupdf wasm pipeline with native function

- Replaced the custom MuPDF-WASM PDF extraction and rendering pipeline with the new `pdfToMarkdown` native function from `@oh-my-pi/pi-natives`.
- Removed legacy MuPDF extraction modules, WASM embedding scripts, and PDF image extraction tools.
- Added OCR warnings and browser/text redirection for unsupported PDF image reads.
- Updated native package definitions, documentation, and test suites for the new PDF inspection capability.
This commit is contained in:
can1357
2026-08-14 14:08:28 +02:00
parent 1e1ee2c330
commit 04fab5ecb4
42 changed files with 1153 additions and 3620 deletions
+3 -9
View File
@@ -162,11 +162,11 @@ jobs:
bazelisk --bazelrc="${{ steps.cache.outputs.rc }}" test //crates/... bazelisk --bazelrc="${{ steps.cache.outputs.rc }}" test //crates/...
# Clippy scope mirrors `cargo clippy --workspace` (libraries only, no # Clippy scope mirrors `cargo clippy --workspace` (libraries only, no
# test targets) plus the strict/default split: crates with # test targets) plus the strict/default split: crates with
# `[lints] workspace = true` get the workspace policy, the vendored # `[lints] workspace = true` get the workspace policy, except
# brush-core fork is exempt (same as run-rs-task.ts's cargo excludes). # brush-core (a vendored fork excluded from the Cargo task too).
# pi-builtins allows every clippy group in its own manifest (ported # pi-builtins allows every clippy group in its own manifest (ported
# brush/uutils/jaq code) but is still held to zero rustc warnings; # brush/uutils/jaq code) but is still held to zero rustc warnings;
# cargo honors that via `[lints]`, bazel via the clippy-ported config. # Cargo honors that via `[lints]`, Bazel via the clippy-ported config.
- name: Clippy (workspace lint policy on opted-in crates) - name: Clippy (workspace lint policy on opted-in crates)
run: | run: |
bazelisk query "kind('rust_library|rust_shared_library', //crates/pi-ast/... + //crates/pi-iso/... + //crates/pi-natives/... + //crates/pi-shell/... + //crates/pi-voice/... + //crates/pi-walker/...)" \ bazelisk query "kind('rust_library|rust_shared_library', //crates/pi-ast/... + //crates/pi-iso/... + //crates/pi-natives/... + //crates/pi-shell/... + //crates/pi-voice/... + //crates/pi-walker/...)" \
@@ -411,12 +411,6 @@ jobs:
- name: Test coding-agent native/unit bucket - name: Test coding-agent native/unit bucket
env: env:
OMP_TEST_CONCURRENCY: "4" OMP_TEST_CONCURRENCY: "4"
# The mupdf/PDF-extraction chunk measures ~7 min on burstable
# runners under a full 8-wide fan-out; the default 600 s chunk
# watchdog SIGKILLed it (release run 30519992654). The watchdog
# exists to catch wedged children, not slow-but-progressing
# chunks — give this bucket a wider budget.
OMP_TEST_CHUNK_TIMEOUT: "1200"
run: bun run ci:test:coding-agent:native run: bun run ci:test:coding-agent:native
test_smoke: test_smoke:
-1
View File
@@ -64,7 +64,6 @@ pi-*.html
# Generated files # Generated files
packages/coding-agent/src/export/html/tool-views.generated.js packages/coding-agent/src/export/html/tool-views.generated.js
packages/natives/npm/ packages/natives/npm/
packages/coding-agent/src/utils/mupdf-wasm.wasm
/runs/ /runs/
python/omp-rpc/src/omp_rpc.egg-info/ python/omp-rpc/src/omp_rpc.egg-info/
python/omp-rpc/build/ python/omp-rpc/build/
Generated
+139
View File
@@ -1868,6 +1868,15 @@ version = "1.0.20"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555" checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
[[package]]
name = "ecb"
version = "0.1.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1a8bfa975b1aec2145850fcaa1c6fe269a16578c44705a532ae3edc92b8881c7"
dependencies = [
"cipher",
]
[[package]] [[package]]
name = "ecdsa" name = "ecdsa"
version = "0.16.9" version = "0.16.9"
@@ -1979,6 +1988,29 @@ dependencies = [
"syn 2.0.119", "syn 2.0.119",
] ]
[[package]]
name = "env_filter"
version = "2.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "900d271a03799a1ee8d1ca9b19893b48ca674a9284fefcfb85f05e74ed314217"
dependencies = [
"log",
"regex",
]
[[package]]
name = "env_logger"
version = "0.11.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "de671bd27a75a797dc9ae289ba1e77276e75e2026408aab65185384e2d5cd3f6"
dependencies = [
"anstream",
"anstyle",
"env_filter",
"jiff",
"log",
]
[[package]] [[package]]
name = "equivalent" name = "equivalent"
version = "1.0.2" version = "1.0.2"
@@ -3152,6 +3184,25 @@ dependencies = [
"quick-error", "quick-error",
] ]
[[package]]
name = "include_dir"
version = "0.7.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
dependencies = [
"include_dir_macros",
]
[[package]]
name = "include_dir_macros"
version = "0.7.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
dependencies = [
"proc-macro2",
"quote",
]
[[package]] [[package]]
name = "indenter" name = "indenter"
version = "0.3.4" version = "0.3.4"
@@ -3640,6 +3691,37 @@ version = "0.4.33"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad" checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
[[package]]
name = "lopdf"
version = "0.42.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
dependencies = [
"aes",
"bitflags 2.13.1",
"cbc",
"chrono",
"ecb",
"encoding_rs",
"flate2",
"getrandom 0.4.3",
"indexmap",
"itoa",
"jiff",
"log",
"md-5",
"nom 8.0.0",
"rand 0.10.2",
"rangemap",
"rayon",
"sha2",
"stringprep",
"thiserror 2.0.20",
"time",
"ttf-parser",
"weezl",
]
[[package]] [[package]]
name = "lscolors" name = "lscolors"
version = "0.21.0" version = "0.21.0"
@@ -4539,6 +4621,24 @@ dependencies = [
"pkg-config", "pkg-config",
] ]
[[package]]
name = "pdf-inspector"
version = "1.14.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e024ae242c514e2adf6aee186678e0eabdc2e5ecfbb2159186881b4498593cb"
dependencies = [
"env_logger",
"include_dir",
"log",
"lopdf",
"once_cell",
"rayon",
"regex",
"thiserror 2.0.20",
"ttf-parser",
"unicode-normalization",
]
[[package]] [[package]]
name = "peg" name = "peg"
version = "0.8.6" version = "0.8.6"
@@ -4982,6 +5082,7 @@ dependencies = [
"objc2-core-graphics", "objc2-core-graphics",
"objc2-foundation", "objc2-foundation",
"parking_lot", "parking_lot",
"pdf-inspector",
"phf 0.13.1", "phf 0.13.1",
"pi-ast", "pi-ast",
"pi-iso", "pi-iso",
@@ -5505,6 +5606,12 @@ dependencies = [
"rand_core 0.10.1", "rand_core 0.10.1",
] ]
[[package]]
name = "rangemap"
version = "1.8.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a611d15b50743feb4c76b7d03edcb0e64f399c26961e4efe6975bc398be6aa3d"
[[package]] [[package]]
name = "rayon" name = "rayon"
version = "1.12.0" version = "1.12.0"
@@ -6210,6 +6317,17 @@ dependencies = [
"quote", "quote",
] ]
[[package]]
name = "stringprep"
version = "0.1.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7b4df3d392d81bd458a8a621b8bffbd2302a12ffe288a9d931670948749463b1"
dependencies = [
"unicode-bidi",
"unicode-normalization",
"unicode-properties",
]
[[package]] [[package]]
name = "strsim" name = "strsim"
version = "0.11.1" version = "0.11.1"
@@ -7375,12 +7493,33 @@ version = "2.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dbc4bc3a9f746d862c45cb89d705aa10f187bb96c76001afab07a0d35ce60142" checksum = "dbc4bc3a9f746d862c45cb89d705aa10f187bb96c76001afab07a0d35ce60142"
[[package]]
name = "unicode-bidi"
version = "0.3.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5c1cb5db39152898a79168971543b1cb5020dff7fe43c8dc468b0885f5e29df5"
[[package]] [[package]]
name = "unicode-ident" name = "unicode-ident"
version = "1.0.24" version = "1.0.24"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75" checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
[[package]]
name = "unicode-normalization"
version = "0.1.25"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8"
dependencies = [
"tinyvec",
]
[[package]]
name = "unicode-properties"
version = "0.1.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7df058c713841ad818f1dc5d3fd88063241cc61f49f5fbea4b951e8cf5a8d71d"
[[package]] [[package]]
name = "unicode-segmentation" name = "unicode-segmentation"
version = "1.13.3" version = "1.13.3"
+1
View File
@@ -229,6 +229,7 @@ clap = { version = "4", features = ["derive"] }
# ────────────────────────────────────────────────────────────────────────────── # ──────────────────────────────────────────────────────────────────────────────
# Text Processing & Parsing # Text Processing & Parsing
# ────────────────────────────────────────────────────────────────────────────── # ──────────────────────────────────────────────────────────────────────────────
pdf-inspector = "1"
regex = "1" regex = "1"
similar = "3.1.0" similar = "3.1.0"
unicode-segmentation = "1.13" unicode-segmentation = "1.13"
+596 -425
View File
File diff suppressed because one or more lines are too long
-4
View File
@@ -105,7 +105,6 @@
"@opentelemetry/sdk-metrics": "catalog:", "@opentelemetry/sdk-metrics": "catalog:",
"@opentelemetry/sdk-trace-base": "catalog:", "@opentelemetry/sdk-trace-base": "catalog:",
"@opentelemetry/sdk-trace-node": "catalog:", "@opentelemetry/sdk-trace-node": "catalog:",
"mupdf": "catalog:",
"puppeteer-core": "catalog:", "puppeteer-core": "catalog:",
}, },
"devDependencies": { "devDependencies": {
@@ -386,7 +385,6 @@
"ghostty-web": "^0.4.0", "ghostty-web": "^0.4.0",
"lint-staged": "^17.0.8", "lint-staged": "^17.0.8",
"lucide-react": "^1.24.0", "lucide-react": "^1.24.0",
"mupdf": "^1.28.0",
"onnxruntime-node": "1.26.0", "onnxruntime-node": "1.26.0",
"postcss": "^8.5.16", "postcss": "^8.5.16",
"prettier": "^3.9.5", "prettier": "^3.9.5",
@@ -1219,8 +1217,6 @@
"ms": ["ms@2.1.3", "", {}, "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="], "ms": ["ms@2.1.3", "", {}, "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="],
"mupdf": ["mupdf@1.28.0", "", {}, "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg=="],
"mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="], "mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="],
"nanoid": ["nanoid@3.3.18", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w=="], "nanoid": ["nanoid@3.3.18", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w=="],
+1
View File
@@ -44,6 +44,7 @@ inferno.workspace = true
napi.workspace = true napi.workspace = true
napi-derive.workspace = true napi-derive.workspace = true
parking_lot.workspace = true parking_lot.workspace = true
pdf-inspector.workspace = true
phf.workspace = true phf.workspace = true
flume.workspace = true flume.workspace = true
pi-ast.workspace = true pi-ast.workspace = true
+4 -2
View File
@@ -2,7 +2,7 @@
//! //!
//! # Overview //! # Overview
//! High-performance primitives for clipboard access, grep, file discovery, //! High-performance primitives for clipboard access, grep, file discovery,
//! ANSI-aware text measurement, syntax highlighting, HTML-to-Markdown //! ANSI-aware text measurement, syntax highlighting, HTML/PDF-to-Markdown
//! conversion, and terminal SIXEL encoding. //! conversion, and terminal SIXEL encoding.
//! //!
//! # Example //! # Example
@@ -15,7 +15,7 @@
//! //!
//! # Architecture //! # Architecture
//! ```text //! ```text
//! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/highlight/sixel/text) //! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/pdf/highlight/sixel/text)
//! ``` //! ```
#![allow(clippy::trailing_empty_array, reason = "generated by napi macro")] #![allow(clippy::trailing_empty_array, reason = "generated by napi macro")]
@@ -41,6 +41,8 @@ pub mod html;
pub mod iofs; pub mod iofs;
pub mod keys; pub mod keys;
pub mod live; pub mod live;
/// PDF inspection and Markdown conversion.
pub mod pdf;
pub mod sixel; pub mod sixel;
pub mod snapcompact; pub mod snapcompact;
pub use pi_ast::language; pub use pi_ast::language;
+165
View File
@@ -0,0 +1,165 @@
//! PDF inspection and Markdown conversion backed by `pdf-inspector`.
use napi::{Result, bindgen_prelude::Uint8Array};
use napi_derive::napi;
use pdf_inspector::{MarkdownOptions, PdfOptions, process_pdf_mem_with_options};
use crate::task;
/// Markdown and inspection metadata produced from a PDF document.
#[napi(object)]
pub struct PdfMarkdownResult {
/// Extracted document content in Markdown format.
pub markdown: String,
/// Document title from PDF metadata, when present.
pub title: Option<String>,
/// Total number of pages in the document.
pub page_count: u32,
/// One-indexed page numbers whose content requires OCR.
pub pages_needing_ocr: Vec<u32>,
/// Whether the document contains text encoding problems.
pub has_encoding_issues: bool,
}
/// Convert an in-memory PDF to Markdown and return its inspection metadata.
///
/// Conversion copies the typed array before dispatch so JavaScript mutation
/// cannot race the native worker.
///
/// # Errors
/// Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
/// be parsed or converted.
#[napi(js_name = "pdfToMarkdown")]
pub fn pdf_to_markdown(input: Uint8Array) -> task::Promise<PdfMarkdownResult> {
let input = input.to_vec();
task::blocking("pdf.to_markdown", (), move |_| convert_pdf(&input))
}
fn convert_pdf(input: &[u8]) -> Result<PdfMarkdownResult> {
let options = PdfOptions::new()
.markdown(MarkdownOptions { include_page_numbers: true, ..Default::default() });
let converted = process_pdf_mem_with_options(input, options)
.map_err(|error| napi::Error::from_reason(format!("PDF conversion failed: {error}")))?;
let markdown = match converted.markdown {
Some(markdown) => markdown,
None if !converted.pages_needing_ocr.is_empty() => String::new(),
None => {
return Err(napi::Error::from_reason(
"PDF conversion failed: converter returned no Markdown",
));
},
};
Ok(PdfMarkdownResult {
markdown,
title: converted.title,
page_count: converted.page_count,
pages_needing_ocr: converted.pages_needing_ocr,
has_encoding_issues: converted.has_encoding_issues,
})
}
#[cfg(test)]
mod tests {
use super::*;
fn pdf_fixture(page_contents: &[&str], title: Option<&str>) -> Vec<u8> {
let font_id = 3 + page_contents.len() * 2;
let info_id = title.map(|_| font_id + 1);
let mut objects = Vec::with_capacity(font_id + usize::from(info_id.is_some()));
objects.push("<< /Type /Catalog /Pages 2 0 R >>".to_string());
let kids = (0..page_contents.len())
.map(|index| format!("{} 0 R", 3 + index * 2))
.collect::<Vec<_>>()
.join(" ");
objects.push(format!("<< /Type /Pages /Kids [{kids}] /Count {} >>", page_contents.len()));
for (index, content) in page_contents.iter().enumerate() {
let content_id = 4 + index * 2;
objects.push(format!(
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 \
{font_id} 0 R >> >> /Contents {content_id} 0 R >>"
));
objects.push(format!("<< /Length {} >>\nstream\n{content}\nendstream", content.len()));
}
objects.push(
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>"
.to_string(),
);
if let Some(title) = title {
objects.push(format!("<< /Title ({title}) >>"));
}
let mut pdf = b"%PDF-1.4\n".to_vec();
let mut offsets = Vec::with_capacity(objects.len());
for (index, object) in objects.iter().enumerate() {
offsets.push(pdf.len());
pdf.extend_from_slice(format!("{} 0 obj\n{object}\nendobj\n", index + 1).as_bytes());
}
let xref_offset = pdf.len();
pdf.extend_from_slice(
format!("xref\n0 {}\n0000000000 65535 f \n", objects.len() + 1).as_bytes(),
);
for offset in offsets {
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
}
let info = info_id.map_or_else(String::new, |id| format!(" /Info {id} 0 R"));
pdf.extend_from_slice(
format!(
"trailer\n<< /Size {} /Root 1 0 R{info} >>\nstartxref\n{xref_offset}\n%%EOF\n",
objects.len() + 1
)
.as_bytes(),
);
pdf
}
#[test]
fn converts_text_title_and_page_markers() {
let pdf = pdf_fixture(
&[
"BT /F1 12 Tf 72 720 Td (First page text) Tj 0 -18 Td (More first page text) Tj 0 -18 \
Td (End first page) Tj ET",
"BT /F1 12 Tf 72 720 Td (Second page text) Tj 0 -18 Td (More second page text) Tj 0 \
-18 Td (End second page) Tj ET",
],
Some("Fixture Title"),
);
let result = convert_pdf(&pdf).expect("fixture PDF should convert");
assert_eq!(result.title.as_deref(), Some("Fixture Title"));
assert_eq!(result.page_count, 2);
assert!(result.markdown.contains("First page text"), "{}", result.markdown);
assert!(result.markdown.contains("Second page text"), "{}", result.markdown);
assert!(result.markdown.contains("<!-- Page 1 -->"), "{}", result.markdown);
assert!(result.markdown.contains("<!-- Page 2 -->"), "{}", result.markdown);
assert!(!result.has_encoding_issues);
}
#[test]
fn reports_empty_pages_as_needing_ocr() {
let pdf = pdf_fixture(&[""], None);
let result = convert_pdf(&pdf).expect("empty-page PDF should still convert");
assert_eq!(result.page_count, 1);
assert_eq!(result.pages_needing_ocr, vec![1]);
}
#[test]
fn prefixes_malformed_pdf_errors() {
let error = convert_pdf(b"not a PDF")
.err()
.expect("malformed input should fail");
assert!(
error.reason.starts_with("PDF conversion failed:"),
"unexpected error: {}",
error.reason
);
}
}
-4
View File
@@ -1738,10 +1738,6 @@
url = "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz"; url = "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz";
hash = "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="; hash = "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==";
}; };
"mupdf@1.28.0" = fetchurl {
url = "https://registry.npmjs.org/mupdf/-/mupdf-1.28.0.tgz";
hash = "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg==";
};
"mute-stream@3.0.0" = fetchurl { "mute-stream@3.0.0" = fetchurl {
url = "https://registry.npmjs.org/mute-stream/-/mute-stream-3.0.0.tgz"; url = "https://registry.npmjs.org/mute-stream/-/mute-stream-3.0.0.tgz";
hash = "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="; hash = "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw==";
-3
View File
@@ -61,7 +61,6 @@
"ghostty-web": "^0.4.0", "ghostty-web": "^0.4.0",
"lint-staged": "^17.0.8", "lint-staged": "^17.0.8",
"lucide-react": "^1.24.0", "lucide-react": "^1.24.0",
"mupdf": "^1.28.0",
"onnxruntime-node": "1.26.0", "onnxruntime-node": "1.26.0",
"postcss": "^8.5.16", "postcss": "^8.5.16",
"prettier": "^3.9.5", "prettier": "^3.9.5",
@@ -170,8 +169,6 @@
"gen:nix": "bun scripts/gen-nix-bun.ts", "gen:nix": "bun scripts/gen-nix-bun.ts",
"gen:tool-views": "bun --cwd=packages/collab-web run gen:tool-views", "gen:tool-views": "bun --cwd=packages/collab-web run gen:tool-views",
"gen:bundle": "bun --cwd=packages/coding-agent run gen:bundle", "gen:bundle": "bun --cwd=packages/coding-agent run gen:bundle",
"gen:mupdf": "bun --cwd=packages/coding-agent run gen:mupdf",
"gen:mupdf:reset": "bun --cwd=packages/coding-agent run gen:mupdf:reset",
"gen:native": "bun --cwd=packages/natives run gen:native", "gen:native": "bun --cwd=packages/natives run gen:native",
"gen:native:reset": "bun --cwd=packages/natives run gen:native:reset", "gen:native:reset": "bun --cwd=packages/natives run gen:native:reset",
"check-spoofed-versions": "bun scripts/check-spoofed-versions.ts" "check-spoofed-versions": "bun scripts/check-spoofed-versions.ts"
+5
View File
@@ -2,6 +2,11 @@
## [Unreleased] ## [Unreleased]
### Changed
- Replaced the MuPDF-WASM PDF document backend with `pdf-inspector` through `@oh-my-pi/pi-natives`, preserving cached text conversion and PDF line selectors while reporting pages that need OCR.
- Removed `read <pdf>:` image listings and `read <pdf>:<image>.png` extraction because `pdf-inspector` does not rasterize pages; these reads now direct users to the Puppeteer browser tool for rendering or to read the PDF path for extracted text.
## [17.3.3] - 2026-08-14 ## [17.3.3] - 2026-08-14
### Fixed ### Fixed
-3
View File
@@ -41,8 +41,6 @@
"format-prompts": "bun scripts/format-prompts.ts", "format-prompts": "bun scripts/format-prompts.ts",
"gen:tool-views": "bun --cwd=../collab-web run gen:tool-views", "gen:tool-views": "bun --cwd=../collab-web run gen:tool-views",
"gen:bundle": "bun scripts/bundle-dist.ts", "gen:bundle": "bun scripts/bundle-dist.ts",
"gen:mupdf": "bun scripts/embed-mupdf-wasm.ts --generate",
"gen:mupdf:reset": "bun scripts/embed-mupdf-wasm.ts --reset",
"gen:native": "bun --cwd=../natives run gen:native", "gen:native": "bun --cwd=../natives run gen:native",
"gen:native:reset": "bun --cwd=../natives run gen:native:reset", "gen:native:reset": "bun --cwd=../natives run gen:native:reset",
"prepack": "bun run gen:tool-views && bun run gen:bundle", "prepack": "bun run gen:tool-views && bun run gen:bundle",
@@ -73,7 +71,6 @@
"@opentelemetry/sdk-metrics": "catalog:", "@opentelemetry/sdk-metrics": "catalog:",
"@opentelemetry/sdk-trace-base": "catalog:", "@opentelemetry/sdk-trace-base": "catalog:",
"@opentelemetry/sdk-trace-node": "catalog:", "@opentelemetry/sdk-trace-node": "catalog:",
"mupdf": "catalog:",
"puppeteer-core": "catalog:" "puppeteer-core": "catalog:"
}, },
"optionalDependencies": { "optionalDependencies": {
@@ -88,7 +88,6 @@ async function main(): Promise<void> {
["bun", "--cwd=../natives", "run", "gen:native"], ["bun", "--cwd=../natives", "run", "gen:native"],
crossBuild ? { ...Bun.env, TARGET_PLATFORM: crossBuild.platform, TARGET_ARCH: crossBuild.arch } : Bun.env, crossBuild ? { ...Bun.env, TARGET_PLATFORM: crossBuild.platform, TARGET_ARCH: crossBuild.arch } : Bun.env,
); );
await runCommand(["bun", "run", "gen:mupdf"]);
try { try {
await compileCodingAgent({ await compileCodingAgent({
repoRoot, repoRoot,
@@ -104,7 +103,6 @@ async function main(): Promise<void> {
await runCommand(["codesign", "--force", "--sign", "-", outputPath]); await runCommand(["codesign", "--force", "--sign", "-", outputPath]);
} }
} finally { } finally {
await runCommand(["bun", "run", "gen:mupdf:reset"]);
await runCommand(["bun", "--cwd=../natives", "run", "gen:native:reset"]); await runCommand(["bun", "--cwd=../natives", "run", "gen:native:reset"]);
} }
} finally { } finally {
@@ -15,7 +15,6 @@ const legacyHtmlExportAssetPattern = /^(?:template-[^.]+\.(?:css|html|js)|tool-v
// `omp-legacy-pi-modules` exists only in compiled binaries via the build plugin; // `omp-legacy-pi-modules` exists only in compiled binaries via the build plugin;
// the npm bundle never executes that `isCompiledBinary()` branch. // the npm bundle never executes that `isCompiledBinary()` branch.
const ALWAYS_EXTERNAL = [ const ALWAYS_EXTERNAL = [
"mupdf",
"@oh-my-pi/pi-natives", "@oh-my-pi/pi-natives",
"@huggingface/transformers", "@huggingface/transformers",
"fastembed", "fastembed",
@@ -1,67 +0,0 @@
#!/usr/bin/env bun
// Embeds mupdf's `mupdf-wasm.wasm` into the compiled single-file binary.
//
// mupdf loads its wasm by reading the `mupdf-wasm.wasm` sibling of its own
// module via `new URL(..., import.meta.url)` + `readFileSync`. A `bun --compile`
// binary has no node_modules, so that read fails (`ENOENT .../mupdf-wasm.wasm`),
// and marking mupdf `--external` instead makes `bun --compile` eagerly fail to
// resolve the package at startup (the static `import * as mupdf` lives in a lazy
// chunk but is hoisted). So the binary build bundles mupdf and embeds the wasm
// bytes here, handing them to the WASM module as `$libmupdf_wasm_Module.wasmBinary`
// (see src/utils/markit.ts).
//
// `--generate` copies the wasm next to src/utils/mupdf-wasm-embed.ts and rewrites
// that module to import it via `with { type: "file" }`; `--reset` restores the
// checked-in placeholder and removes the copy. The npm `dist/cli.js` bundle never
// runs this — it keeps mupdf external and loads the wasm from node_modules.
import * as fs from "node:fs/promises";
import { createRequire } from "node:module";
import * as path from "node:path";
const utilsDir = path.join(import.meta.dir, "..", "src", "utils");
const helperPath = path.join(utilsDir, "mupdf-wasm-embed.ts");
const wasmCopyPath = path.join(utilsDir, "mupdf-wasm.wasm");
const placeholder = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
//
// Compiled single-file binaries cannot let mupdf resolve its \`mupdf-wasm.wasm\`
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
// wasm bytes via \`with { type: "file" }\` and copies the wasm next to it. Source
// checkouts, \`bun test\`, and the npm \`dist/cli.js\` bundle keep mupdf external and
// load the wasm from node_modules, so this placeholder returns undefined and the
// build resets back to it afterward.
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
\treturn undefined;
}
`;
const generated = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit or commit.
import { readFileSync } from "node:fs";
import wasmPath from "./mupdf-wasm.wasm" with { type: "file" };
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
\treturn readFileSync(wasmPath);
}
`;
if (process.argv.includes("--reset")) {
await Bun.write(helperPath, placeholder);
try {
await fs.unlink(wasmCopyPath);
} catch (err) {
if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err;
}
process.exit(0);
}
const wasmSource = path.join(path.dirname(createRequire(import.meta.url).resolve("mupdf")), "mupdf-wasm.wasm");
const wasmFile = Bun.file(wasmSource);
if (!(await wasmFile.exists())) {
throw new Error(`mupdf wasm not found at ${wasmSource}; run \`bun install\` first.`);
}
await Bun.write(wasmCopyPath, wasmFile);
await Bun.write(helperPath, generated);
console.log(`Embedded mupdf wasm (${wasmFile.size} bytes) into ${path.relative(process.cwd(), wasmCopyPath)}`);
+8 -8
View File
@@ -1,15 +1,15 @@
This directory contains an in-house document-to-markdown engine adapted from Portions of this in-house document-to-markdown engine are adapted from
markit-ai (https://github.com/Michaelliv/markit), used under the MIT License. markit-ai (https://github.com/Michaelliv/markit), used under the MIT License.
This attribution covers the shared registry/types and the DOCX, PPTX, XLSX,
and EPUB converters. The PDF converter is implemented separately and is not
derived from markit-ai.
Copyright (c) 2026 Michael Liv Copyright (c) 2026 Michael Liv
Only the converters for the document formats omp supports are ported (pdf, The CLI, plugin/provider, and unused converters (html, image, audio,
docx, pptx, xlsx, epub); the CLI, plugin/provider, and unused converters plain-text, rss, github, wikipedia, csv, json, yaml, ipynb, iwork, zip, xml)
(html, image, audio, plain-text, rss, github, wikipedia, csv, json, yaml, were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and `.rtf` have no converter
ipynb, iwork, zip, xml) were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and and surface a conversion error.
`.rtf` are routed by the read/fetch tools but have no converter — they surface
a conversion error, exactly as upstream markit did. Logic is ported faithfully
so conversion output matches the upstream package.
MIT License MIT License
@@ -1,103 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Multi-column layout detection and text box reordering.
*
* Many PDFs (legal documents, datasheets, academic papers) use two-column
* layouts. Without column detection, text boxes are ordered by Y position
* only, interleaving left and right column content.
*
* Algorithm:
* 1. Collect left edges of all text boxes on the page
* 2. Find the largest horizontal gap between consecutive left edges
* 3. If gap > MIN_GAP_RATIO of the text width and both sides have
* enough boxes → multi-column detected
* 4. Assign each text box to a column based on its center X
* 5. Return columns in reading order (left-to-right, top-to-bottom)
*
* This only detects the column structure. The caller is responsible for
* processing each column's text boxes independently (table detection,
* rendering, etc.).
*/
import type { TextBox } from "./types";
export interface ColumnLayout {
/** Number of columns detected (1 = single column, 2+ = multi-column). */
columnCount: number;
/** Text boxes grouped by column, in reading order (left to right). */
columns: TextBox[][];
/** X positions of column boundaries (between columns). */
boundaries: number[];
}
/**
* Minimum gap as a fraction of the total text width to consider a column
* boundary. A two-column layout typically has ~50% gap; we use a lower
* threshold to catch asymmetric columns.
*/
const MIN_GAP_RATIO = 0.15;
/** Minimum number of text boxes on each side of the gap. */
const MIN_BOXES_PER_COLUMN = 4;
/** Minimum gap in absolute points to avoid splitting on small whitespace. */
const MIN_GAP_PTS = 40;
/**
* Detect column layout and return text boxes grouped by column.
*
* For single-column pages, returns all boxes in one group.
* For multi-column pages, returns boxes split by column in reading order.
*/
export function detectColumns(textBoxes: TextBox[]): ColumnLayout {
if (textBoxes.length < MIN_BOXES_PER_COLUMN * 2) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Collect unique left edges (rounded to avoid float noise)
const lefts = [...new Set(textBoxes.map(tb => Math.round(tb.bounds.left)))].sort((a, b) => a - b);
if (lefts.length < 2) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
const textXMin = lefts[0];
const textXMax = Math.max(...textBoxes.map(tb => Math.round(tb.bounds.right)));
const textWidth = textXMax - textXMin;
if (textWidth <= 0) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Find the largest gap between consecutive left-edge positions
let maxGap = 0;
let gapLeft = 0;
let gapRight = 0;
for (let i = 1; i < lefts.length; i++) {
const gap = lefts[i] - lefts[i - 1];
if (gap > maxGap) {
maxGap = gap;
gapLeft = lefts[i - 1];
gapRight = lefts[i];
}
}
const gapRatio = maxGap / textWidth;
if (gapRatio < MIN_GAP_RATIO || maxGap < MIN_GAP_PTS) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Split point is the midpoint of the gap
const splitX = (gapLeft + gapRight) / 2;
// Assign boxes to columns based on center X
const leftCol: TextBox[] = [];
const rightCol: TextBox[] = [];
for (const tb of textBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
if (cx < splitX) {
leftCol.push(tb);
} else {
rightCol.push(tb);
}
}
// Validate both columns have enough content
if (leftCol.length < MIN_BOXES_PER_COLUMN || rightCol.length < MIN_BOXES_PER_COLUMN) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
return {
columnCount: 2,
columns: [leftCol, rightCol],
boundaries: [splitX],
};
}
@@ -1,598 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* PDF content extraction using mupdf.
*
* Extracts text boxes (with position, font size, bold) and vector line
* segments (table borders) from each page. Uses mupdf's native WASM
* engine for fast parsing, and reads raw content streams for vector graphics.
*
* Coordinate system: PDF native (origin = bottom-left, Y increases upward).
*/
import type * as mupdf from "mupdf";
import type { ImageRegion, PageContent, Segment, TextBox } from "./types";
// mupdf instantiates its WASM module via a top-level await. A static
// `import * as mupdf` would pull that await into this module's init, which makes
// the whole bundled markit chunk's `__esm` init async — and bun's compiled
// bundler fails to await that init transitively through the `../markit` barrel,
// exposing the converter classes before their module-level consts initialize
// (e.g. `EXTENSIONS` reads as undefined). Importing mupdf lazily keeps the chunk
// init synchronous and also keeps the ~10MB wasm off non-PDF conversions.
let mupdfModule: typeof mupdf | undefined;
async function loadMupdf(): Promise<typeof mupdf> {
if (!mupdfModule) {
mupdfModule = await import("mupdf");
}
return mupdfModule;
}
/** mupdf structured-text JSON bounding box (top-left origin). */
interface StextBBox {
x: number;
y: number;
w: number;
h: number;
}
/** Font metadata attached to a structured-text line. */
interface StextFont {
size?: number;
weight?: string;
name?: string;
}
/** A line within a text block in mupdf structured-text JSON. */
interface StextLine {
text?: string;
font?: StextFont;
bbox: StextBBox;
}
/** A block (text or image) in mupdf structured-text JSON. */
interface StextBlock {
type: string;
bbox: StextBBox;
lines: StextLine[];
}
/** Parsed mupdf structured-text JSON for a page. */
interface StructuredTextJSON {
blocks: StextBlock[];
}
/** A raw text fragment before merging into word/phrase boxes. */
interface RawTextItem {
text: string;
x: number;
y: number;
width: number;
height: number;
fontSize: number;
isBold: boolean;
}
// ---------------------------------------------------------------------------
// Text extraction
// ---------------------------------------------------------------------------
/** Y tolerance for merging text fragments on the same visual line. */
const SAME_LINE_Y_TOLERANCE = 2;
/** Max horizontal gap (pts) to merge adjacent fragments into one text box. */
const MAX_MERGE_GAP = 14;
/**
* Merge horizontally adjacent raw text items on the same visual line into
* word/phrase-level text boxes.
*/
function mergeIntoWords(raws: RawTextItem[]): RawTextItem[] {
if (raws.length === 0) return [];
// Sort by Y descending (top-first in bottom-left coords), then X ascending
const sorted = [...raws].sort((a, b) => {
const dy = b.y - a.y;
return Math.abs(dy) > SAME_LINE_Y_TOLERANCE ? dy : a.x - b.x;
});
const merged: RawTextItem[] = [];
let cur = { ...sorted[0] };
for (let i = 1; i < sorted.length; i++) {
const next = sorted[i];
const sameY = Math.abs(next.y - cur.y) <= SAME_LINE_Y_TOLERANCE;
const close = next.x <= cur.x + cur.width + MAX_MERGE_GAP;
if (sameY && close) {
const gap = next.x - (cur.x + cur.width);
const sep = gap > 1 ? " " : "";
cur.text += sep + next.text;
cur.width = next.x + next.width - cur.x;
cur.height = Math.max(cur.height, next.height);
cur.fontSize = Math.max(cur.fontSize, next.fontSize);
cur.isBold = cur.isBold || next.isBold;
} else {
merged.push(cur);
cur = { ...next };
}
}
merged.push(cur);
return merged;
}
/**
* Extract text boxes from a mupdf page using structured text output.
*
* mupdf's structured text JSON uses top-left origin; we convert to
* bottom-left (standard PDF coordinates) using the page height.
*/
function extractTextBoxes(
page: mupdf.Page,
pageNumber: number,
pageHeight: number,
stext?: StructuredTextJSON,
): TextBox[] {
if (!stext) {
stext = JSON.parse(page.toStructuredText("preserve-whitespace").asJSON()) as StructuredTextJSON;
}
const raws: RawTextItem[] = [];
for (const block of stext.blocks) {
if (block.type !== "text") continue;
for (const line of block.lines) {
const text = line.text?.trim();
if (!text) continue;
const fontSize = line.font?.size ?? 0;
const weight = line.font?.weight ?? "normal";
const fontName = line.font?.name ?? "";
const isBold = weight === "bold" || /bold/i.test(fontName) || /Black|Heavy/i.test(fontName);
// mupdf bbox: {x, y, w, h} in top-left coords
// Convert to bottom-left: pdfY = pageHeight - (bbox.y + bbox.h)
const bboxY = line.bbox.y;
const bboxH = line.bbox.h;
const pdfY = pageHeight - (bboxY + bboxH);
raws.push({
text,
x: line.bbox.x,
y: pdfY,
width: line.bbox.w,
height: bboxH,
fontSize,
isBold,
});
}
}
const words = mergeIntoWords(raws);
return words
.map((w, i) => ({
id: `p${pageNumber}-t${i}`,
text: w.text.trim(),
pageNumber,
fontSize: w.fontSize,
isBold: w.isBold,
bounds: {
left: w.x,
right: w.x + w.width,
bottom: w.y,
top: w.y + w.height,
},
}))
.filter(b => b.text.length > 0);
}
// ---------------------------------------------------------------------------
// Vector segment extraction from raw content stream
// ---------------------------------------------------------------------------
/** Minimum aspect ratio for a filled rect to be considered a line. */
const LINE_ASPECT_THRESHOLD = 6;
/** Minimum length (pts) for a segment to count. */
const MIN_LENGTH = 2;
/** Maximum thickness (pts) for a border line (filters out filled areas). */
const MAX_THICKNESS = 3;
/**
* Convert a thin filled rectangle to a horizontal or vertical segment.
* Returns null if the rect doesn't look like a border line.
*/
function thinRectToSegment(id: string, x: number, y: number, w: number, h: number): Segment | null {
const aw = Math.abs(w);
const ah = Math.abs(h);
if (aw > ah * LINE_ASPECT_THRESHOLD && aw >= MIN_LENGTH && ah <= MAX_THICKNESS) {
// Horizontal line
const cy = y + ah / 2;
return { id, x1: x, y1: cy, x2: x + aw, y2: cy };
}
if (ah > aw * LINE_ASPECT_THRESHOLD && ah >= MIN_LENGTH && aw <= MAX_THICKNESS) {
// Vertical line
const cx = x + aw / 2;
return { id, x1: cx, y1: y, x2: cx, y2: y + ah };
}
return null;
}
/**
* Emit 4 edge segments from a stroked rectangle.
*/
function pushStrokedRectEdges(segments: Segment[], id: string, x: number, y: number, w: number, h: number): void {
const aw = Math.abs(w);
const ah = Math.abs(h);
const base = id;
if (aw >= MIN_LENGTH) {
segments.push({ id: `${base}-b`, x1: x, y1: y, x2: x + aw, y2: y });
segments.push({
id: `${base}-t`,
x1: x,
y1: y + ah,
x2: x + aw,
y2: y + ah,
});
}
if (ah >= MIN_LENGTH) {
segments.push({ id: `${base}-l`, x1: x, y1: y, x2: x, y2: y + ah });
segments.push({
id: `${base}-r`,
x1: x + aw,
y1: y,
x2: x + aw,
y2: y + ah,
});
}
}
const CTM_IDENTITY = [1, 0, 0, 1, 0, 0];
/** Concatenate two affine matrices: result = parent × child. */
function ctmConcat(p: number[], c: number[]): number[] {
return [
p[0] * c[0] + p[2] * c[1],
p[1] * c[0] + p[3] * c[1],
p[0] * c[2] + p[2] * c[3],
p[1] * c[2] + p[3] * c[3],
p[0] * c[4] + p[2] * c[5] + p[4],
p[1] * c[4] + p[3] * c[5] + p[5],
];
}
function ctmApply(m: number[], x: number, y: number): [number, number] {
return [m[0] * x + m[2] * y + m[4], m[1] * x + m[3] * y + m[5]];
}
// ---------------------------------------------------------------------------
// Content stream parsing
// ---------------------------------------------------------------------------
/**
* Parse a PDF content stream and extract line segments from thin filled
* rectangles (re+f), stroked rectangles (re+S), and explicit lines (m/l+S).
* Tracks the CTM via q/Q/cm operators so coordinates are in page space.
*/
function extractSegmentsFromContentStream(raw: string, pageNumber: number): Segment[] {
const segments: Segment[] = [];
const tokens = tokenizeContentStream(raw);
let idx = 0;
let strokeWidth = 1.0;
// Graphics state stack (q/Q): saves CTM + strokeWidth
let ctm = [...CTM_IDENTITY];
const stateStack: Array<{ ctm: number[]; strokeWidth: number }> = [];
// State for path building (in user coordinates, pre-CTM)
let curX = 0;
let curY = 0;
let pathStartX = 0;
let pathStartY = 0;
const pendingRects: Array<{ x: number; y: number; w: number; h: number }> = [];
const pendingLines: Array<{ x1: number; y1: number; x2: number; y2: number }> = [];
function flushPath(mode: "fill" | "stroke"): void {
const sid = () => `p${pageNumber}-s${segments.length}`;
if (mode === "fill") {
for (const r of pendingRects) {
// Transform the rect corners through CTM, then check if it's a thin line
const [x0, y0] = ctmApply(ctm, r.x, r.y);
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
const seg = thinRectToSegment(
sid(),
Math.min(x0, x1),
Math.min(y0, y1),
Math.abs(x1 - x0),
Math.abs(y1 - y0),
);
if (seg) segments.push(seg);
}
} else if (mode === "stroke" && strokeWidth <= MAX_THICKNESS) {
for (const r of pendingRects) {
const [x0, y0] = ctmApply(ctm, r.x, r.y);
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
pushStrokedRectEdges(
segments,
sid(),
Math.min(x0, x1),
Math.min(y0, y1),
Math.abs(x1 - x0),
Math.abs(y1 - y0),
);
}
for (const l of pendingLines) {
const [lx1, ly1] = ctmApply(ctm, l.x1, l.y1);
const [lx2, ly2] = ctmApply(ctm, l.x2, l.y2);
const dx = Math.abs(lx2 - lx1);
const dy = Math.abs(ly2 - ly1);
// Only keep H/V lines
if ((dx >= MIN_LENGTH && dy < 1) || (dy >= MIN_LENGTH && dx < 1)) {
segments.push({ id: sid(), x1: lx1, y1: ly1, x2: lx2, y2: ly2 });
}
}
}
pendingRects.length = 0;
pendingLines.length = 0;
}
while (idx < tokens.length) {
const t = tokens[idx];
if (t === "q") {
stateStack.push({ ctm: [...ctm], strokeWidth });
} else if (t === "Q") {
const saved = stateStack.pop();
if (saved) {
ctm = saved.ctm;
strokeWidth = saved.strokeWidth;
}
} else if (t === "cm" && idx >= 6) {
const a = Number(tokens[idx - 6]);
const b = Number(tokens[idx - 5]);
const c = Number(tokens[idx - 4]);
const d = Number(tokens[idx - 3]);
const e = Number(tokens[idx - 2]);
const f = Number(tokens[idx - 1]);
ctm = ctmConcat(ctm, [a, b, c, d, e, f]);
} else if (t === "w" && idx >= 1) {
strokeWidth = Number(tokens[idx - 1]) || strokeWidth;
} else if (t === "re" && idx >= 4) {
const x = Number(tokens[idx - 4]);
const y = Number(tokens[idx - 3]);
const w = Number(tokens[idx - 2]);
const h = Number(tokens[idx - 1]);
if (Number.isFinite(x + y + w + h)) {
pendingRects.push({ x, y, w, h });
}
} else if (t === "m" && idx >= 2) {
curX = Number(tokens[idx - 2]);
curY = Number(tokens[idx - 1]);
pathStartX = curX;
pathStartY = curY;
} else if (t === "l" && idx >= 2) {
const x2 = Number(tokens[idx - 2]);
const y2 = Number(tokens[idx - 1]);
pendingLines.push({ x1: curX, y1: curY, x2, y2 });
curX = x2;
curY = y2;
} else if (t === "h") {
// closePath: line back to start
if (curX !== pathStartX || curY !== pathStartY) {
pendingLines.push({
x1: curX,
y1: curY,
x2: pathStartX,
y2: pathStartY,
});
}
curX = pathStartX;
curY = pathStartY;
} else if (t === "f" || t === "F" || t === "f*") {
flushPath("fill");
} else if (t === "S" || t === "s") {
if (t === "s") {
// closeStroke: implicit closePath
if (curX !== pathStartX || curY !== pathStartY) {
pendingLines.push({
x1: curX,
y1: curY,
x2: pathStartX,
y2: pathStartY,
});
}
}
flushPath("stroke");
} else if (t === "B" || t === "B*" || t === "b" || t === "b*") {
// fill + stroke combined
flushPath("fill");
flushPath("stroke");
} else if (t === "n") {
// end path without painting — discard
pendingRects.length = 0;
pendingLines.length = 0;
}
idx++;
}
return segments;
}
/**
* Fast tokenizer for PDF content streams.
* Splits on whitespace, skipping comments, string literals, and inline image payloads.
*/
function tokenizeContentStream(raw: string): string[] {
const tokens: string[] = [];
const len = raw.length;
let i = 0;
let inInlineImage = false;
while (i < len) {
const ch = raw.charCodeAt(i);
// Skip whitespace
if (ch <= 32) {
i++;
continue;
}
// Skip comments
if (ch === 37 /* % */) {
while (i < len && raw.charCodeAt(i) !== 10) i++;
continue;
}
// Skip string literals (...)
if (ch === 40 /* ( */) {
let depth = 1;
i++;
while (i < len && depth > 0) {
const c = raw.charCodeAt(i);
if (c === 92 /* \ */) {
i++;
} else if (c === 40) {
depth++;
} else if (c === 41) {
depth--;
}
i++;
}
continue;
}
// Skip hex strings <...>
if (ch === 60 /* < */ && i + 1 < len && raw.charCodeAt(i + 1) !== 60) {
i++;
while (i < len && raw.charCodeAt(i) !== 62) i++;
i++; // skip >
continue;
}
// Skip dict delimiters << >>
if (ch === 60 && i + 1 < len && raw.charCodeAt(i + 1) === 60) {
i += 2;
continue;
}
if (ch === 62 && i + 1 < len && raw.charCodeAt(i + 1) === 62) {
i += 2;
continue;
}
// Skip stray closing delimiters from malformed streams. They cannot start
// a token, so leaving i unchanged would spin forever.
if (ch === 41 || ch === 62) {
i++;
continue;
}
// Regular token: read until whitespace or delimiter
const start = i;
while (i < len) {
const c = raw.charCodeAt(i);
if (c <= 32 || c === 40 || c === 41 || c === 60 || c === 62 || c === 37) break;
i++;
}
if (i > start) {
const token = raw.substring(start, i);
tokens.push(token);
if (token === "BI") {
inInlineImage = true;
} else if (token === "ID" && inInlineImage) {
while (i < len && raw.charCodeAt(i) <= 32) i++;
while (i < len) {
const c = raw.charCodeAt(i);
const prev = i === 0 ? 32 : raw.charCodeAt(i - 1);
const next = i + 2 >= len ? 32 : raw.charCodeAt(i + 2);
if (c === 69 && raw.charCodeAt(i + 1) === 73 && prev <= 32 && next <= 32) {
i += 2;
break;
}
i++;
}
inInlineImage = false;
}
}
}
return tokens;
}
// ---------------------------------------------------------------------------
// Image region detection
// ---------------------------------------------------------------------------
/** Minimum area (pts²) for an image to be considered a diagram, not an icon. */
const MIN_IMAGE_AREA = 5000;
function extractImageRegions(stext: StructuredTextJSON, pageNumber: number, pageHeight: number): ImageRegion[] {
const regions: ImageRegion[] = [];
for (const block of stext.blocks) {
if (block.type !== "image") continue;
const { x, y, w, h } = block.bbox;
if (w * h < MIN_IMAGE_AREA) continue; // skip tiny icons
// Convert Y from mupdf (top-left) to PDF (bottom-left) for ordering
const pdfTopY = pageHeight - y;
regions.push({
id: `p${pageNumber}-img${regions.length}`,
pageNumber,
bbox: { x, y, w, h },
topY: pdfTopY,
});
}
return regions;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/**
* Render an image region from a PDF page as a PNG buffer.
* Uses mupdf's DrawDevice to render just the cropped area at 2x resolution.
*/
export async function renderImageRegion(input: Uint8Array, region: ImageRegion): Promise<Uint8Array> {
const m = await loadMupdf();
const doc = m.Document.openDocument(input, "application/pdf");
const page = doc.loadPage(region.pageNumber - 1);
const pad = 10;
const bx = region.bbox.x - pad;
const by = region.bbox.y - pad;
const bw = region.bbox.w + 2 * pad;
const bh = region.bbox.h + 2 * pad;
const scale = 2;
const pw = Math.round(bw * scale);
const ph = Math.round(bh * scale);
const pix = new m.Pixmap(m.ColorSpace.DeviceRGB, [0, 0, pw, ph], false);
pix.clear(255);
const matrix: mupdf.Matrix = [scale, 0, 0, scale, -bx * scale, -by * scale];
const dl = page.toDisplayList();
const dev = new m.DrawDevice(matrix, pix);
dl.run(dev, m.Matrix.identity);
dev.close();
return pix.asPNG();
}
/**
* Extract text boxes and vector segments from all pages of a PDF buffer.
*/
export async function extractPages(input: Uint8Array): Promise<PageContent[]> {
const m = await loadMupdf();
const doc = m.Document.openDocument(input, "application/pdf");
const pages: PageContent[] = [];
for (let i = 0; i < doc.countPages(); i++) {
const pageNumber = i + 1;
const page = doc.loadPage(i);
const bounds = page.getBounds();
const pageHeight = bounds[3] - bounds[1];
// Single structured text pass with both flags
const stext = JSON.parse(
page.toStructuredText("preserve-whitespace,preserve-images").asJSON(),
) as StructuredTextJSON;
// Extract text boxes and image regions from the same parse
const textBoxes = extractTextBoxes(page, pageNumber, pageHeight, stext);
const images = extractImageRegions(stext, pageNumber, pageHeight);
// Extract vector segments from raw content stream
let segments: Segment[] = [];
try {
const pageObj = (page as mupdf.PDFPage).getObject();
const contents = pageObj.get("Contents");
if (contents) {
let rawBytes: Uint8Array;
if (contents.isArray()) {
// Multiple content streams — concatenate
const parts: Uint8Array[] = [];
const len = contents.length ?? 0;
for (let j = 0; j < len; j++) {
const stream = contents.get(j);
if (stream?.readStream) {
parts.push(stream.readStream().asUint8Array());
}
}
const totalLen = parts.reduce((s, p) => s + p.length, 0);
rawBytes = new Uint8Array(totalLen);
let offset = 0;
for (const part of parts) {
rawBytes.set(part, offset);
offset += part.length;
}
} else {
rawBytes = contents.readStream().asUint8Array();
}
const raw = new TextDecoder().decode(rawBytes);
segments = extractSegmentsFromContentStream(raw, pageNumber);
}
} catch {
// Content stream extraction failed — proceed with text only
}
pages.push({ pageNumber, textBoxes, segments, images });
}
return pages;
}
@@ -1,780 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Table grid detection from vector segments and text boxes.
*
* Ported from @oharato/pdf2md-ts with TypeScript types and without
* CJK-specific borderless table heuristics. The core algorithm:
*
* 1. Classify segments as horizontal or vertical lines
* 2. Group horizontal Y-lines into table groups (split by vertical gaps)
* 3. For each group:
* a. Full grid (H+V lines): build cells from grid intersections,
* place text via raycasting
* b. H-line only (no V lines): infer columns from text X positions
* 4. Prune empty rows/cols
*
* Coordinate system: PDF native (bottom-left origin, Y increases upward).
*/
import type { Segment, TableCell, TableGrid, TextBox } from "./types";
export interface GridResult {
grids: TableGrid[];
consumedIds: string[];
}
type RayDirection = "up" | "down" | "left" | "right";
interface Ray {
direction: RayDirection;
segmentId: string | null;
distance: number;
}
interface Interval {
min: number;
max: number;
}
function castRaysForTextBox(textBox: TextBox, segments: Segment[]): Ray[] {
const cx = (textBox.bounds.left + textBox.bounds.right) / 2;
const cy = (textBox.bounds.top + textBox.bounds.bottom) / 2;
let up: Ray = { direction: "up", segmentId: null, distance: Infinity };
let down: Ray = { direction: "down", segmentId: null, distance: Infinity };
let left: Ray = { direction: "left", segmentId: null, distance: Infinity };
let right: Ray = {
direction: "right",
segmentId: null,
distance: Infinity,
};
for (const seg of segments) {
const isH = Math.abs(seg.y1 - seg.y2) < 0.5;
const isV = Math.abs(seg.x1 - seg.x2) < 0.5;
if (isH) {
const minX = Math.min(seg.x1, seg.x2);
const maxX = Math.max(seg.x1, seg.x2);
if (cx >= minX && cx <= maxX) {
const d = seg.y1 - cy;
if (d >= 0 && d < up.distance) up = { direction: "up", segmentId: seg.id, distance: d };
const dd = cy - seg.y1;
if (dd >= 0 && dd < down.distance) down = { direction: "down", segmentId: seg.id, distance: dd };
}
}
if (isV) {
const minY = Math.min(seg.y1, seg.y2);
const maxY = Math.max(seg.y1, seg.y2);
if (cy >= minY && cy <= maxY) {
const d = cx - seg.x1;
if (d >= 0 && d < left.distance) left = { direction: "left", segmentId: seg.id, distance: d };
const rd = seg.x1 - cx;
if (rd >= 0 && rd < right.distance) right = { direction: "right", segmentId: seg.id, distance: rd };
}
}
}
return [up, down, left, right];
}
// ---------------------------------------------------------------------------
// Utility
// ---------------------------------------------------------------------------
const AXIS_EPSILON = 0.8;
const PAGE_MARGIN = 20;
function uniqueSorted(values: number[]): number[] {
const sorted = [...values].sort((a, b) => a - b);
const result: number[] = [];
for (const v of sorted) {
if (result.length === 0 || Math.abs(result[result.length - 1] - v) > 1) result.push(v);
}
return result;
}
// ---------------------------------------------------------------------------
// Y-line group splitting
// ---------------------------------------------------------------------------
function chainCoversRange(intervals: Interval[], lowerY: number, upperY: number, eps: number): boolean {
const sorted = [...intervals].sort((a, b) => a.min - b.min);
let covered = lowerY;
for (const iv of sorted) {
if (iv.min > covered + eps) break;
if (iv.max > covered) covered = iv.max;
if (covered >= upperY - eps) return true;
}
return false;
}
function countBridgingVLineCols(upperY: number, lowerY: number, verticals: Segment[]): number {
const eps = 1.5;
const byX = new Map<number, Interval[]>();
for (const seg of verticals) {
const rx = Math.round(seg.x1);
if (!byX.has(rx)) byX.set(rx, []);
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
}
let count = 0;
for (const intervals of byX.values()) {
if (chainCoversRange(intervals, lowerY, upperY, eps)) count++;
}
return count;
}
function bridgingXSet(upperY: number, lowerY: number, verticals: Segment[]): Set<number> {
const eps = 1.5;
const xs = new Set<number>();
const byX = new Map<number, Interval[]>();
for (const seg of verticals) {
const rx = Math.round(seg.x1);
if (!byX.has(rx)) byX.set(rx, []);
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
}
for (const [rx, intervals] of byX) {
if (chainCoversRange(intervals, lowerY, upperY, eps)) xs.add(rx);
}
return xs;
}
const MIN_RICH_BRIDGING_COLS = 3;
function splitYLinesIntoGroups(yLines: number[], verticals: Segment[]): number[][] {
if (yLines.length === 0) return [];
const eps = 1.5;
const allX = verticals.map(s => Math.round(s.x1));
const globalXMin = allX.length > 0 ? Math.min(...allX) : 0;
const globalXMax = allX.length > 0 ? Math.max(...allX) : 0;
const groups: number[][] = [];
let currentGroup = [yLines[0]];
let prevBridgingCols = -1;
for (let i = 1; i < yLines.length; i++) {
const upperY = yLines[i - 1];
const lowerY = yLines[i];
const cols = countBridgingVLineCols(upperY, lowerY, verticals);
if (cols === 0) {
groups.push(currentGroup);
currentGroup = [yLines[i]];
prevBridgingCols = -1;
continue;
}
if (prevBridgingCols >= MIN_RICH_BRIDGING_COLS && cols < MIN_RICH_BRIDGING_COLS) {
const bxs = bridgingXSet(upperY, lowerY, verticals);
const isOuterFrameOnly = [...bxs].every(
x => Math.abs(x - globalXMin) <= eps || Math.abs(x - globalXMax) <= eps,
);
if (!isOuterFrameOnly) {
groups.push(currentGroup);
currentGroup = [yLines[i - 1], yLines[i]];
prevBridgingCols = cols;
continue;
}
}
currentGroup.push(yLines[i]);
prevBridgingCols = cols;
}
groups.push(currentGroup);
return groups;
}
// ---------------------------------------------------------------------------
// Sub-row Y-cluster expansion
// ---------------------------------------------------------------------------
const Y_CLUSTER_GAP = 10;
const MIN_COLS_IN_TOP_CLUSTER = 2;
function assignToYCluster(y: number, clusters: number[]): number {
let closest = 0;
let closestDist = Math.abs(y - clusters[0]);
for (let k = 1; k < clusters.length; k++) {
const d = Math.abs(y - clusters[k]);
if (d < closestDist) {
closestDist = d;
closest = k;
}
}
return closest;
}
function expandSubRowsByYClusters(
originalRows: number,
cols: number,
cells: TableCell[],
cellBoxes: Map<TableCell, TextBox[]>,
): number {
let addedRows = 0;
for (let origRow = 0; origRow < originalRows; origRow++) {
const currentRow = origRow + addedRows;
const rowCellInfos: Array<{ cell: TableCell; col: number; boxes: TextBox[] }> = [];
for (let col = 0; col < cols; col++) {
const cell = cells.find(c => c.row === currentRow && c.col === col);
if (!cell) continue;
const boxes = cellBoxes.get(cell);
if (boxes && boxes.length > 0) rowCellInfos.push({ cell, col, boxes });
}
if (rowCellInfos.length === 0) continue;
const allMidYs = rowCellInfos.flatMap(({ boxes }) => boxes.map(b => (b.bounds.top + b.bounds.bottom) / 2));
const sortedY = [...new Set(allMidYs.map(y => Math.round(y * 10) / 10))].sort((a, b) => b - a);
const clusters = [sortedY[0]];
for (let i = 1; i < sortedY.length; i++) {
if (clusters[clusters.length - 1] - sortedY[i] > Y_CLUSTER_GAP) {
clusters.push(sortedY[i]);
}
}
if (clusters.length < 2) continue;
const colsInTopCluster = new Set<number>();
const totalNonEmptyCols = new Set<number>();
for (const { col, boxes } of rowCellInfos) {
totalNonEmptyCols.add(col);
if (boxes.some(b => assignToYCluster((b.bounds.top + b.bounds.bottom) / 2, clusters) === 0)) {
colsInTopCluster.add(col);
}
}
if (colsInTopCluster.size < MIN_COLS_IN_TOP_CLUSTER) continue;
if (colsInTopCluster.size >= totalNonEmptyCols.size) continue;
const sparseColsHaveMultipleBoxes = rowCellInfos.some(
({ col, boxes }) => !colsInTopCluster.has(col) && boxes.length > 1,
);
if (!sparseColsHaveMultipleBoxes) continue;
const numSubRows = clusters.length;
const numNewRows = numSubRows - 1;
for (const cell of cells) {
if (cell.row > currentRow) cell.row += numNewRows;
}
for (let subRow = 1; subRow < numSubRows; subRow++) {
for (let col = 0; col < cols; col++) {
cells.push({
row: currentRow + subRow,
col,
text: "",
rowSpan: 1,
colSpan: 1,
});
}
}
for (const { cell: origCell, col, boxes } of rowCellInfos) {
const subRowBoxGroups: TextBox[][] = Array.from({ length: numSubRows }, () => []);
for (const box of boxes) {
const cy = (box.bounds.top + box.bounds.bottom) / 2;
subRowBoxGroups[assignToYCluster(cy, clusters)].push(box);
}
cellBoxes.set(origCell, subRowBoxGroups[0]);
if (subRowBoxGroups[0].length === 0) cellBoxes.delete(origCell);
for (let subRow = 1; subRow < numSubRows; subRow++) {
if (subRowBoxGroups[subRow].length > 0) {
const newCell = cells.find(c => c.row === currentRow + subRow && c.col === col);
if (newCell) cellBoxes.set(newCell, subRowBoxGroups[subRow]);
}
}
}
addedRows += numNewRows;
}
return originalRows + addedRows;
}
// ---------------------------------------------------------------------------
// Cross-column text box splitting
// ---------------------------------------------------------------------------
/**
* Find which column a horizontal position falls into.
* Returns -1 if outside the grid.
*/
function findCol(x: number, xLines: number[]): number {
for (let i = 0; i < xLines.length - 1; i++) {
if (x >= xLines[i] && x <= xLines[i + 1]) return i;
}
return -1;
}
/**
* When a text box spans across one or more vertical column boundaries,
* split it into multiple virtual text boxes — one per column — with the
* text divided proportionally by width.
*
* We split at word boundaries closest to the proportional split point
* so we don't chop words in half.
*/
function splitCrossColumnBoxes(textBoxes: TextBox[], xLines: number[]): TextBox[] {
const result: TextBox[] = [];
const MARGIN = 5; // allow small overlap before considering it cross-column
for (const tb of textBoxes) {
const leftCol = findCol(tb.bounds.left + MARGIN, xLines);
const rightCol = findCol(tb.bounds.right - MARGIN, xLines);
// Not spanning columns, or outside grid — keep as-is
if (leftCol < 0 || rightCol < 0 || leftCol === rightCol) {
result.push(tb);
continue;
}
// Text box spans from leftCol to rightCol — split it
const totalWidth = tb.bounds.right - tb.bounds.left;
if (totalWidth <= 0) {
result.push(tb);
continue;
}
const words = tb.text.split(/\s+/);
if (words.length <= 1) {
// Single word spanning columns — just assign to whichever col has more overlap
result.push(tb);
continue;
}
// For each column boundary crossing, find the best word-boundary split
let remainingWords = [...words];
let currentLeft = tb.bounds.left;
for (let col = leftCol; col <= rightCol && remainingWords.length > 0; col++) {
const colRight = col < xLines.length - 1 ? xLines[col + 1] : tb.bounds.right;
const segmentRight = Math.min(colRight, tb.bounds.right);
if (col === rightCol) {
// Last column — take all remaining words
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: remainingWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: tb.bounds.right,
},
});
remainingWords = [];
} else {
// Find how many words fit in this column segment proportionally
const segmentWidth = segmentRight - currentLeft;
const fractionOfTotal = segmentWidth / totalWidth;
const approxChars = Math.round(fractionOfTotal * tb.text.length);
// Walk words to find the split closest to the proportional point
let charCount = 0;
let splitIdx = 0;
for (let w = 0; w < remainingWords.length; w++) {
const nextCount = charCount + remainingWords[w].length + (w > 0 ? 1 : 0);
if (nextCount > approxChars && splitIdx > 0) break;
charCount = nextCount;
splitIdx = w + 1;
}
if (splitIdx === 0) splitIdx = 1; // take at least one word
if (splitIdx >= remainingWords.length) {
// All remaining words fit here
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: remainingWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: segmentRight,
},
});
remainingWords = [];
} else {
const partWords = remainingWords.slice(0, splitIdx);
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: partWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: segmentRight,
},
});
remainingWords = remainingWords.slice(splitIdx);
currentLeft = segmentRight;
}
}
}
}
return result;
}
// ---------------------------------------------------------------------------
// Full grid table (H + V lines)
// ---------------------------------------------------------------------------
function buildCells(rows: number, cols: number): TableCell[] {
const cells: TableCell[] = [];
for (let row = 0; row < rows; row++) {
for (let col = 0; col < cols; col++) {
cells.push({ row, col, text: "", rowSpan: 1, colSpan: 1 });
}
}
return cells;
}
function buildTableGrid(
pageNumber: number,
yLines: number[],
xLines: number[],
filteredSegments: Segment[],
textBoxes: TextBox[],
): { grid: TableGrid; consumedIds: string[] } {
let rows = yLines.length - 1;
const cols = xLines.length - 1;
const cells = buildCells(rows, cols);
const consumedIds: string[] = [];
const yMin = yLines[yLines.length - 1];
const yMax = yLines[0];
const xMin = xLines[0];
const xMax = xLines[xLines.length - 1];
// Split text boxes that span multiple columns before placement
const splitBoxes = splitCrossColumnBoxes(textBoxes, xLines);
// Track which split piece IDs get placed in cells, so we can consume
// the original (unsplit) text box IDs too.
const placedSplitIds = new Set<string>();
// Look for header text boxes just above the grid.
// Use the ORIGINAL (unsplit) text boxes for header detection so that
// wide paragraph text isn't falsely split into column-sized header chunks.
// Reject boxes wider than 1.5 columns — those are paragraph text, not headers.
const avgColWidth = (xMax - xMin) / cols;
const maxHeaderBoxWidth = avgColWidth * 1.5;
const headerBoxes = textBoxes.filter(tb => {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const boxWidth = tb.bounds.right - tb.bounds.left;
return cy > yMax && cy <= yMax + 20 && cx >= xMin && cx <= xMax && boxWidth <= maxHeaderBoxWidth;
});
if (headerBoxes.length > 0) {
rows += 1;
for (const cell of cells) cell.row += 1;
for (let col = 0; col < cols; col++) {
cells.push({ row: 0, col, text: "", rowSpan: 1, colSpan: 1 });
}
for (const tb of headerBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col >= 0 && col < cols) {
const cell = cells.find(c => c.row === 0 && c.col === col);
if (cell) {
cell.text = cell.text.length === 0 ? tb.text : `${cell.text} ${tb.text}`;
consumedIds.push(tb.id);
}
}
}
}
const cellBoxes = new Map<TableCell, TextBox[]>();
for (const tb of splitBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
if (cy < yMin || cy > yMax || cx < xMin || cx > xMax) continue;
const rays = castRaysForTextBox(tb, filteredSegments);
const rayConfidence = rays.filter(r => r.segmentId !== null).length;
let row = yLines.findIndex((lineY, idx) => {
const next = yLines[idx + 1];
return next !== undefined && cy <= lineY && cy >= next;
});
if (row < 0 || row >= (headerBoxes.length > 0 ? rows - 1 : rows)) continue;
if (headerBoxes.length > 0) row += 1;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col < 0 || col >= cols) continue;
if (rayConfidence === 0) continue;
const cell = cells.find(c => c.row === row && c.col === col);
if (!cell) continue;
if (!cellBoxes.has(cell)) cellBoxes.set(cell, []);
cellBoxes.get(cell)?.push(tb);
consumedIds.push(tb.id);
if (tb.id.includes("-split")) placedSplitIds.add(tb.id);
}
rows = expandSubRowsByYClusters(rows, cols, cells, cellBoxes);
// Merge text boxes within each cell into cell text
for (const [cell, boxes] of cellBoxes.entries()) {
boxes.sort((a, b) => b.bounds.top - a.bounds.top);
const lines: string[] = [];
let currentLine: string[] = [];
let currentY = boxes[0].bounds.top;
for (const box of boxes) {
if (Math.abs(box.bounds.top - currentY) > 5) {
lines.push(currentLine.join(" "));
currentLine = [box.text];
currentY = box.bounds.top;
} else {
currentLine.push(box.text);
}
}
if (currentLine.length > 0) lines.push(currentLine.join(" "));
cell.text = lines.join("<br>");
}
const grid = pruneEmptyRowsAndCols({
pageNumber,
rows,
cols,
cells,
warnings: [],
topY: yLines[0],
isBorderless: false,
});
// Also consume the original (unsplit) text box IDs when any of their
// split pieces were placed in a cell.
for (const splitId of placedSplitIds) {
const origId = splitId.replace(/-split\d+$/, "");
if (!consumedIds.includes(origId)) {
consumedIds.push(origId);
}
}
return { grid, consumedIds };
}
// ---------------------------------------------------------------------------
// H-line-only table (inferred columns)
// ---------------------------------------------------------------------------
const COL_GAP_THRESHOLD = 20;
const HONLY_ROW_GAP = 30;
const HONLY_ROW_TOLERANCE = 8;
const MIN_TABLE_HEIGHT = 24;
const MIN_LEFT_SPREAD = 50;
function inferXLinesFromBoxes(textBoxes: TextBox[], xMin: number, xMax: number): number[] {
const centers = textBoxes.map(tb => (tb.bounds.left + tb.bounds.right) / 2).sort((a, b) => a - b);
if (centers.length === 0) return [xMin, xMax];
const boundaries = [xMin];
for (let i = 1; i < centers.length; i++) {
if (centers[i] - centers[i - 1] >= COL_GAP_THRESHOLD) {
boundaries.push((centers[i - 1] + centers[i]) / 2);
}
}
boundaries.push(xMax);
return boundaries;
}
function buildHLineOnlyTable(
pageNumber: number,
yLines: number[],
xMin: number,
xMax: number,
textBoxes: TextBox[],
alreadyConsumed: Set<string>,
): { grid: TableGrid; consumedIds: string[] } | null {
const yMax = yLines[0];
const yMin = yLines[yLines.length - 1];
const candidates = textBoxes.filter(tb => !alreadyConsumed.has(tb.id));
const BOX_LEFT_TOLERANCE = 30;
const inRange = candidates.filter(tb => {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
return (
tb.bounds.left >= xMin - BOX_LEFT_TOLERANCE &&
tb.bounds.right <= xMax + BOX_LEFT_TOLERANCE &&
cy >= yMin &&
cy <= yMax
);
});
// Extend downward below yMin
const belowYMin = candidates
.filter(tb => {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
return cx >= xMin && cx <= xMax && cy < yMin;
})
.sort((a, b) => (b.bounds.top + b.bounds.bottom) / 2 - (a.bounds.top + a.bounds.bottom) / 2);
const extensionBoxes: TextBox[] = [];
let lastY = yMin;
for (const tb of belowYMin) {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
if (lastY - cy > HONLY_ROW_GAP) break;
extensionBoxes.push(tb);
lastY = cy;
}
const allBoxes = [...inRange, ...extensionBoxes];
if (allBoxes.length === 0) return null;
const leftEdges = allBoxes.map(tb => tb.bounds.left);
if (Math.max(...leftEdges) - Math.min(...leftEdges) < MIN_LEFT_SPREAD) return null;
const xLines = inferXLinesFromBoxes(allBoxes, xMin, xMax);
if (xLines.length < 2) return null;
const cols = xLines.length - 1;
// Build visual rows
const visualRows: Array<{ midY: number; boxes: TextBox[] }> = [];
const sortedBoxes = [...allBoxes].sort((a, b) => {
const ya = (a.bounds.top + a.bounds.bottom) / 2;
const yb = (b.bounds.top + b.bounds.bottom) / 2;
if (Math.abs(ya - yb) > 0.5) return yb - ya;
return a.bounds.left - b.bounds.left;
});
for (const box of sortedBoxes) {
const cy = (box.bounds.top + box.bounds.bottom) / 2;
const last = visualRows[visualRows.length - 1];
if (last && Math.abs(last.midY - cy) <= HONLY_ROW_TOLERANCE) {
last.boxes.push(box);
} else {
visualRows.push({ midY: cy, boxes: [box] });
}
}
if (visualRows.length === 0) return null;
const cells: TableCell[] = [];
const consumedIds: string[] = [];
for (let rowIdx = 0; rowIdx < visualRows.length; rowIdx++) {
const vrow = visualRows[rowIdx];
const colBoxes = new Map<number, TextBox[]>();
for (const box of vrow.boxes) {
const cx = (box.bounds.left + box.bounds.right) / 2;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col >= 0 && col < cols) {
if (!colBoxes.has(col)) colBoxes.set(col, []);
colBoxes.get(col)?.push(box);
}
}
for (let c = 0; c < cols; c++) {
const cbs = (colBoxes.get(c) ?? []).sort((a, b) => a.bounds.left - b.bounds.left);
cells.push({
row: rowIdx,
col: c,
text: cbs.map(b => b.text).join(" "),
rowSpan: 1,
colSpan: 1,
});
consumedIds.push(...cbs.map(b => b.id));
}
}
const contentTopY = visualRows.length > 0 ? visualRows[0].midY : yMax;
const grid = pruneEmptyRowsAndCols({
pageNumber,
rows: visualRows.length,
cols,
cells,
warnings: [],
topY: contentTopY,
isBorderless: false,
});
return { grid, consumedIds };
}
// ---------------------------------------------------------------------------
// Pruning
// ---------------------------------------------------------------------------
function pruneEmptyRowsAndCols(table: TableGrid): TableGrid {
const occupiedRows = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.row));
const occupiedCols = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.col));
if (occupiedRows.size === 0) return table;
const rowMap = new Map<number, number>();
let newRow = 0;
for (let r = 0; r < table.rows; r++) {
if (occupiedRows.has(r)) rowMap.set(r, newRow++);
}
const colMap = new Map<number, number>();
let newCol = 0;
for (let c = 0; c < table.cols; c++) {
if (occupiedCols.has(c)) colMap.set(c, newCol++);
}
const prunedCells = table.cells
.filter(c => occupiedRows.has(c.row) && occupiedCols.has(c.col))
.map(c => ({
...c,
row: rowMap.get(c.row) ?? c.row,
col: colMap.get(c.col) ?? c.col,
}));
return { ...table, rows: newRow, cols: newCol, cells: prunedCells };
}
// ---------------------------------------------------------------------------
// Diagram vs table discrimination
// ---------------------------------------------------------------------------
/** Maximum column count for a plausible data table. */
const MAX_TABLE_COLS = 25;
/**
* Returns true if a grid looks like a vector diagram rather than a data table.
*
* Heuristics (any match → diagram):
* 1. Column count > 25 (diagrams create many X-lines from box edges)
* 2. Fill ratio < 25% (most cells empty — scattered boxes)
* 3. Fill < 50% AND duplicate text ratio > 30% (repeating labels in a
* diagram layout, e.g. "Hash", "Transaction" appearing in each column)
* 4. Fill < 50% AND cols >= 6 (moderate sparseness with wide grid)
*/
function isDiagram(grid: TableGrid): boolean {
const totalCells = grid.rows * grid.cols;
if (totalCells === 0) return true;
const filled = grid.cells.filter(c => c.text.trim().length > 0);
const fillRatio = filled.length / totalCells;
// Very high column count
if (grid.cols > MAX_TABLE_COLS) return true;
// Very sparse
if (fillRatio < 0.25) return true;
// Compute duplicate text ratio among non-trivial cells.
// Exclude short values (≤3 chars) like "—", "V", "YES", "NO" which
// naturally repeat in real data tables.
const substantive = filled.filter(c => c.text.trim().length > 3);
const uniqueTexts = new Set(substantive.map(c => c.text.trim())).size;
const dupRatio = substantive.length > 2 ? 1 - uniqueTexts / substantive.length : 0;
// Sparse + highly duplicated substantive text → repeating diagram
if (fillRatio < 0.5 && dupRatio > 0.3) return true;
// High duplication + wide grid → repeating diagram even at moderate fill
if (dupRatio > 0.4 && grid.cols >= 6) return true;
// Sparse + wide grid with no substantive text to judge
if (fillRatio < 0.4 && grid.cols >= 6) return true;
return false;
}
/**
* Detect all table grids on a single page from its text boxes and segments.
*/
export function resolveTableGrids(pageNumber: number, textBoxes: TextBox[], segments: Segment[]): GridResult {
const vertical = segments.filter(s => Math.abs(s.x1 - s.x2) <= AXIS_EPSILON);
const horizontal = segments.filter(s => Math.abs(s.y1 - s.y2) <= AXIS_EPSILON);
// Filter segments to the text's visible area
const textYValues = textBoxes.flatMap(t => [t.bounds.bottom, t.bounds.top]);
const textYMin = textYValues.length > 0 ? Math.min(...textYValues) - PAGE_MARGIN : -Infinity;
const textYMax = textYValues.length > 0 ? Math.max(...textYValues) + PAGE_MARGIN : Infinity;
const textXValues = textBoxes.flatMap(t => [t.bounds.left, t.bounds.right]);
const textXMin = textXValues.length > 0 ? Math.min(...textXValues) - 100 : -Infinity;
const textXMax = textXValues.length > 0 ? Math.max(...textXValues) + 100 : Infinity;
const filteredH = horizontal.filter(
s => s.y1 >= textYMin && s.y1 <= textYMax && s.x1 <= textXMax && s.x2 >= textXMin,
);
const hMaxX2 = filteredH.length > 0 ? Math.max(...filteredH.map(s => s.x2)) : textXMax;
const vSegXMax = Math.max(textXMax, hMaxX2 + PAGE_MARGIN);
const filteredV = vertical.filter(s => {
const segMin = Math.min(s.y1, s.y2);
const segMax = Math.max(s.y1, s.y2);
return segMax >= textYMin && segMin <= textYMax && s.x1 >= textXMin && s.x1 <= vSegXMax;
});
const allYLines = uniqueSorted(filteredH.flatMap(s => [s.y1, s.y2])).sort((a, b) => b - a);
if (allYLines.length < 2) {
return { grids: [], consumedIds: [] };
}
const filteredSegments = [...filteredH, ...filteredV];
const yGroups = splitYLinesIntoGroups(allYLines, filteredV);
const grids: TableGrid[] = [];
const gridConsumedIds: string[][] = [];
// Flat set for the alreadyConsumed check in H-line-only tables
const allConsumedIds: string[] = [];
for (const yLines of yGroups) {
if (yLines.length < 2) continue;
const yMin = yLines[yLines.length - 1];
const yMax = yLines[0];
const groupVerticals = filteredV.filter(s => {
const segMin = Math.min(s.y1, s.y2);
const segMax = Math.max(s.y1, s.y2);
return segMin < yMax - 1.5 && segMax > yMin + 1.5;
});
const groupXLines = uniqueSorted(groupVerticals.flatMap(s => [s.x1, s.x2]));
if (groupXLines.length < 2) {
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
const groupHoriz = filteredH.filter(s => s.y1 >= yMin - 1.5 && s.y1 <= yMax + 1.5);
if (groupHoriz.length === 0) continue;
const hxMin = Math.min(...groupHoriz.map(s => s.x1));
const hxMax = Math.max(...groupHoriz.map(s => s.x2));
const result = buildHLineOnlyTable(pageNumber, yLines, hxMin, hxMax, textBoxes, new Set(allConsumedIds));
if (result) {
grids.push(result.grid);
gridConsumedIds.push(result.consumedIds);
allConsumedIds.push(...result.consumedIds);
}
continue;
}
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
const result = buildTableGrid(pageNumber, yLines, groupXLines, filteredSegments, textBoxes);
grids.push(result.grid);
gridConsumedIds.push(result.consumedIds);
allConsumedIds.push(...result.consumedIds);
}
// Filter out grids that look like vector diagrams, not data tables.
// Their consumed text box IDs are released so the text becomes free text.
const filteredGrids: TableGrid[] = [];
const filteredConsumedIds: string[] = [];
for (let i = 0; i < grids.length; i++) {
if (isDiagram(grids[i])) continue;
filteredGrids.push(grids[i]);
filteredConsumedIds.push(...gridConsumedIds[i]);
}
return { grids: filteredGrids, consumedIds: filteredConsumedIds };
}
@@ -1,106 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Running header/footer detection and removal.
*
* Many PDFs have repeated text at the top or bottom of every page:
* document titles, chapter names, page numbers, copyright notices.
* These pollute the markdown output as false headings or noise.
*
* Algorithm:
* 1. For each page, bucket text boxes by Y position (top/bottom zones)
* 2. Collect the text content at each zone across all pages
* 3. Text appearing on >20% of pages OR 8+ consecutive pages is a
* running header/footer
* 4. Remove matching text boxes before further processing
*/
import type { PageContent } from "./types";
/** Minimum number of pages to enable header/footer detection. */
const MIN_PAGES = 5;
/** Minimum Y position for top zone (from bottom of page in PDF coords). */
const TOP_ZONE_MIN_Y = 700;
/** Maximum Y position for bottom zone. */
const BOTTOM_ZONE_MAX_Y = 80;
/**
* Minimum consecutive pages a text must appear on to be considered a
* running header/footer. Catches both document-wide headers (appearing
* on every page) and chapter-specific headers (appearing on 4+ consecutive
* pages within a chapter).
*/
const MIN_CONSECUTIVE_PAGES = 8;
/**
* Detect and remove running headers and footers from all pages.
* Mutates the pages array in place, removing header/footer text boxes.
*
* Uses two strategies:
* 1. Global frequency: text appearing on > 20% of all pages
* 2. Consecutive runs: text appearing on 8+ consecutive pages
*/
export function stripHeadersFooters(pages: PageContent[]): void {
if (pages.length < MIN_PAGES) return;
// Step 1: Build per-page zone text sets
const pageZoneTexts: Set<string>[] = [];
for (const page of pages) {
const zoneTexts = new Set<string>();
for (const tb of page.textBoxes) {
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
if (midY >= TOP_ZONE_MIN_Y || midY <= BOTTOM_ZONE_MAX_Y) {
const key = tb.text.trim().replace(/\s+/g, " ");
if (key.length > 0) zoneTexts.add(key);
}
}
pageZoneTexts.push(zoneTexts);
}
// Step 2: Count global frequency AND longest consecutive run for each text
const globalCount = new Map<string, number>();
const maxConsecutive = new Map<string, number>();
// Collect all unique zone texts
const allTexts = new Set<string>();
for (const zts of pageZoneTexts) {
for (const t of zts) allTexts.add(t);
}
for (const text of allTexts) {
let total = 0;
let consecutive = 0;
let maxRun = 0;
for (const zts of pageZoneTexts) {
if (zts.has(text)) {
total++;
consecutive++;
if (consecutive > maxRun) maxRun = consecutive;
} else {
consecutive = 0;
}
}
globalCount.set(text, total);
maxConsecutive.set(text, maxRun);
}
// Step 3: Identify running headers/footers
const globalThreshold = Math.max(3, Math.floor(pages.length * 0.2));
const repeatedTexts = new Set<string>();
for (const text of allTexts) {
const gc = globalCount.get(text) ?? 0;
const mc = maxConsecutive.get(text) ?? 0;
// Global: appears on 20%+ of pages
if (gc >= globalThreshold) {
repeatedTexts.add(text);
continue;
}
// Consecutive: appears on 8+ consecutive pages (chapter-level headers)
if (mc >= MIN_CONSECUTIVE_PAGES) {
repeatedTexts.add(text);
}
}
if (repeatedTexts.size === 0) return;
// Step 4: Remove matching text boxes from each page
for (const page of pages) {
page.textBoxes = page.textBoxes.filter(tb => {
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
if (midY < TOP_ZONE_MIN_Y && midY > BOTTOM_ZONE_MAX_Y) return true;
const normalized = tb.text.trim().replace(/\s+/g, " ");
return !repeatedTexts.has(normalized);
});
}
}
@@ -1,48 +1,10 @@
// Adapted from markit-ai (MIT). See ../../NOTICE. import { pdfToMarkdown } from "@oh-my-pi/pi-natives";
/**
* PDF to Markdown converter.
*
* Uses mupdf (native WASM) for fast PDF parsing and a custom pipeline for
* table detection via vector line extraction + raycasting.
*
* Pipeline:
* 1. Extract text boxes + vector segments + image regions per page (mupdf)
* 2. Detect column layout (single vs multi-column)
* 3. Per column: detect table grids from segments (grid detection + raycasting)
* 4. Render diagrams as PNG files (if output directory provided)
* 5. Render tables as markdown tables, free text as paragraphs/headings
*/
import * as path from "node:path";
import type { ConversionResult, Converter, StreamInfo } from "../../types"; import type { ConversionResult, Converter, StreamInfo } from "../../types";
import { detectColumns } from "./columns";
import { extractPages, renderImageRegion } from "./extract";
import { resolveTableGrids } from "./grid";
import { stripHeadersFooters } from "./headers";
import { renderPageContent } from "./render";
import type { Segment, TextBox } from "./types";
const EXTENSIONS = [".pdf"]; const EXTENSIONS = [".pdf"];
const MIMETYPES = ["application/pdf", "application/x-pdf"]; const MIMETYPES = ["application/pdf", "application/x-pdf"];
type ImageBlock = { topY: number; markdown: string }; /** Converts PDF buffers to Markdown through the native `pdf-inspector` bridge. */
/**
* Process a set of text boxes (one column or full page): run table detection,
* separate free text, and render to markdown.
*/
function processColumn(
pageNumber: number,
textBoxes: TextBox[],
segments: Segment[],
imageBlocks: ImageBlock[],
): string {
const { grids, consumedIds } = resolveTableGrids(pageNumber, textBoxes, segments);
const consumedSet = new Set(consumedIds);
const freeTextBoxes = textBoxes.filter(tb => !consumedSet.has(tb.id));
return renderPageContent(freeTextBoxes, grids, imageBlocks, textBoxes);
}
export class PdfConverter implements Converter { export class PdfConverter implements Converter {
name = "pdf"; name = "pdf";
@@ -56,91 +18,17 @@ export class PdfConverter implements Converter {
return false; return false;
} }
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> { async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
const pdfBytes = new Uint8Array(input); const result = await pdfToMarkdown(input);
const pages = await extractPages(pdfBytes); const notice =
// Remove running headers/footers before processing. result.pagesNeedingOcr.length > 0
stripHeadersFooters(pages); ? `Text extraction is incomplete for PDF pages ${result.pagesNeedingOcr.join(", ")}. Use the browser tool to render those pages or OCR them.`
const imageDir = streamInfo.imageDir; : undefined;
const pageMarkdowns: string[] = []; const conversion: ConversionResult = {
for (const page of pages) { markdown: notice ? [result.markdown, notice].filter(Boolean).join("\n\n") : result.markdown,
// Build image blocks for this page. };
const imageBlocks: ImageBlock[] = []; if (result.title !== undefined) conversion.title = result.title;
if (imageDir && page.images.length > 0) { return conversion;
for (const img of page.images) {
const filename = `${img.id}.png`;
const filepath = path.join(imageDir, filename);
try {
const png = await renderImageRegion(pdfBytes, img);
await Bun.write(filepath, png);
imageBlocks.push({ topY: img.topY, markdown: `![${img.id}](${filepath})` });
} catch {
// Image rendering failed — skip.
}
}
} else if (page.images.length > 0) {
for (const img of page.images) {
imageBlocks.push({
topY: img.topY,
markdown: `<!-- image: ${img.id} (page ${img.pageNumber}, ${img.bbox.w}x${img.bbox.h}pt) -->`,
});
}
}
// Detect column layout.
// If the page has vertical segments (tables), suppress column detection
// when one detected column is very narrow — that's a table's first column,
// not a page layout column.
const layout = detectColumns(page.textBoxes);
if (layout.columnCount > 1 && page.segments.some(s => Math.abs(s.x1 - s.x2) <= 0.8)) {
const pageXMin = Math.min(...page.textBoxes.map(tb => tb.bounds.left));
const pageXMax = Math.max(...page.textBoxes.map(tb => tb.bounds.right));
const pageWidth = pageXMax - pageXMin;
const minColFraction = 0.3;
const tooNarrow = layout.columns.some(col => {
const colXMin = Math.min(...col.map(tb => tb.bounds.left));
const colXMax = Math.max(...col.map(tb => tb.bounds.right));
return (colXMax - colXMin) / pageWidth < minColFraction;
});
if (tooNarrow) {
layout.columnCount = 1;
layout.columns = [page.textBoxes];
layout.boundaries = [];
}
}
if (layout.columnCount === 1) {
// Single column — process normally.
const md = processColumn(page.pageNumber, page.textBoxes, page.segments, imageBlocks);
if (md.length > 0) pageMarkdowns.push(md);
} else {
// Multi-column — process each column independently, then join.
const columnMarkdowns: string[] = [];
for (const colBoxes of layout.columns) {
// Filter segments to those within this column's X range.
const colXMin = Math.min(...colBoxes.map(tb => tb.bounds.left));
const colXMax = Math.max(...colBoxes.map(tb => tb.bounds.right));
const margin = 10;
const colSegments = page.segments.filter(seg => {
const segXMin = Math.min(seg.x1, seg.x2);
const segXMax = Math.max(seg.x1, seg.x2);
return segXMax >= colXMin - margin && segXMin <= colXMax + margin;
});
// Images go with the first column only (no X info to split by).
const md = processColumn(
page.pageNumber,
colBoxes,
colSegments,
columnMarkdowns.length === 0 ? imageBlocks : [],
);
if (md.length > 0) columnMarkdowns.push(md);
}
const joined = columnMarkdowns.join("\n\n");
if (joined.length > 0) pageMarkdowns.push(joined);
}
}
return { markdown: pageMarkdowns.join("\n\n") };
} }
} }
@@ -1,501 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Markdown rendering for PDF pages.
*
* Converts table grids and free text boxes into markdown, handling:
* - Table grid → markdown table (`| col | col |`)
* - Free text → paragraphs with heading detection (by font size)
* - Content ordering (top-to-bottom via Y coordinate)
* - Paragraph wrap merging (lines broken across PDF line boundaries)
* - Page number removal
*
* Ported from @oharato/pdf2md-ts, stripped of CJK/TDnet-specific logic.
*/
import type { ContentBlock, TableGrid, TextBox } from "./types";
/** A free-text line grouped from horizontally adjacent text boxes. */
interface RenderLine {
text: string;
topY: number;
fontSize: number;
isBold: boolean;
isTabular: boolean;
}
/** A content block carrying the Y of its last wrapped line during merging. */
type WrapBlock = ContentBlock & { lastTopY: number };
// ---------------------------------------------------------------------------
// Utility
// ---------------------------------------------------------------------------
/** Convert full-width ASCII characters (A→A, !→! etc.) to normal ASCII. */
function normalizeFullWidthAscii(text: string): string {
return text.replace(/[!-~]/g, ch => String.fromCharCode(ch.charCodeAt(0) - 0xfee0));
}
function escapePipes(text: string): string {
return normalizeFullWidthAscii(text).replaceAll("|", "\\|").replaceAll("\n", "<br>");
}
/** Parse a markdown pipe-delimited row into cell strings. */
function parsePipeRow(line: string): string[] {
const trimmed = line.trim();
if (!trimmed.startsWith("|") || !trimmed.endsWith("|")) return [];
return trimmed
.slice(1, -1)
.split("|")
.map(cell => cell.trim());
}
// ---------------------------------------------------------------------------
// Table rendering
// ---------------------------------------------------------------------------
/**
* Render a TableGrid as a markdown table.
*/
export function renderTableToMarkdown(table: TableGrid): string {
if (table.rows === 0 || table.cols === 0) return "";
const matrix = Array.from({ length: table.rows }, () => Array.from({ length: table.cols }, () => ""));
for (const cell of table.cells) {
if (cell.row < table.rows && cell.col < table.cols) {
matrix[cell.row][cell.col] = escapePipes(cell.text.trim());
}
}
const normalized = normalizeShiftedSparseColumns(matrix);
const promoted = promoteSubHeaderPrefixes(normalized);
const header = `| ${promoted[0].join(" | ")} |`;
const divider = `| ${Array.from({ length: promoted[0].length }, () => "---").join(" | ")} |`;
const body = promoted
.slice(1)
.map(row => `| ${row.join(" | ")} |`)
.join("\n");
return [header, divider, body].filter(l => l.length > 0).join("\n");
}
/**
* Fix tables with ≥5 columns where sparse single-value columns are
* misaligned. Shifts those values to the adjacent dense column and
* removes the now-empty sparse columns.
*/
function normalizeShiftedSparseColumns(matrix: string[][]): string[][] {
if (matrix.length === 0 || matrix[0].length < 5) return matrix;
const _rows = matrix.length;
const cols = matrix[0].length;
const counts = Array.from({ length: cols }, (_, c) =>
matrix.reduce((n, row) => n + (row[c].trim().length > 0 ? 1 : 0), 0),
);
const denseCols = new Set(
counts
.map((count, col) => ({ count, col }))
.filter(({ col, count }) => col === 0 || count >= 2)
.map(({ col }) => col),
);
const sparseCols = counts
.map((count, col) => ({ count, col }))
.filter(({ col, count }) => col > 0 && col < cols - 1 && count === 1)
.map(({ col }) => col);
if (sparseCols.length < 2 || denseCols.size < 4) return matrix;
const moves: Array<{ from: number; to: number; row: number }> = [];
for (const from of sparseCols) {
const row = matrix.findIndex(r => r[from].trim().length > 0);
const to = from + 1;
if (row < 0) return matrix;
if (!denseCols.has(to)) return matrix;
if (matrix[row][to].trim().length > 0) return matrix;
moves.push({ from, to, row });
}
const copy = matrix.map(row => [...row]);
for (const { from, to, row } of moves) {
copy[row][to] = copy[row][to].trim().length > 0 ? `${copy[row][to]} ${copy[row][from]}` : copy[row][from];
copy[row][from] = "";
}
const keepCols = Array.from({ length: cols }, (_, c) => c).filter(c => copy.some(row => row[c].trim().length > 0));
if (keepCols.length === cols) return copy;
return copy.map(row => keepCols.map(c => row[c]));
}
/**
* When a data row has ≥2 parenthesized qualifiers in non-first columns
* (and the first column is empty), promote them into the header row.
*/
function promoteSubHeaderPrefixes(matrix: string[][]): string[][] {
if (matrix.length < 2) return matrix;
const PAREN_RE = /^\([^)]{1,40}\)$/;
const result = matrix.map(row => [...row]);
const cols = matrix[0].length;
const rowsToRemove = new Set<number>();
for (let r = 1; r < result.length; r++) {
if (rowsToRemove.has(r)) continue;
const promotable: Array<{ col: number; prefix: string; isFullCell: boolean }> = [];
for (let col = 1; col < cols; col++) {
const cell = (result[r][col] ?? "").trim();
if (!cell) continue;
const parts = cell.split("<br>");
if (parts.length === 1 && PAREN_RE.test(cell)) {
promotable.push({ col, prefix: cell, isFullCell: true });
} else if (parts.length >= 2 && PAREN_RE.test(parts[0].trim())) {
promotable.push({
col,
prefix: parts[0].trim(),
isFullCell: false,
});
}
}
if (promotable.length < 2) continue;
if (promotable.some(p => p.isFullCell) && result[r][0].trim().length > 0) continue;
for (const { col, prefix, isFullCell } of promotable) {
result[0][col] = result[0][col].trim() ? `${result[0][col]} ${prefix}` : prefix;
if (isFullCell) {
result[r][col] = "";
} else {
const parts = result[r][col].split("<br>");
result[r][col] = parts.slice(1).join("<br>");
}
}
if (result[r].every(cell => cell.trim().length === 0)) {
rowsToRemove.add(r);
}
}
return result.filter((_, r) => !rowsToRemove.has(r));
}
// ---------------------------------------------------------------------------
// Free text rendering
// ---------------------------------------------------------------------------
/** Y tolerance for grouping text boxes onto the same visual line. */
const TEXT_LINE_Y_TOLERANCE = 3;
/** Minimum X gap between adjacent boxes to mark line as tabular. */
const TABULAR_X_GAP = 30;
/**
* Minimum font size (pts) to consider when computing the modal body font.
* Tiny labels from diagrams, footnote markers, and superscripts are excluded
* so they don't skew the modal toward small sizes.
*/
const MIN_BODY_FONT_SIZE = 7;
/**
* Compute the most frequent font size among text boxes, ignoring very small
* text that likely comes from diagrams, footnotes, or superscripts.
*/
function modalFontSize(textBoxes: TextBox[]): number {
const counts = new Map<number, number>();
for (const tb of textBoxes) {
const size = Math.round((tb.fontSize ?? 0) * 10) / 10;
if (size < MIN_BODY_FONT_SIZE) continue;
counts.set(size, (counts.get(size) ?? 0) + 1);
}
let modal = 0;
let maxCount = 0;
for (const [size, count] of counts) {
if (count > maxCount) {
maxCount = count;
modal = size;
}
}
return modal;
}
/** Group free text boxes into horizontal lines, sorted top-to-bottom. */
function groupFreeTextIntoLines(textBoxes: TextBox[]): RenderLine[] {
if (textBoxes.length === 0) return [];
const sorted = [...textBoxes].sort((a, b) => {
const ya = (a.bounds.top + a.bounds.bottom) / 2;
const yb = (b.bounds.top + b.bounds.bottom) / 2;
const dy = yb - ya;
if (Math.abs(dy) > TEXT_LINE_Y_TOLERANCE) return dy;
return a.bounds.left - b.bounds.left;
});
const lines: RenderLine[] = [];
let curParts = [sorted[0].text];
let curBoxes = [sorted[0]];
let curY = (sorted[0].bounds.top + sorted[0].bounds.bottom) / 2;
let curTopY = curY;
let curFontSize = sorted[0].fontSize;
let curIsBold = sorted[0].isBold;
const finishLine = () => {
let isTabular = false;
for (let j = 1; j < curBoxes.length; j++) {
if (curBoxes[j].bounds.left - curBoxes[j - 1].bounds.right > TABULAR_X_GAP) {
isTabular = true;
break;
}
}
lines.push({
text: curParts.join(" "),
topY: curTopY,
fontSize: curFontSize,
isBold: curIsBold,
isTabular,
});
};
for (let i = 1; i < sorted.length; i++) {
const box = sorted[i];
const cy = (box.bounds.top + box.bounds.bottom) / 2;
if (Math.abs(cy - curY) <= TEXT_LINE_Y_TOLERANCE) {
curParts.push(box.text);
curBoxes.push(box);
curFontSize = Math.max(curFontSize, box.fontSize);
curIsBold = curIsBold || box.isBold;
} else {
finishLine();
curParts = [box.text];
curBoxes = [box];
curY = cy;
curTopY = cy;
curFontSize = box.fontSize;
curIsBold = box.isBold;
}
}
finishLine();
return lines;
}
/** Determine markdown heading prefix based on font size relative to body. */
function headingPrefix(fontSize: number, bodyFontSize: number, isBold: boolean): string {
if (bodyFontSize <= 0) return "";
const ratio = fontSize / bodyFontSize;
// Large headings (>2x body size)
if (ratio >= 2.0) return "# ";
// Medium headings (~1.5x body size)
if (ratio >= 1.4) return "## ";
// Small headings (bold and slightly larger)
if (ratio >= 1.1 && isBold) return "### ";
return "";
}
// ---------------------------------------------------------------------------
// Block merging
// ---------------------------------------------------------------------------
/** Merge consecutive blocks with the same heading prefix (wrapped headings). */
function mergeConsecutiveHeadings(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
if (blocks.length === 0) return [];
const HEADING_RE = /^(#{1,6} )/;
const maxGap = Math.max(bodyFS * 3, 30);
const merged: ContentBlock[] = [];
let cur: ContentBlock = { ...blocks[0] };
for (let i = 1; i < blocks.length; i++) {
const next = blocks[i];
const curMatch = cur.content.match(HEADING_RE);
const nextMatch = next.content.match(HEADING_RE);
const gap = cur.topY - next.topY;
if (curMatch && nextMatch && curMatch[1] === nextMatch[1] && gap <= maxGap) {
cur = {
topY: cur.topY,
content: `${cur.content} ${next.content.slice(nextMatch[1].length)}`,
isTabular: cur.isTabular || next.isTabular,
};
} else {
merged.push(cur);
cur = { ...next };
}
}
merged.push(cur);
return merged;
}
/**
* Merge consecutive plain-text blocks that are wrapped lines of the same paragraph.
*/
function mergeParagraphWraps(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
if (blocks.length === 0 || bodyFS <= 0) return blocks;
const HEADING_RE = /^#{1,6} /;
const SENTENCE_END_RE = /[.!?…)\]]\s*$/;
const maxGap = bodyFS * 2.0;
const MIN_WRAP_LENGTH = 25;
const merged: ContentBlock[] = [];
let cur: WrapBlock = { ...blocks[0], lastTopY: blocks[0].topY };
for (let i = 1; i < blocks.length; i++) {
const next = blocks[i];
const curIsBody = !HEADING_RE.test(cur.content) && !cur.content.startsWith("|");
const nextIsBody = !HEADING_RE.test(next.content) && !next.content.startsWith("|");
const gap = cur.lastTopY - next.topY;
const isWrap =
curIsBody &&
nextIsBody &&
!cur.isTabular &&
!next.isTabular &&
gap > 0 &&
gap <= maxGap &&
cur.content.length > MIN_WRAP_LENGTH &&
!SENTENCE_END_RE.test(cur.content);
if (isWrap) {
cur = {
topY: cur.topY,
lastTopY: next.topY,
content: `${cur.content.trimEnd()} ${next.content.trimStart()}`,
isTabular: false,
};
} else {
merged.push({ topY: cur.topY, content: cur.content });
cur = { ...next, lastTopY: next.topY };
}
}
merged.push({ topY: cur.topY, content: cur.content });
return merged;
}
/** Remove page number blocks near the bottom of the page. */
function removePageNumbers(blocks: ContentBlock[]): ContentBlock[] {
const PAGE_NUM_RE = /^(?:#{1,6}\s*)?\d+\s*$/;
const BOTTOM_Y = 120;
return blocks.filter((block, idx) => {
const isBottom = idx >= blocks.length - 3;
const isLowY = block.topY <= BOTTOM_Y;
const isPageNum = PAGE_NUM_RE.test(block.content.trim());
return !(isBottom && isLowY && isPageNum);
});
}
// ---------------------------------------------------------------------------
// Detached first-column table reconstruction
// ---------------------------------------------------------------------------
/**
* Fix tables where the first column was emitted as free text blocks
* around a markdown table containing only the right-side columns.
*
* Detects: a plain-text header line with (N+1) tokens above an N-column
* markdown table, plus short label lines whose count matches the table's
* logical row count. Reconstructs into a proper (N+1)-column table.
*/
function normalizeDetachedFirstColumnTables(blocks: ContentBlock[]): ContentBlock[] {
const HEADING_RE = /^#{1,6}\s/;
const isTableBlock = (text: string) => text.trimStart().startsWith("|");
const isPlainBlock = (text: string) => !HEADING_RE.test(text) && !isTableBlock(text);
const isShortLabel = (text: string) => {
const t = text.trim();
return t.length > 0 && t.length <= 40;
};
const splitTokens = (text: string) =>
text
.trim()
.split(/[ \t]+/)
.filter(Boolean);
const replacements = new Map<number, string>();
const remove = new Set<number>();
for (let tableIdx = 0; tableIdx < blocks.length; tableIdx++) {
if (remove.has(tableIdx)) continue;
const tableBlock = blocks[tableIdx];
if (!isTableBlock(tableBlock.content)) continue;
const tableLines = tableBlock.content
.split("\n")
.map(line => line.trim())
.filter(line => line.startsWith("|"));
const dataRows = tableLines
.filter(line => !/^\|\s*[-: ]+\|/.test(line))
.map(parsePipeRow)
.filter(row => row.length > 0);
if (dataRows.length === 0) continue;
const cols = dataRows[0].length;
if (cols < 2 || dataRows.some(row => row.length !== cols)) continue;
// Expand by <br> count to get logical row count
const logicalRows: string[][] = [];
for (const row of dataRows) {
const splitCells = row.map(cell => cell.split("<br>").map(p => p.trim()));
const rowSpan = Math.max(...splitCells.map(parts => parts.length));
for (let k = 0; k < rowSpan; k++) {
logicalRows.push(splitCells.map(parts => parts[k] ?? ""));
}
}
if (logicalRows.length < 2) continue;
// Find header with (cols + 1) non-numeric tokens
let headerIdx = -1;
let headerTokens: string[] = [];
for (let i = Math.max(0, tableIdx - 4); i <= tableIdx - 1; i++) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text)) continue;
const tokens = splitTokens(text);
if (tokens.length === cols + 1 && tokens.every(tok => !/[0-9]/.test(tok))) {
headerIdx = i;
headerTokens = tokens;
}
}
if (headerIdx < 0) continue;
// Collect short label lines above/below table
const aboveLabels: Array<{ idx: number; text: string }> = [];
for (let i = tableIdx - 1; i > headerIdx; i--) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text) || !isShortLabel(text)) break;
aboveLabels.push({ idx: i, text });
}
aboveLabels.reverse();
const belowLabels: Array<{ idx: number; text: string }> = [];
for (let i = tableIdx + 1; i < blocks.length; i++) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text) || !isShortLabel(text)) break;
belowLabels.push({ idx: i, text });
}
const labels = [...aboveLabels, ...belowLabels];
if (labels.length !== logicalRows.length) continue;
// Reconstruct the full table
const normalizedLines: string[] = [];
normalizedLines.push(`| ${headerTokens.join(" | ")} |`);
normalizedLines.push(`| ${Array.from({ length: cols + 1 }, () => "---").join(" | ")} |`);
for (let r = 0; r < logicalRows.length; r++) {
normalizedLines.push(`| ${labels[r].text} | ${logicalRows[r].join(" | ")} |`);
}
replacements.set(tableIdx, normalizedLines.join("\n"));
remove.add(headerIdx);
for (const label of labels) remove.add(label.idx);
}
if (replacements.size === 0 && remove.size === 0) return blocks;
const out: ContentBlock[] = [];
for (let i = 0; i < blocks.length; i++) {
if (remove.has(i)) continue;
const replaced = replacements.get(i);
if (replaced) {
out.push({ topY: blocks[i].topY, content: replaced });
} else {
out.push(blocks[i]);
}
}
return out;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/**
* Render one page's content: free text and tables interleaved top-to-bottom.
*/
export function renderPageContent(
freeTextBoxes: TextBox[],
tables: TableGrid[],
imageBlocks: Array<{ topY: number; markdown: string }> = [],
allTextBoxes?: TextBox[],
): string {
const blocks: ContentBlock[] = [];
// Use ALL text boxes (before table/diagram filtering) for modal font size,
// so that diagram labels released as free text don't skew the body size.
const bodyFS = modalFontSize(allTextBoxes ?? freeTextBoxes);
// Free text lines
for (const line of groupFreeTextIntoLines(freeTextBoxes)) {
const prefix = headingPrefix(line.fontSize, bodyFS, line.isBold);
blocks.push({
topY: line.topY,
content: prefix + line.text,
isTabular: prefix === "" && line.isTabular,
});
}
// Tables
for (const table of tables) {
const md = renderTableToMarkdown(table);
if (md.length > 0) {
blocks.push({ topY: table.topY, content: md });
}
}
// Images
for (const img of imageBlocks) {
blocks.push({ topY: img.topY, content: img.markdown });
}
// Sort top-to-bottom (higher Y = higher on page = comes first)
blocks.sort((a, b) => b.topY - a.topY);
const cleaned = removePageNumbers(blocks);
const headingsMerged = mergeConsecutiveHeadings(cleaned, bodyFS);
const merged = mergeParagraphWraps(headingsMerged, bodyFS);
const normalized = normalizeDetachedFirstColumnTables(merged);
return normalized
.map(b => b.content)
.join("\n\n")
.trim();
}
@@ -1,84 +0,0 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/** Bounding box in PDF coordinate space (origin = bottom-left). */
export type Bounds = {
left: number;
right: number;
/** Higher value = higher on the page. */
top: number;
bottom: number;
};
/** A text fragment with position and font metadata. */
export type TextBox = {
id: string;
text: string;
bounds: Bounds;
pageNumber: number;
/** Dominant font size in points. */
fontSize: number;
/** True if rendered bold (font name or rendering mode). */
isBold: boolean;
};
/** A horizontal or vertical line segment extracted from vector graphics. */
export type Segment = {
id: string;
x1: number;
y1: number;
x2: number;
y2: number;
};
/** A single cell in a resolved table grid. */
export type TableCell = {
row: number;
col: number;
text: string;
rowSpan: number;
colSpan: number;
};
/** A resolved table grid ready for markdown rendering. */
export type TableGrid = {
pageNumber: number;
rows: number;
cols: number;
cells: TableCell[];
warnings: string[];
/** Top Y coordinate (PDF space: larger = higher on page). */
topY: number;
/** True for tables detected without vector borders. */
isBorderless: boolean;
};
/** An image/diagram region detected on a page. */
export type ImageRegion = {
id: string;
pageNumber: number;
/** Bounding box in mupdf coordinates (top-left origin). */
bbox: {
x: number;
y: number;
w: number;
h: number;
};
/** Y position in PDF coordinates (bottom-left) for ordering. */
topY: number;
};
/** Result of extracting content from a single PDF page. */
export type PageContent = {
pageNumber: number;
textBoxes: TextBox[];
segments: Segment[];
images: ImageRegion[];
};
/** A block of rendered content (text paragraph or table). */
export type ContentBlock = {
topY: number;
content: string;
/** True if this line has wide gaps between text boxes (column headers). */
isTabular?: boolean;
};
@@ -1,250 +0,0 @@
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { isEexist, isEnotempty, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
import type { ToolSession } from "../sdk";
import { loadImageInput, MAX_IMAGE_INPUT_BYTES, webpExclusionForModel } from "../utils/image-loading";
import { convertFileWithMarkit } from "../utils/markit";
import type { ReadToolDetails } from "./read";
import { prependSuffixResolutionNotice } from "./read-format";
import { isNotFoundError } from "./read-path-resolution";
import { formatBytes } from "./render-utils";
import { ToolError } from "./tool-errors";
import { toolResult } from "./tool-result";
const MAX_IMAGE_SIZE = MAX_IMAGE_INPUT_BYTES;
const PDF_IMAGE_PLACEHOLDER_RE = /<!--\s*image:\s*([^\s<>]+)(.*?)-->/g;
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
const PDF_IMAGE_MEMBER_EXTENSION_RE = /\.png$/i;
const PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH = 96;
interface PdfImageSnapshot {
directory: string;
filePath: string;
digest: string;
}
interface PdfImageExtraction {
controller: AbortController;
promise: Promise<string>;
settled: boolean;
waiters: number;
}
const pdfImageExtractions = new Map<string, PdfImageExtraction>();
function pdfImageMemberPath(pdfPath: string, imageId: string): string {
const member = PDF_IMAGE_MEMBER_EXTENSION_RE.test(imageId) ? imageId : `${imageId}.png`;
return `${pdfPath}:${member}`;
}
export function rewritePdfImagePlaceholders(markdown: string, pdfPath: string): string {
return markdown.replace(PDF_IMAGE_PLACEHOLDER_RE, (_match: string, imageId: string, metadataText: string) => {
const metadata = metadataText.trim();
const suffix = metadata.length > 0 ? ` (${metadata})` : "";
return `Image ${imageId}${suffix}: read \`${pdfImageMemberPath(pdfPath, imageId)}\``;
});
}
export function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; member: string } | null {
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
if (!match) return null;
const pdfPath = match[1];
const member = match[2];
if (pdfPath === undefined || member === undefined) return null;
if (member.length !== 0 && !PDF_IMAGE_MEMBER_EXTENSION_RE.test(member)) return null;
return { pdfPath, member };
}
function pdfImageCacheDir(session: ToolSession, absolutePdfPath: string, contentDigest: string): string {
const artifactsDir = session.getArtifactsDir?.();
let root = artifactsDir ?? undefined;
if (root === undefined) {
const sessionFile = session.getSessionFile();
root = sessionFile?.endsWith(".jsonl") ? sessionFile.slice(0, -6) : path.join(os.tmpdir(), "omp-read-pdf-images");
}
const basename = path
.basename(absolutePdfPath)
.replace(/[^A-Za-z0-9._-]/g, "_")
.slice(0, PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH);
const pathDigest = Bun.hash(absolutePdfPath).toString(36);
return path.join(root, "read-pdf-images", `${basename}-${pathDigest}-${contentDigest}`);
}
async function snapshotPdfSource(absolutePdfPath: string, signal?: AbortSignal): Promise<PdfImageSnapshot> {
const directory = await fs.mkdtemp(path.join(os.tmpdir(), "omp-read-pdf-"));
try {
const bytes = await untilAborted(signal, () => Bun.file(absolutePdfPath).bytes());
signal?.throwIfAborted();
const digest = new Bun.CryptoHasher("sha256").update(bytes).digest("hex");
const filePath = path.join(directory, "source.pdf");
await Bun.write(filePath, bytes);
signal?.throwIfAborted();
return { directory, filePath, digest };
} catch (error) {
await fs.rm(directory, { recursive: true, force: true });
throw error;
}
}
async function listPdfImageMembers(imageDir: string): Promise<string[]> {
try {
const entries = await fs.readdir(imageDir, { withFileTypes: true });
const members: string[] = [];
for (const entry of entries) {
if (entry.isFile() && PDF_IMAGE_MEMBER_EXTENSION_RE.test(entry.name)) members.push(entry.name);
}
return members.sort();
} catch (error) {
if (isNotFoundError(error)) return [];
throw error;
}
}
async function extractPdfImages(snapshot: PdfImageSnapshot, imageDir: string, signal: AbortSignal): Promise<string> {
const markerPath = path.join(imageDir, ".extracted");
try {
await fs.stat(markerPath);
return imageDir;
} catch (error) {
if (!isNotFoundError(error)) throw error;
}
await fs.mkdir(path.dirname(imageDir), { recursive: true });
const stagingDir = await fs.mkdtemp(`${imageDir}.tmp-`);
let published = false;
try {
const result = await convertFileWithMarkit(snapshot.filePath, signal, { imageDir: stagingDir });
if (!result.ok) {
throw new ToolError(`Cannot extract images from PDF: ${result.error ?? "conversion failed"}`);
}
await Bun.write(path.join(stagingDir, ".extracted"), "ok");
try {
await fs.rename(stagingDir, imageDir);
published = true;
} catch (error) {
if (!isEexist(error) && !isEnotempty(error)) throw error;
try {
await fs.stat(markerPath);
} catch (markerError) {
if (isNotFoundError(markerError)) throw error;
throw markerError;
}
}
return imageDir;
} finally {
if (!published) await fs.rm(stagingDir, { recursive: true, force: true });
}
}
function createPdfImageExtraction(snapshot: PdfImageSnapshot, imageDir: string): PdfImageExtraction {
const controller = new AbortController();
const promise = extractPdfImages(snapshot, imageDir, controller.signal).finally(() =>
fs.rm(snapshot.directory, { recursive: true, force: true }),
);
const extraction: PdfImageExtraction = { controller, promise, settled: false, waiters: 0 };
const settle = () => {
extraction.settled = true;
if (pdfImageExtractions.get(imageDir) === extraction) pdfImageExtractions.delete(imageDir);
};
void promise.then(settle, settle);
return extraction;
}
async function waitForPdfImageExtraction(
extraction: PdfImageExtraction,
signal: AbortSignal | undefined,
): Promise<string> {
extraction.waiters++;
try {
return await untilAborted(signal, extraction.promise);
} finally {
extraction.waiters--;
if (extraction.waiters === 0 && !extraction.settled) {
extraction.controller.abort();
try {
await extraction.promise;
} catch {}
}
}
}
async function ensurePdfImageCache(
session: ToolSession,
absolutePdfPath: string,
signal?: AbortSignal,
): Promise<string> {
const snapshot = await snapshotPdfSource(absolutePdfPath, signal);
const imageDir = pdfImageCacheDir(session, absolutePdfPath, snapshot.digest);
const existing = pdfImageExtractions.get(imageDir);
if (existing && !existing.settled && !existing.controller.signal.aborted) {
await fs.rm(snapshot.directory, { recursive: true, force: true });
return waitForPdfImageExtraction(existing, signal);
}
const extraction = createPdfImageExtraction(snapshot, imageDir);
pdfImageExtractions.set(imageDir, extraction);
return waitForPdfImageExtraction(extraction, signal);
}
export async function readPdfImageMember(
session: ToolSession,
autoResizeImages: boolean,
absolutePdfPath: string,
pdfDisplayPath: string,
member: string,
suffixResolution: { from: string; to: string } | undefined,
signal?: AbortSignal,
): Promise<AgentToolResult<ReadToolDetails>> {
const imageDir = await ensurePdfImageCache(session, absolutePdfPath, signal);
const members = await listPdfImageMembers(imageDir);
if (member.length === 0) {
const text =
members.length === 0
? "No extractable PDF image members found."
: `Extractable PDF image members:\n${members
.map(imageMember => `- read \`${pdfDisplayPath}:${imageMember}\``)
.join("\n")}`;
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
.text(prependSuffixResolutionNotice(text, suffixResolution))
.sourcePath(absolutePdfPath)
.done();
}
if (!members.includes(member)) {
const available = members.length === 0 ? "(none)" : members.join(", ");
throw new ToolError(`PDF image member '${member}' not found. Available members: ${available}`);
}
const imagePath = path.join(imageDir, member);
const imageStat = await Bun.file(imagePath).stat();
if (imageStat.size > MAX_IMAGE_SIZE) {
const sizeStr = formatBytes(imageStat.size);
const maxStr = formatBytes(MAX_IMAGE_SIZE);
throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`);
}
const metadata = await readImageMetadata(imagePath);
const mimeType = metadata?.mimeType;
if (!mimeType) throw new ToolError(`PDF image member '${member}' is not a supported image.`);
const imageInput = await loadImageInput({
path: `${pdfDisplayPath}:${member}`,
cwd: session.cwd,
autoResize: autoResizeImages,
maxBytes: MAX_IMAGE_SIZE,
resolvedPath: imagePath,
detectedMimeType: mimeType,
excludeWebP: webpExclusionForModel(session.getActiveModel?.()),
});
if (!imageInput) {
throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`);
}
const textNote = prependSuffixResolutionNotice(imageInput.textNote, suffixResolution);
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
.content([
{ type: "text", text: textNote },
{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType },
])
.sourcePath(imageInput.resolvedPath)
.done();
}
@@ -0,0 +1,13 @@
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
/** Parse a former PDF image-member read without claiming normal selectors. */
export function splitUnsupportedPdfImageReadPath(readPath: string): { pdfPath: string } | null {
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
const pdfPath = match?.[1];
return pdfPath ? { pdfPath } : null;
}
/** Explain how to render a PDF now that the text backend has no rasterizer. */
export function pdfImageRenderingUnsupportedMessage(pdfPath: string): string {
return `pdf-inspector cannot render PDF images. Use the Puppeteer browser tool to render '${pdfPath}', or read '${pdfPath}' for extracted text.`;
}
+7 -33
View File
@@ -100,7 +100,7 @@ import {
isRemoteMountPath, isRemoteMountPath,
type SuffixMatchCache, type SuffixMatchCache,
} from "./read-path-resolution"; } from "./read-path-resolution";
import { readPdfImageMember, rewritePdfImagePlaceholders, splitPdfImageMemberReadPath } from "./read-pdf-images"; import { pdfImageRenderingUnsupportedMessage, splitUnsupportedPdfImageReadPath } from "./read-pdf";
import { isMultiRange, isRawSelector, type ParsedSelector, parseSel, selToOffsetLimit } from "./read-selector"; import { isMultiRange, isRawSelector, type ParsedSelector, parseSel, selToOffsetLimit } from "./read-selector";
import { readSqlite, resolveSqliteReadPath } from "./read-sqlite"; import { readSqlite, resolveSqliteReadPath } from "./read-sqlite";
import { isProseSummaryPath, renderSummary, routeReadThroughBridge, trySummarize } from "./read-summary"; import { isProseSummaryPath, renderSummary, routeReadThroughBridge, trySummarize } from "./read-summary";
@@ -892,7 +892,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
// Prefer a literal filesystem match over selector interpretation so real // Prefer a literal filesystem match over selector interpretation so real
// POSIX filenames containing selector-looking suffixes win over structured // POSIX filenames containing selector-looking suffixes win over structured
// archive / sqlite / pdf-image dispatch. A selector promoted from local:// // archive / sqlite / unsupported PDF-image dispatch. A selector promoted from local://
// remains separate so it cannot be mistaken for part of the resolved path. // remains separate so it cannot be mistaken for part of the resolved path.
const literalSplit = const literalSplit =
promotedSelector === undefined promotedSelector === undefined
@@ -925,35 +925,10 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
return readSqlite(sqlitePath, signal); return readSqlite(sqlitePath, signal);
} }
const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath); const unsupportedPdfImageRead =
if (pdfImageMemberPath) { literalSplit.sel === undefined ? splitUnsupportedPdfImageReadPath(readPath) : null;
let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd); if (unsupportedPdfImageRead && (await probeLiteralPathExists(readPath, this.session.cwd)) === "missing") {
let suffixResolution: { from: string; to: string } | undefined; throw new ToolError(pdfImageRenderingUnsupportedMessage(unsupportedPdfImageRead.pdfPath));
try {
const stat = await Bun.file(absolutePdfPath).stat();
if (stat.isDirectory())
throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`);
} catch (error) {
if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error;
const suffixMatch = await findSuffixMatchCached(
this.session,
suffixCache,
pdfImageMemberPath.pdfPath,
signal,
);
if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`);
absolutePdfPath = suffixMatch.absolutePath;
suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath };
}
return readPdfImageMember(
this.session,
this.#autoResizeImages,
absolutePdfPath,
pdfImageMemberPath.pdfPath,
pdfImageMemberPath.member,
suffixResolution,
signal,
);
} }
} }
@@ -1102,8 +1077,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
// Convert document via markit. // Convert document via markit.
const result = await convertFileWithMarkit(absolutePath, signal); const result = await convertFileWithMarkit(absolutePath, signal);
if (result.ok) { if (result.ok) {
const renderedContent = const renderedContent = result.content;
ext === ".pdf" ? rewritePdfImagePlaceholders(result.content, resolvedDisplayPath) : result.content;
// Route the converted markdown through the in-memory text builder // Route the converted markdown through the in-memory text builder
// so line-range selectors (`file.pdf:50-100`, `:5-16,40-80`) and // so line-range selectors (`file.pdf:50-100`, `:5-16,40-80`) and
// raw mode apply against the converted output. Without this, // raw mode apply against the converted output. Without this,
+6 -44
View File
@@ -1,5 +1,5 @@
import * as path from "node:path"; import * as path from "node:path";
import { logger, untilAborted } from "@oh-my-pi/pi-utils"; import { untilAborted } from "@oh-my-pi/pi-utils";
import type { ConversionResult, Markit, StreamInfo } from "../markit"; import type { ConversionResult, Markit, StreamInfo } from "../markit";
import { ToolAbortError } from "../tools/tool-errors"; import { ToolAbortError } from "../tools/tool-errors";
import { import {
@@ -8,7 +8,6 @@ import {
readMarkitConversionCache, readMarkitConversionCache,
writeMarkitConversionCache, writeMarkitConversionCache,
} from "./markit-cache"; } from "./markit-cache";
import { loadEmbeddedMupdfWasm } from "./mupdf-wasm-embed";
/** /**
* File extensions markit can actually convert to markdown — one per registered * File extensions markit can actually convert to markdown — one per registered
@@ -31,53 +30,16 @@ export interface MarkitConversionResult {
export interface MarkitFileConversionOptions { export interface MarkitFileConversionOptions {
/** /**
* Directory the PDF converter writes extracted images/diagrams into. When * Directory converters may use for extracted image or diagram files. Since
* set, each embedded image is rendered to `<id>.png` and referenced by path * those files are conversion side effects, conversions using this option
* in the markdown; when unset, markit emits an `<!-- image: <id> ... -->` * bypass the markdown cache.
* placeholder comment instead.
*/ */
imageDir?: string; imageDir?: string;
} }
interface MuPdfWasmModuleConfig {
print?: (...values: unknown[]) => void;
printErr?: (...values: unknown[]) => void;
wasmBinary?: Uint8Array;
}
function logMuPdfWasmOutput(stream: "stdout" | "stderr", values: unknown[]): void {
const message = values.length === 1 && typeof values[0] === "string" ? values[0] : values.map(String).join(" ");
logger.debug("mupdf wasm output", { stream, message });
}
// `$libmupdf_wasm_Module` is declared globally (as `any`) by the mupdf package.
// Install print hooks before the WASM module initializes so its stdout/stderr
// route to the file logger instead of corrupting the TUI.
function installMuPdfWasmLogger(): void {
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
moduleConfig.print = (...values: unknown[]) => logMuPdfWasmOutput("stdout", values);
moduleConfig.printErr = (...values: unknown[]) => logMuPdfWasmOutput("stderr", values);
globalThis.$libmupdf_wasm_Module = moduleConfig;
}
// Hand the WASM module its bytes directly when the compiled binary embedded them
// (scripts/embed-mupdf-wasm.ts); a single-file binary has no node_modules for
// mupdf to read `mupdf-wasm.wasm` from. Source/npm builds get undefined here and
// mupdf loads its own wasm. Must run before the mupdf module evaluates.
function installEmbeddedMupdfWasm(): void {
const wasmBinary = loadEmbeddedMupdfWasm();
if (!wasmBinary) return;
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
moduleConfig.wasmBinary = wasmBinary;
globalThis.$libmupdf_wasm_Module = moduleConfig;
}
installMuPdfWasmLogger();
let markit: () => Markit | Promise<Markit> = async () => { let markit: () => Markit | Promise<Markit> = async () => {
// Lazy: keep the document engine (mammoth/mupdf) off the startup // Lazy: keep the document engine off the startup import graph — it loads
// import graph — it loads only when a document is first converted. // only when a document is first converted.
installEmbeddedMupdfWasm();
const promise = import("../markit").then(({ Markit }) => { const promise = import("../markit").then(({ Markit }) => {
const instance = new Markit(); const instance = new Markit();
markit = () => instance; markit = () => instance;
@@ -1,12 +0,0 @@
// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
//
// Compiled single-file binaries cannot let mupdf resolve its `mupdf-wasm.wasm`
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
// wasm bytes via `with { type: "file" }` and copies the wasm next to it. Source
// checkouts, `bun test`, and the npm `dist/cli.js` bundle keep mupdf external and
// load the wasm from node_modules, so this placeholder returns undefined and the
// build resets back to it afterward.
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
return undefined;
}
@@ -0,0 +1,35 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import * as piNatives from "@oh-my-pi/pi-natives";
import { PdfConverter } from "../src/markit/converters/pdf";
describe("PdfConverter", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("keeps accepting PDF extensions and MIME types", () => {
const converter = new PdfConverter();
expect(converter.accepts({ extension: ".pdf" })).toBe(true);
expect(converter.accepts({ mimetype: "application/pdf" })).toBe(true);
expect(converter.accepts({ mimetype: "application/pdf; charset=binary" })).toBe(true);
expect(converter.accepts({ mimetype: "application/x-pdf" })).toBe(true);
expect(converter.accepts({ extension: ".txt", mimetype: "text/plain" })).toBe(false);
});
it("returns a browser and OCR notice for an image-only PDF", async () => {
vi.spyOn(piNatives, "pdfToMarkdown").mockResolvedValue({
markdown: "",
pageCount: 3,
pagesNeedingOcr: [1, 3],
hasEncodingIssues: false,
});
const result = await new PdfConverter().convert(Buffer.from("image-only pdf"), { extension: ".pdf" });
expect(result.markdown).toBe(
"Text extraction is incomplete for PDF pages 1, 3. Use the browser tool to render those pages or OCR them.",
);
expect(result.markdown.length).toBeGreaterThan(0);
});
});
@@ -1,447 +0,0 @@
/**
* PDF image extraction: markit emits inert `<!-- image: <id> ... -->`
* placeholders for embedded PDF images. The read tool rewrites those into
* browsable `read <pdf>:<id>.png` handles, and serves the actual PNG when that
* handle is read — extracting via markit's `imageDir` into a session-artifact
* cache. These lock the rewrite, the member extraction, member validation, and
* the caching contract.
*/
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
import * as piUtils from "@oh-my-pi/pi-utils";
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
// 1x1 transparent PNG — small enough to pass through image loading untouched.
const TINY_PNG = Buffer.from(
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg==",
"base64",
);
function makeSession(testDir: string): ToolSession {
const sessionFile = path.join(testDir, "session.jsonl");
const artifactsDir = sessionFile.slice(0, -6);
return {
cwd: testDir,
hasUI: false,
getSessionFile: () => sessionFile,
getArtifactsDir: () => artifactsDir,
getSessionSpawns: () => null,
settings: Settings.isolated({ "images.autoResize": false }),
} as unknown as ToolSession;
}
/** Spy on markit so PDF "extraction" writes the given members into imageDir. */
function mockExtraction(members: Record<string, Buffer> = { "p11-img0.png": TINY_PNG }) {
return vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_filePath: string, _signal, options) => {
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
for (const name in members) {
fs.writeFileSync(path.join(options.imageDir, name), members[name]!);
}
}
return { ok: true, content: "" };
});
}
function imageBytes(result: AgentToolResult<ReadToolDetails>): Buffer {
const image = result.content.find(content => content.type === "image");
if (image?.type !== "image") throw new Error("Expected an image result");
return Buffer.from(image.data, "base64");
}
function mockBlockedExtraction() {
const entered = Promise.withResolvers<void>();
const release = Promise.withResolvers<void>();
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_sourcePath, signal, options) => {
entered.resolve();
await release.promise;
signal?.throwIfAborted();
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
}
return { ok: true, content: "" };
});
return { entered, release, spy };
}
/**
* Resolves once `count` callers are attached as waiters on the shared PDF
* extraction. Waiter attachment is the only `untilAborted` call that receives
* a promise (source snapshots pass thunks), so counting promise arguments
* observes it. The abort tests need this barrier: aborting a caller while it
* is the sole waiter tears the extraction down and deadlocks against the
* blocked conversion mock.
*/
function extractionWaitersAttached(count: number): Promise<void> {
const attached = Promise.withResolvers<void>();
const original = piUtils.untilAborted;
let seen = 0;
vi.spyOn(piUtils, "untilAborted").mockImplementation((signal, pr) => {
if (typeof pr !== "function" && ++seen === count) attached.resolve();
return original(signal, pr);
});
return attached.promise;
}
describe("read PDF image extraction", () => {
let testDir: string;
let pdfPath: string;
beforeEach(() => {
testDir = path.join(os.tmpdir(), `read-pdf-img-${Snowflake.next()}`);
fs.mkdirSync(testDir, { recursive: true });
pdfPath = path.join(testDir, "doc.pdf");
fs.writeFileSync(pdfPath, "%PDF-stub");
});
afterEach(() => {
vi.restoreAllMocks();
removeSyncWithRetries(testDir);
});
it("rewrites image placeholders into browse handles on a full read", async () => {
const converted = [
"Heading",
"",
"<!-- image: p11-img0 (page 11, 199x124pt) -->",
"",
"<!-- image: p11-img1 (page 11, 199x54pt) -->",
"",
"Footer",
].join("\n");
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: converted });
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: pdfPath });
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).not.toContain("<!-- image:");
expect(text).toContain("read `doc.pdf:p11-img0.png`");
expect(text).toContain("read `doc.pdf:p11-img1.png`");
// Page/size metadata is preserved in the handle text.
expect(text).toContain("page 11, 199x124pt");
});
it("rewrites placeholders inside a line-range view", async () => {
const lines = Array.from({ length: 20 }, (_, i) => `pdf line ${i + 1}`);
lines[9] = "<!-- image: p3-img0 (page 3, 100x50pt) -->"; // line 10
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: lines.join("\n") });
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: `${pdfPath}:8-12` });
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).not.toContain("<!-- image:");
expect(text).toContain("read `doc.pdf:p3-img0.png`");
});
it("extracts a PDF image member as an inline image block", async () => {
const spy = mockExtraction();
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
const image = result.content.find(c => c.type === "image");
expect(image).toBeDefined();
expect(image && "mimeType" in image ? image.mimeType : undefined).toBe("image/png");
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).toContain("Read image file");
// Extraction was driven through markit with an imageDir target.
expect(spy).toHaveBeenCalledTimes(1);
expect(spy.mock.calls[0]?.[2]?.imageDir).toBeTruthy();
});
it("reuses the extraction cache across member reads", async () => {
const spy = mockExtraction();
const tool = new ReadTool(makeSession(testDir));
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
// Second read is served from the `.extracted` cache, not re-converted.
expect(spy).toHaveBeenCalledTimes(1);
});
it("re-extracts image members after same-path PDF replacement", async () => {
const sourceA = Buffer.from("%PDF-source-a");
const sourceB = Buffer.from("%PDF-source-b");
fs.writeFileSync(pdfPath, sourceA);
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(
path.join(options.imageDir, "p11-img0.png"),
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
);
}
return { ok: true, content: "" };
});
const tool = new ReadTool(makeSession(testDir));
const originalStat = fs.statSync(pdfPath);
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
fs.writeFileSync(pdfPath, sourceB);
fs.utimesSync(pdfPath, originalStat.atime, originalStat.mtime);
const second = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
expect(imageBytes(first).subarray(TINY_PNG.length)).toEqual(sourceA);
expect(imageBytes(second).subarray(TINY_PNG.length)).toEqual(sourceB);
expect(spy).toHaveBeenCalledTimes(2);
});
it("converts an immutable snapshot when the source changes during extraction", async () => {
const sourceA = Buffer.from("%PDF-source-a");
const sourceB = Buffer.from("%PDF-source-b");
fs.writeFileSync(pdfPath, sourceA);
const entered = Promise.withResolvers<void>();
const release = Promise.withResolvers<void>();
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
entered.resolve();
await release.promise;
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(
path.join(options.imageDir, "p11-img0.png"),
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
);
}
return { ok: true, content: "" };
});
const pending = new ReadTool(makeSession(testDir)).execute("call", { path: `${pdfPath}:p11-img0.png` });
await entered.promise;
fs.writeFileSync(pdfPath, sourceB);
release.resolve();
const result = await pending;
expect(imageBytes(result).subarray(TINY_PNG.length)).toEqual(sourceA);
});
it("coalesces concurrent cold image extraction", async () => {
const { entered, release, spy } = mockBlockedExtraction();
const tool = new ReadTool(makeSession(testDir));
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
const second = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await entered.promise;
const conversionCount = spy.mock.calls.length;
release.resolve();
const [firstResult, secondResult] = await Promise.all([first, second]);
expect(conversionCount).toBe(1);
expect(imageBytes(firstResult)).toEqual(imageBytes(secondResult));
});
it("keeps shared extraction running when its owner aborts", async () => {
const { entered, release, spy } = mockBlockedExtraction();
const bothAttached = extractionWaitersAttached(2);
const tool = new ReadTool(makeSession(testDir));
const ownerController = new AbortController();
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, ownerController.signal);
await entered.promise;
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await bothAttached;
ownerController.abort();
await expect(owner).rejects.toThrow(/Aborted|Cancelled/);
release.resolve();
const result = await joiner;
expect(result.content.some(content => content.type === "image")).toBe(true);
expect(spy).toHaveBeenCalledTimes(1);
});
it("keeps shared extraction running when a joiner aborts", async () => {
const { entered, release, spy } = mockBlockedExtraction();
const bothAttached = extractionWaitersAttached(2);
const tool = new ReadTool(makeSession(testDir));
const joinerController = new AbortController();
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await entered.promise;
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, joinerController.signal);
await bothAttached;
joinerController.abort();
await expect(joiner).rejects.toThrow(/Aborted|Cancelled/);
release.resolve();
const result = await owner;
expect(result.content.some(content => content.type === "image")).toBe(true);
expect(spy).toHaveBeenCalledTimes(1);
});
it("cleans temporary extraction state when the only caller aborts", async () => {
const entered = Promise.withResolvers<void>();
let snapshotPath: string | undefined;
let stagingDir: string | undefined;
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, signal, options) => {
snapshotPath = sourcePath;
stagingDir = options?.imageDir;
entered.resolve();
const aborted = Promise.withResolvers<void>();
const onAbort = () => aborted.resolve();
if (signal?.aborted) onAbort();
else signal?.addEventListener("abort", onAbort, { once: true });
await aborted.promise;
signal?.removeEventListener("abort", onAbort);
signal?.throwIfAborted();
return { ok: true, content: "" };
});
const controller = new AbortController();
const pending = new ReadTool(makeSession(testDir)).execute(
"call",
{ path: `${pdfPath}:p11-img0.png` },
controller.signal,
);
await entered.promise;
controller.abort();
await expect(pending).rejects.toThrow(/Aborted|Cancelled/);
if (!snapshotPath || !stagingDir) throw new Error("Expected extraction paths");
expect(fs.existsSync(path.dirname(snapshotPath))).toBe(false);
expect(fs.existsSync(stagingDir)).toBe(false);
});
it("does not let a failed generation delete a replacement generation", async () => {
const sourceA = Buffer.from("%PDF-source-a");
const sourceB = Buffer.from("%PDF-source-b");
fs.writeFileSync(pdfPath, sourceA);
const firstEntered = Promise.withResolvers<void>();
const failFirst = Promise.withResolvers<void>();
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
const source = fs.readFileSync(sourcePath);
if (source.equals(sourceA)) {
firstEntered.resolve();
await failFirst.promise;
return { ok: false, content: "", error: "generation A failed" };
}
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), Buffer.concat([TINY_PNG, source]));
}
return { ok: true, content: "" };
});
const tool = new ReadTool(makeSession(testDir));
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
await firstEntered.promise;
fs.writeFileSync(pdfPath, sourceB);
const replacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
failFirst.resolve();
await expect(first).rejects.toThrow(/Cannot extract images/);
const cachedReplacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
expect(imageBytes(replacement).subarray(TINY_PNG.length)).toEqual(sourceB);
expect(imageBytes(cachedReplacement)).toEqual(imageBytes(replacement));
expect(spy).toHaveBeenCalledTimes(2);
});
it("isolates equal-content PDFs with the same basename in different directories", async () => {
const otherDir = path.join(testDir, "other");
const otherPdfPath = path.join(otherDir, path.basename(pdfPath));
fs.mkdirSync(otherDir, { recursive: true });
fs.writeFileSync(otherPdfPath, fs.readFileSync(pdfPath));
let conversion = 0;
const spy = vi
.spyOn(markit, "convertFileWithMarkit")
.mockImplementation(async (_sourcePath, _signal, options) => {
conversion++;
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(
path.join(options.imageDir, "p11-img0.png"),
Buffer.concat([TINY_PNG, Buffer.from(String(conversion))]),
);
}
return { ok: true, content: "" };
});
const tool = new ReadTool(makeSession(testDir));
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
const second = await tool.execute("call", { path: `${otherPdfPath}:p11-img0.png` });
expect(imageBytes(first).subarray(TINY_PNG.length).toString()).toBe("1");
expect(imageBytes(second).subarray(TINY_PNG.length).toString()).toBe("2");
expect(spy).toHaveBeenCalledTimes(2);
});
it("supports PDF basenames at the filesystem component limit", async () => {
const longPdfPath = path.join(testDir, `${"a".repeat(250)}.pdf`);
fs.writeFileSync(longPdfPath, "%PDF-stub");
mockExtraction();
const result = await new ReadTool(makeSession(testDir)).execute("call", {
path: `${longPdfPath}:p11-img0.png`,
});
expect(result.content.some(content => content.type === "image")).toBe(true);
});
it("errors with the available members for an unknown member", async () => {
mockExtraction();
const tool = new ReadTool(makeSession(testDir));
await expect(tool.execute("call", { path: `${pdfPath}:does-not-exist.png` })).rejects.toThrow(
/not found.*p11-img0\.png/s,
);
});
it("rejects member traversal attempts", async () => {
mockExtraction();
const tool = new ReadTool(makeSession(testDir));
// `../../escape.png` matches the image-member shape but is not a known
// basename, so it must be refused rather than joined into the cache path.
await expect(tool.execute("call", { path: `${pdfPath}:../../escape.png` })).rejects.toThrow(/not found/);
});
it("lists extractable members for a trailing-colon read", async () => {
mockExtraction({ "p1-img0.png": TINY_PNG, "p2-img0.png": TINY_PNG });
const tool = new ReadTool(makeSession(testDir));
const result = await tool.execute("call", { path: `${pdfPath}:` });
const text = result.content
.filter(c => c.type === "text")
.map(c => c.text)
.join("\n");
expect(text).toContain(`read \`${pdfPath}:p1-img0.png\``);
expect(text).toContain(`read \`${pdfPath}:p2-img0.png\``);
});
it("does not cache a failed conversion", async () => {
let failedSnapshotPath: string | undefined;
let failedImageDir: string | undefined;
const spy = vi.spyOn(markit, "convertFileWithMarkit");
spy.mockImplementationOnce(async (sourcePath, _signal, options) => {
failedSnapshotPath = sourcePath;
failedImageDir = options?.imageDir;
return { ok: false, content: "", error: "boom" };
});
const tool = new ReadTool(makeSession(testDir));
await expect(tool.execute("call", { path: `${pdfPath}:p11-img0.png` })).rejects.toThrow(/Cannot extract images/);
if (!failedSnapshotPath || !failedImageDir) throw new Error("Expected failed extraction paths");
expect(fs.existsSync(path.dirname(failedSnapshotPath))).toBe(false);
expect(fs.existsSync(path.join(failedImageDir, ".extracted"))).toBe(false);
spy.mockImplementationOnce(async (_filePath: string, _signal, options) => {
if (options?.imageDir) {
fs.mkdirSync(options.imageDir, { recursive: true });
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
}
return { ok: true, content: "" };
});
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
expect(result.content.some(c => c.type === "image")).toBe(true);
expect(spy).toHaveBeenCalledTimes(2);
});
});
@@ -0,0 +1,79 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
import { removeWithRetries } from "@oh-my-pi/pi-utils";
function makeSession(cwd: string): ToolSession {
return {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated({ "images.autoResize": false }),
} as ToolSession;
}
function textOf(result: AgentToolResult<ReadToolDetails>): string {
return result.content
.filter(entry => entry.type === "text")
.map(entry => entry.text)
.join("\n");
}
describe("read unsupported PDF image members", () => {
let testDir: string;
let pdfPath: string;
beforeEach(async () => {
testDir = await fs.mkdtemp(path.join(os.tmpdir(), "read-pdf-image-unsupported-"));
pdfPath = path.join(testDir, "doc.pdf");
await fs.writeFile(pdfPath, `%PDF-stub-${testDir}`);
});
afterEach(async () => {
vi.restoreAllMocks();
await removeWithRetries(testDir);
});
it("directs former image listing and PNG member reads to browser rendering", async () => {
const tool = new ReadTool(makeSession(testDir));
for (const readPath of [`${pdfPath}:`, `${pdfPath}:p1-img0.png`]) {
try {
await tool.execute("read-pdf-image", { path: readPath });
throw new Error("Expected the PDF image read to fail");
} catch (error) {
expect(error).toBeInstanceOf(Error);
const message = (error as Error).message;
expect(message).toContain("pdf-inspector cannot render PDF images");
expect(message).toContain("Puppeteer browser tool");
expect(message).toContain(`read '${pdfPath}' for extracted text`);
}
}
});
it("preserves a literal filename that looks like a PDF image listing", async () => {
const literalPath = `${pdfPath}:`;
await fs.writeFile(literalPath, "literal colon path wins\n");
const result = await new ReadTool(makeSession(testDir)).execute("read-literal", { path: literalPath });
expect(textOf(result)).toContain("literal colon path wins");
});
it("routes PDF line selectors through normal document conversion", async () => {
const convert = vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({
ok: true,
content: "first line\nselected line\nthird line\n",
});
const result = await new ReadTool(makeSession(testDir)).execute("read-pdf-lines", { path: `${pdfPath}:2-2` });
expect(convert).toHaveBeenCalledTimes(1);
expect(textOf(result)).toContain("selected line");
});
});
@@ -105,7 +105,7 @@ describe("document conversion cache", () => {
it("skips cache for imageDir conversions", async () => { it("skips cache for imageDir conversions", async () => {
const convert = vi.spyOn(Markit.prototype, "convert").mockResolvedValue({ markdown: "image body" }); const convert = vi.spyOn(Markit.prototype, "convert").mockResolvedValue({ markdown: "image body" });
const docPath = path.join(testDir, "image-doc.pdf"); const docPath = path.join(testDir, "image-doc.docx");
await fs.writeFile(docPath, new TextEncoder().encode("image bytes")); await fs.writeFile(docPath, new TextEncoder().encode("image bytes"));
const imageDir = path.join(testDir, "images"); const imageDir = path.join(testDir, "images");
+4
View File
@@ -2,6 +2,10 @@
## [Unreleased] ## [Unreleased]
### Added
- Added the async `pdfToMarkdown` native API backed by `pdf-inspector`, with page numbering, page-count, OCR-needed-page, and encoding-issue metadata.
### Changed ### Changed
- Docker images (`Dockerfile`, `scripts/install-tests/*.dockerfile`) build the native addon through the cargo/napi-rs backend (`OMP_NATIVE_BUILD_BACKEND=cargo`) instead of Bazel: a single fixed host target gains nothing from hermetic cross toolchains, and none of those images shipped bazelisk. `OMP_NATIVE_CARGO_PROFILE` picks the profile for that path (images use `ci`, local default stays `local`). - Docker images (`Dockerfile`, `scripts/install-tests/*.dockerfile`) build the native addon through the cargo/napi-rs backend (`OMP_NATIVE_BUILD_BACKEND=cargo`) instead of Bazel: a single fixed host target gains nothing from hermetic cross toolchains, and none of those images shipped bazelisk. `OMP_NATIVE_CARGO_PROFILE` picks the profile for that path (images use `ci`, local default stays `local`).
+6 -1
View File
@@ -10,6 +10,7 @@ Native Rust functionality via N-API.
- **Audio**: Cross-platform low-latency microphone capture and gapless speaker playback - **Audio**: Cross-platform low-latency microphone capture and gapless speaker playback
- **WebRTC**: Native Opus media, SDP offer/answer negotiation, and data-channel events for live sessions - **WebRTC**: Native Opus media, SDP offer/answer negotiation, and data-channel events for live sessions
- **File locking**: Process-owned cross-process locks with in-memory kernel names on Linux/Windows and `flock(2)` sidecars on other Unix platforms - **File locking**: Process-owned cross-process locks with in-memory kernel names on Linux/Windows and `flock(2)` sidecars on other Unix platforms
- **PDF**: In-memory PDF-to-Markdown extraction with OCR-page classification via `pdf-inspector`
General-purpose image processing (decode/resize/encode for files and buffers) General-purpose image processing (decode/resize/encode for files and buffers)
lives in [`Bun.Image`](https://bun.com/docs/runtime/image) on the JS side; this lives in [`Bun.Image`](https://bun.com/docs/runtime/image) on the JS side; this
@@ -19,7 +20,7 @@ that terminal protocol.
## Usage ## Usage
```typescript ```typescript
import { grep, find, encodeSixel } from "@oh-my-pi/pi-natives"; import { encodeSixel, grep, pdfToMarkdown } from "@oh-my-pi/pi-natives";
// Grep for a pattern // Grep for a pattern
const results = await grep({ const results = await grep({
@@ -38,6 +39,10 @@ const files = await find({
// SIXEL encode for a terminal cell box (px) // SIXEL encode for a terminal cell box (px)
const sequence = encodeSixel(pngBytes, widthPx, heightPx); const sequence = encodeSixel(pngBytes, widthPx, heightPx);
// Extract PDF text and identify pages that still need OCR
const pdf = await pdfToMarkdown(pdfBytes);
console.log(pdf.markdown, pdf.pagesNeedingOcr);
``` ```
## Building ## Building
+26
View File
@@ -1578,6 +1578,32 @@ export interface PatchHunk {
lines: Array<string> lines: Array<string>
} }
/** Markdown and inspection metadata produced from a PDF document. */
export interface PdfMarkdownResult {
/** Extracted document content in Markdown format. */
markdown: string
/** Document title from PDF metadata, when present. */
title?: string
/** Total number of pages in the document. */
pageCount: number
/** One-indexed page numbers whose content requires OCR. */
pagesNeedingOcr: Array<number>
/** Whether the document contains text encoding problems. */
hasEncodingIssues: boolean
}
/**
* Convert an in-memory PDF to Markdown and return its inspection metadata.
*
* Conversion copies the typed array before dispatch so JavaScript mutation
* cannot race the native worker.
*
* # Errors
* Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
* be parsed or converted.
*/
export declare function pdfToMarkdown(input: Uint8Array): Promise<PdfMarkdownResult>
export interface PointerOptions { export interface PointerOptions {
button?: string button?: string
count?: number count?: number
+1
View File
@@ -69,6 +69,7 @@ export const matchesLegacySequence = nativeBindings.matchesLegacySequence;
export const mmrRerankIndices = nativeBindings.mmrRerankIndices; export const mmrRerankIndices = nativeBindings.mmrRerankIndices;
export const parseKey = nativeBindings.parseKey; export const parseKey = nativeBindings.parseKey;
export const parseKittySequence = nativeBindings.parseKittySequence; export const parseKittySequence = nativeBindings.parseKittySequence;
export const pdfToMarkdown = nativeBindings.pdfToMarkdown;
export const readImageFromClipboard = nativeBindings.readImageFromClipboard; export const readImageFromClipboard = nativeBindings.readImageFromClipboard;
export const renderSnapcompactPng = nativeBindings.renderSnapcompactPng; export const renderSnapcompactPng = nativeBindings.renderSnapcompactPng;
export const search = nativeBindings.search; export const search = nativeBindings.search;
+1
View File
@@ -26,6 +26,7 @@ export interface DetectCompiledBinaryInput {
export function detectCompiledBinary(input: DetectCompiledBinaryInput): boolean; export function detectCompiledBinary(input: DetectCompiledBinaryInput): boolean;
export interface GetAddonFilenamesInput { export interface GetAddonFilenamesInput {
tag: string; tag: string;
arch: string; arch: string;
-1
View File
@@ -87,7 +87,6 @@ export function detectCompiledBinary({ embeddedAddon, env, importMetaUrl }) {
} }
return false; return false;
} }
/** /**
* @param {{ tag: string; arch: string; variant: "modern" | "baseline" | null | undefined }} input * @param {{ tag: string; arch: string; variant: "modern" | "baseline" | null | undefined }} input
* @returns {string[]} * @returns {string[]}
+1 -1
View File
@@ -1,7 +1,7 @@
{ {
"name": "@oh-my-pi/pi-natives", "name": "@oh-my-pi/pi-natives",
"version": "17.3.3", "version": "17.3.3",
"description": "Native Rust bindings for audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "description": "Native Rust bindings for PDF conversion, audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API",
"type": "module", "type": "module",
"homepage": "https://omp.sh", "homepage": "https://omp.sh",
"author": "Can Boluk", "author": "Can Boluk",
+38
View File
@@ -23,6 +23,7 @@ import {
matchesKey, matchesKey,
PtySession, PtySession,
parseKey, parseKey,
pdfToMarkdown,
summarizeCode, summarizeCode,
supportsLanguage, supportsLanguage,
truncateToWidth, truncateToWidth,
@@ -87,6 +88,30 @@ async function createFifo(fifoPath: string) {
throw new Error(await new Response(process.stderr).text()); throw new Error(await new Response(process.stderr).text());
} }
function textPdf(text: string): Uint8Array {
const stream = `BT /F1 12 Tf 72 720 Td (${text}) Tj ET`;
const objects = [
"<< /Type /Catalog /Pages 2 0 R >>",
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
`<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`,
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>",
];
let document = "%PDF-1.4\n";
const offsets: number[] = [];
for (const [index, object] of objects.entries()) {
offsets.push(document.length);
document += `${index + 1} 0 obj\n${object}\nendobj\n`;
}
const xrefOffset = document.length;
document += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
for (const offset of offsets) {
document += `${offset.toString().padStart(10, "0")} 00000 n \n`;
}
document += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xrefOffset}\n%%EOF\n`;
return Buffer.from(document);
}
describe("pi-natives", () => { describe("pi-natives", () => {
beforeAll(async () => { beforeAll(async () => {
await setupFixtures(); await setupFixtures();
@@ -763,6 +788,19 @@ describe("pi-natives", () => {
expect(await Bun.file(markerPath).exists()).toBe(false); expect(await Bun.file(markerPath).exists()).toBe(false);
}); });
}); });
describe("pdfToMarkdown", () => {
it("isolates blocking conversion from later JavaScript buffer mutation", async () => {
const input = textPdf("Copied PDF bytes");
const conversion = pdfToMarkdown(input);
input.fill(0);
const result = await conversion;
expect(result.pageCount).toBe(1);
expect(result.markdown).toContain("Copied PDF bytes");
});
});
describe("htmlToMarkdown", () => { describe("htmlToMarkdown", () => {
it("should convert basic HTML to markdown", async () => { it("should convert basic HTML to markdown", async () => {
const html = "<h1>Hello World</h1><p>This is a paragraph.</p>"; const html = "<h1>Hello World</h1><p>This is a paragraph.</p>";
-4
View File
@@ -158,24 +158,20 @@ async function generateBundle(): Promise<void> {
if (isDryRun) { if (isDryRun) {
console.log("DRY RUN bun run gen:stats"); console.log("DRY RUN bun run gen:stats");
console.log("DRY RUN bun --cwd=packages/collab-web run gen:tool-views"); console.log("DRY RUN bun --cwd=packages/collab-web run gen:tool-views");
console.log("DRY RUN bun run gen:mupdf");
return; return;
} }
await runCommand(["bun", "run", "gen:stats"], repoRoot); await runCommand(["bun", "run", "gen:stats"], repoRoot);
await runCommand(["bun", "--cwd=packages/collab-web", "run", "gen:tool-views"], repoRoot); await runCommand(["bun", "--cwd=packages/collab-web", "run", "gen:tool-views"], repoRoot);
await runCommand(["bun", "run", "gen:mupdf"], repoRoot);
} }
async function resetArtifacts(): Promise<void> { async function resetArtifacts(): Promise<void> {
if (isDryRun) { if (isDryRun) {
console.log("DRY RUN bun run gen:native:reset"); console.log("DRY RUN bun run gen:native:reset");
console.log("DRY RUN bun run gen:stats:reset"); console.log("DRY RUN bun run gen:stats:reset");
console.log("DRY RUN bun run gen:mupdf:reset");
return; return;
} }
await runCommand(["bun", "run", "gen:native:reset"], repoRoot); await runCommand(["bun", "run", "gen:native:reset"], repoRoot);
await runCommand(["bun", "run", "gen:stats:reset"], repoRoot); await runCommand(["bun", "run", "gen:stats:reset"], repoRoot);
await runCommand(["bun", "run", "gen:mupdf:reset"], repoRoot);
} }
async function main(): Promise<void> { async function main(): Promise<void> {