feat: replaced custom mupdf wasm pipeline with native function
- Replaced the custom MuPDF-WASM PDF extraction and rendering pipeline with the new `pdfToMarkdown` native function from `@oh-my-pi/pi-natives`. - Removed legacy MuPDF extraction modules, WASM embedding scripts, and PDF image extraction tools. - Added OCR warnings and browser/text redirection for unsupported PDF image reads. - Updated native package definitions, documentation, and test suites for the new PDF inspection capability.
This commit is contained in:
@@ -162,11 +162,11 @@ jobs:
|
|||||||
bazelisk --bazelrc="${{ steps.cache.outputs.rc }}" test //crates/...
|
bazelisk --bazelrc="${{ steps.cache.outputs.rc }}" test //crates/...
|
||||||
# Clippy scope mirrors `cargo clippy --workspace` (libraries only, no
|
# Clippy scope mirrors `cargo clippy --workspace` (libraries only, no
|
||||||
# test targets) plus the strict/default split: crates with
|
# test targets) plus the strict/default split: crates with
|
||||||
# `[lints] workspace = true` get the workspace policy, the vendored
|
# `[lints] workspace = true` get the workspace policy, except
|
||||||
# brush-core fork is exempt (same as run-rs-task.ts's cargo excludes).
|
# brush-core (a vendored fork excluded from the Cargo task too).
|
||||||
# pi-builtins allows every clippy group in its own manifest (ported
|
# pi-builtins allows every clippy group in its own manifest (ported
|
||||||
# brush/uutils/jaq code) but is still held to zero rustc warnings;
|
# brush/uutils/jaq code) but is still held to zero rustc warnings;
|
||||||
# cargo honors that via `[lints]`, bazel via the clippy-ported config.
|
# Cargo honors that via `[lints]`, Bazel via the clippy-ported config.
|
||||||
- name: Clippy (workspace lint policy on opted-in crates)
|
- name: Clippy (workspace lint policy on opted-in crates)
|
||||||
run: |
|
run: |
|
||||||
bazelisk query "kind('rust_library|rust_shared_library', //crates/pi-ast/... + //crates/pi-iso/... + //crates/pi-natives/... + //crates/pi-shell/... + //crates/pi-voice/... + //crates/pi-walker/...)" \
|
bazelisk query "kind('rust_library|rust_shared_library', //crates/pi-ast/... + //crates/pi-iso/... + //crates/pi-natives/... + //crates/pi-shell/... + //crates/pi-voice/... + //crates/pi-walker/...)" \
|
||||||
@@ -411,12 +411,6 @@ jobs:
|
|||||||
- name: Test coding-agent native/unit bucket
|
- name: Test coding-agent native/unit bucket
|
||||||
env:
|
env:
|
||||||
OMP_TEST_CONCURRENCY: "4"
|
OMP_TEST_CONCURRENCY: "4"
|
||||||
# The mupdf/PDF-extraction chunk measures ~7 min on burstable
|
|
||||||
# runners under a full 8-wide fan-out; the default 600 s chunk
|
|
||||||
# watchdog SIGKILLed it (release run 30519992654). The watchdog
|
|
||||||
# exists to catch wedged children, not slow-but-progressing
|
|
||||||
# chunks — give this bucket a wider budget.
|
|
||||||
OMP_TEST_CHUNK_TIMEOUT: "1200"
|
|
||||||
run: bun run ci:test:coding-agent:native
|
run: bun run ci:test:coding-agent:native
|
||||||
|
|
||||||
test_smoke:
|
test_smoke:
|
||||||
|
|||||||
@@ -64,7 +64,6 @@ pi-*.html
|
|||||||
# Generated files
|
# Generated files
|
||||||
packages/coding-agent/src/export/html/tool-views.generated.js
|
packages/coding-agent/src/export/html/tool-views.generated.js
|
||||||
packages/natives/npm/
|
packages/natives/npm/
|
||||||
packages/coding-agent/src/utils/mupdf-wasm.wasm
|
|
||||||
/runs/
|
/runs/
|
||||||
python/omp-rpc/src/omp_rpc.egg-info/
|
python/omp-rpc/src/omp_rpc.egg-info/
|
||||||
python/omp-rpc/build/
|
python/omp-rpc/build/
|
||||||
|
|||||||
Generated
+139
@@ -1868,6 +1868,15 @@ version = "1.0.20"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
|
checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "ecb"
|
||||||
|
version = "0.1.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1a8bfa975b1aec2145850fcaa1c6fe269a16578c44705a532ae3edc92b8881c7"
|
||||||
|
dependencies = [
|
||||||
|
"cipher",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "ecdsa"
|
name = "ecdsa"
|
||||||
version = "0.16.9"
|
version = "0.16.9"
|
||||||
@@ -1979,6 +1988,29 @@ dependencies = [
|
|||||||
"syn 2.0.119",
|
"syn 2.0.119",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "env_filter"
|
||||||
|
version = "2.0.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "900d271a03799a1ee8d1ca9b19893b48ca674a9284fefcfb85f05e74ed314217"
|
||||||
|
dependencies = [
|
||||||
|
"log",
|
||||||
|
"regex",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "env_logger"
|
||||||
|
version = "0.11.11"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "de671bd27a75a797dc9ae289ba1e77276e75e2026408aab65185384e2d5cd3f6"
|
||||||
|
dependencies = [
|
||||||
|
"anstream",
|
||||||
|
"anstyle",
|
||||||
|
"env_filter",
|
||||||
|
"jiff",
|
||||||
|
"log",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "equivalent"
|
name = "equivalent"
|
||||||
version = "1.0.2"
|
version = "1.0.2"
|
||||||
@@ -3152,6 +3184,25 @@ dependencies = [
|
|||||||
"quick-error",
|
"quick-error",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "include_dir"
|
||||||
|
version = "0.7.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
|
||||||
|
dependencies = [
|
||||||
|
"include_dir_macros",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "include_dir_macros"
|
||||||
|
version = "0.7.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
|
||||||
|
dependencies = [
|
||||||
|
"proc-macro2",
|
||||||
|
"quote",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "indenter"
|
name = "indenter"
|
||||||
version = "0.3.4"
|
version = "0.3.4"
|
||||||
@@ -3640,6 +3691,37 @@ version = "0.4.33"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
|
checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "lopdf"
|
||||||
|
version = "0.42.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||||
|
dependencies = [
|
||||||
|
"aes",
|
||||||
|
"bitflags 2.13.1",
|
||||||
|
"cbc",
|
||||||
|
"chrono",
|
||||||
|
"ecb",
|
||||||
|
"encoding_rs",
|
||||||
|
"flate2",
|
||||||
|
"getrandom 0.4.3",
|
||||||
|
"indexmap",
|
||||||
|
"itoa",
|
||||||
|
"jiff",
|
||||||
|
"log",
|
||||||
|
"md-5",
|
||||||
|
"nom 8.0.0",
|
||||||
|
"rand 0.10.2",
|
||||||
|
"rangemap",
|
||||||
|
"rayon",
|
||||||
|
"sha2",
|
||||||
|
"stringprep",
|
||||||
|
"thiserror 2.0.20",
|
||||||
|
"time",
|
||||||
|
"ttf-parser",
|
||||||
|
"weezl",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "lscolors"
|
name = "lscolors"
|
||||||
version = "0.21.0"
|
version = "0.21.0"
|
||||||
@@ -4539,6 +4621,24 @@ dependencies = [
|
|||||||
"pkg-config",
|
"pkg-config",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "pdf-inspector"
|
||||||
|
version = "1.14.2"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "1e024ae242c514e2adf6aee186678e0eabdc2e5ecfbb2159186881b4498593cb"
|
||||||
|
dependencies = [
|
||||||
|
"env_logger",
|
||||||
|
"include_dir",
|
||||||
|
"log",
|
||||||
|
"lopdf",
|
||||||
|
"once_cell",
|
||||||
|
"rayon",
|
||||||
|
"regex",
|
||||||
|
"thiserror 2.0.20",
|
||||||
|
"ttf-parser",
|
||||||
|
"unicode-normalization",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "peg"
|
name = "peg"
|
||||||
version = "0.8.6"
|
version = "0.8.6"
|
||||||
@@ -4982,6 +5082,7 @@ dependencies = [
|
|||||||
"objc2-core-graphics",
|
"objc2-core-graphics",
|
||||||
"objc2-foundation",
|
"objc2-foundation",
|
||||||
"parking_lot",
|
"parking_lot",
|
||||||
|
"pdf-inspector",
|
||||||
"phf 0.13.1",
|
"phf 0.13.1",
|
||||||
"pi-ast",
|
"pi-ast",
|
||||||
"pi-iso",
|
"pi-iso",
|
||||||
@@ -5505,6 +5606,12 @@ dependencies = [
|
|||||||
"rand_core 0.10.1",
|
"rand_core 0.10.1",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "rangemap"
|
||||||
|
version = "1.8.0"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "a611d15b50743feb4c76b7d03edcb0e64f399c26961e4efe6975bc398be6aa3d"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "rayon"
|
name = "rayon"
|
||||||
version = "1.12.0"
|
version = "1.12.0"
|
||||||
@@ -6210,6 +6317,17 @@ dependencies = [
|
|||||||
"quote",
|
"quote",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "stringprep"
|
||||||
|
version = "0.1.5"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "7b4df3d392d81bd458a8a621b8bffbd2302a12ffe288a9d931670948749463b1"
|
||||||
|
dependencies = [
|
||||||
|
"unicode-bidi",
|
||||||
|
"unicode-normalization",
|
||||||
|
"unicode-properties",
|
||||||
|
]
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "strsim"
|
name = "strsim"
|
||||||
version = "0.11.1"
|
version = "0.11.1"
|
||||||
@@ -7375,12 +7493,33 @@ version = "2.9.0"
|
|||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "dbc4bc3a9f746d862c45cb89d705aa10f187bb96c76001afab07a0d35ce60142"
|
checksum = "dbc4bc3a9f746d862c45cb89d705aa10f187bb96c76001afab07a0d35ce60142"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-bidi"
|
||||||
|
version = "0.3.18"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "5c1cb5db39152898a79168971543b1cb5020dff7fe43c8dc468b0885f5e29df5"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "unicode-ident"
|
name = "unicode-ident"
|
||||||
version = "1.0.24"
|
version = "1.0.24"
|
||||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
|
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-normalization"
|
||||||
|
version = "0.1.25"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8"
|
||||||
|
dependencies = [
|
||||||
|
"tinyvec",
|
||||||
|
]
|
||||||
|
|
||||||
|
[[package]]
|
||||||
|
name = "unicode-properties"
|
||||||
|
version = "0.1.4"
|
||||||
|
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||||
|
checksum = "7df058c713841ad818f1dc5d3fd88063241cc61f49f5fbea4b951e8cf5a8d71d"
|
||||||
|
|
||||||
[[package]]
|
[[package]]
|
||||||
name = "unicode-segmentation"
|
name = "unicode-segmentation"
|
||||||
version = "1.13.3"
|
version = "1.13.3"
|
||||||
|
|||||||
@@ -229,6 +229,7 @@ clap = { version = "4", features = ["derive"] }
|
|||||||
# ──────────────────────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────────────────────
|
||||||
# Text Processing & Parsing
|
# Text Processing & Parsing
|
||||||
# ──────────────────────────────────────────────────────────────────────────────
|
# ──────────────────────────────────────────────────────────────────────────────
|
||||||
|
pdf-inspector = "1"
|
||||||
regex = "1"
|
regex = "1"
|
||||||
similar = "3.1.0"
|
similar = "3.1.0"
|
||||||
unicode-segmentation = "1.13"
|
unicode-segmentation = "1.13"
|
||||||
|
|||||||
Generated
+596
-425
File diff suppressed because one or more lines are too long
@@ -105,7 +105,6 @@
|
|||||||
"@opentelemetry/sdk-metrics": "catalog:",
|
"@opentelemetry/sdk-metrics": "catalog:",
|
||||||
"@opentelemetry/sdk-trace-base": "catalog:",
|
"@opentelemetry/sdk-trace-base": "catalog:",
|
||||||
"@opentelemetry/sdk-trace-node": "catalog:",
|
"@opentelemetry/sdk-trace-node": "catalog:",
|
||||||
"mupdf": "catalog:",
|
|
||||||
"puppeteer-core": "catalog:",
|
"puppeteer-core": "catalog:",
|
||||||
},
|
},
|
||||||
"devDependencies": {
|
"devDependencies": {
|
||||||
@@ -386,7 +385,6 @@
|
|||||||
"ghostty-web": "^0.4.0",
|
"ghostty-web": "^0.4.0",
|
||||||
"lint-staged": "^17.0.8",
|
"lint-staged": "^17.0.8",
|
||||||
"lucide-react": "^1.24.0",
|
"lucide-react": "^1.24.0",
|
||||||
"mupdf": "^1.28.0",
|
|
||||||
"onnxruntime-node": "1.26.0",
|
"onnxruntime-node": "1.26.0",
|
||||||
"postcss": "^8.5.16",
|
"postcss": "^8.5.16",
|
||||||
"prettier": "^3.9.5",
|
"prettier": "^3.9.5",
|
||||||
@@ -1219,8 +1217,6 @@
|
|||||||
|
|
||||||
"ms": ["ms@2.1.3", "", {}, "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="],
|
"ms": ["ms@2.1.3", "", {}, "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="],
|
||||||
|
|
||||||
"mupdf": ["mupdf@1.28.0", "", {}, "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg=="],
|
|
||||||
|
|
||||||
"mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="],
|
"mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="],
|
||||||
|
|
||||||
"nanoid": ["nanoid@3.3.18", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w=="],
|
"nanoid": ["nanoid@3.3.18", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w=="],
|
||||||
|
|||||||
@@ -44,6 +44,7 @@ inferno.workspace = true
|
|||||||
napi.workspace = true
|
napi.workspace = true
|
||||||
napi-derive.workspace = true
|
napi-derive.workspace = true
|
||||||
parking_lot.workspace = true
|
parking_lot.workspace = true
|
||||||
|
pdf-inspector.workspace = true
|
||||||
phf.workspace = true
|
phf.workspace = true
|
||||||
flume.workspace = true
|
flume.workspace = true
|
||||||
pi-ast.workspace = true
|
pi-ast.workspace = true
|
||||||
|
|||||||
@@ -2,7 +2,7 @@
|
|||||||
//!
|
//!
|
||||||
//! # Overview
|
//! # Overview
|
||||||
//! High-performance primitives for clipboard access, grep, file discovery,
|
//! High-performance primitives for clipboard access, grep, file discovery,
|
||||||
//! ANSI-aware text measurement, syntax highlighting, HTML-to-Markdown
|
//! ANSI-aware text measurement, syntax highlighting, HTML/PDF-to-Markdown
|
||||||
//! conversion, and terminal SIXEL encoding.
|
//! conversion, and terminal SIXEL encoding.
|
||||||
//!
|
//!
|
||||||
//! # Example
|
//! # Example
|
||||||
@@ -15,7 +15,7 @@
|
|||||||
//!
|
//!
|
||||||
//! # Architecture
|
//! # Architecture
|
||||||
//! ```text
|
//! ```text
|
||||||
//! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/highlight/sixel/text)
|
//! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/pdf/highlight/sixel/text)
|
||||||
//! ```
|
//! ```
|
||||||
|
|
||||||
#![allow(clippy::trailing_empty_array, reason = "generated by napi macro")]
|
#![allow(clippy::trailing_empty_array, reason = "generated by napi macro")]
|
||||||
@@ -41,6 +41,8 @@ pub mod html;
|
|||||||
pub mod iofs;
|
pub mod iofs;
|
||||||
pub mod keys;
|
pub mod keys;
|
||||||
pub mod live;
|
pub mod live;
|
||||||
|
/// PDF inspection and Markdown conversion.
|
||||||
|
pub mod pdf;
|
||||||
pub mod sixel;
|
pub mod sixel;
|
||||||
pub mod snapcompact;
|
pub mod snapcompact;
|
||||||
pub use pi_ast::language;
|
pub use pi_ast::language;
|
||||||
|
|||||||
@@ -0,0 +1,165 @@
|
|||||||
|
//! PDF inspection and Markdown conversion backed by `pdf-inspector`.
|
||||||
|
|
||||||
|
use napi::{Result, bindgen_prelude::Uint8Array};
|
||||||
|
use napi_derive::napi;
|
||||||
|
use pdf_inspector::{MarkdownOptions, PdfOptions, process_pdf_mem_with_options};
|
||||||
|
|
||||||
|
use crate::task;
|
||||||
|
|
||||||
|
/// Markdown and inspection metadata produced from a PDF document.
|
||||||
|
#[napi(object)]
|
||||||
|
pub struct PdfMarkdownResult {
|
||||||
|
/// Extracted document content in Markdown format.
|
||||||
|
pub markdown: String,
|
||||||
|
/// Document title from PDF metadata, when present.
|
||||||
|
pub title: Option<String>,
|
||||||
|
/// Total number of pages in the document.
|
||||||
|
pub page_count: u32,
|
||||||
|
/// One-indexed page numbers whose content requires OCR.
|
||||||
|
pub pages_needing_ocr: Vec<u32>,
|
||||||
|
/// Whether the document contains text encoding problems.
|
||||||
|
pub has_encoding_issues: bool,
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Convert an in-memory PDF to Markdown and return its inspection metadata.
|
||||||
|
///
|
||||||
|
/// Conversion copies the typed array before dispatch so JavaScript mutation
|
||||||
|
/// cannot race the native worker.
|
||||||
|
///
|
||||||
|
/// # Errors
|
||||||
|
/// Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
|
||||||
|
/// be parsed or converted.
|
||||||
|
#[napi(js_name = "pdfToMarkdown")]
|
||||||
|
pub fn pdf_to_markdown(input: Uint8Array) -> task::Promise<PdfMarkdownResult> {
|
||||||
|
let input = input.to_vec();
|
||||||
|
task::blocking("pdf.to_markdown", (), move |_| convert_pdf(&input))
|
||||||
|
}
|
||||||
|
|
||||||
|
fn convert_pdf(input: &[u8]) -> Result<PdfMarkdownResult> {
|
||||||
|
let options = PdfOptions::new()
|
||||||
|
.markdown(MarkdownOptions { include_page_numbers: true, ..Default::default() });
|
||||||
|
let converted = process_pdf_mem_with_options(input, options)
|
||||||
|
.map_err(|error| napi::Error::from_reason(format!("PDF conversion failed: {error}")))?;
|
||||||
|
let markdown = match converted.markdown {
|
||||||
|
Some(markdown) => markdown,
|
||||||
|
None if !converted.pages_needing_ocr.is_empty() => String::new(),
|
||||||
|
None => {
|
||||||
|
return Err(napi::Error::from_reason(
|
||||||
|
"PDF conversion failed: converter returned no Markdown",
|
||||||
|
));
|
||||||
|
},
|
||||||
|
};
|
||||||
|
|
||||||
|
Ok(PdfMarkdownResult {
|
||||||
|
markdown,
|
||||||
|
title: converted.title,
|
||||||
|
page_count: converted.page_count,
|
||||||
|
pages_needing_ocr: converted.pages_needing_ocr,
|
||||||
|
has_encoding_issues: converted.has_encoding_issues,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
fn pdf_fixture(page_contents: &[&str], title: Option<&str>) -> Vec<u8> {
|
||||||
|
let font_id = 3 + page_contents.len() * 2;
|
||||||
|
let info_id = title.map(|_| font_id + 1);
|
||||||
|
let mut objects = Vec::with_capacity(font_id + usize::from(info_id.is_some()));
|
||||||
|
objects.push("<< /Type /Catalog /Pages 2 0 R >>".to_string());
|
||||||
|
|
||||||
|
let kids = (0..page_contents.len())
|
||||||
|
.map(|index| format!("{} 0 R", 3 + index * 2))
|
||||||
|
.collect::<Vec<_>>()
|
||||||
|
.join(" ");
|
||||||
|
objects.push(format!("<< /Type /Pages /Kids [{kids}] /Count {} >>", page_contents.len()));
|
||||||
|
|
||||||
|
for (index, content) in page_contents.iter().enumerate() {
|
||||||
|
let content_id = 4 + index * 2;
|
||||||
|
objects.push(format!(
|
||||||
|
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 \
|
||||||
|
{font_id} 0 R >> >> /Contents {content_id} 0 R >>"
|
||||||
|
));
|
||||||
|
objects.push(format!("<< /Length {} >>\nstream\n{content}\nendstream", content.len()));
|
||||||
|
}
|
||||||
|
|
||||||
|
objects.push(
|
||||||
|
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>"
|
||||||
|
.to_string(),
|
||||||
|
);
|
||||||
|
if let Some(title) = title {
|
||||||
|
objects.push(format!("<< /Title ({title}) >>"));
|
||||||
|
}
|
||||||
|
|
||||||
|
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||||
|
let mut offsets = Vec::with_capacity(objects.len());
|
||||||
|
for (index, object) in objects.iter().enumerate() {
|
||||||
|
offsets.push(pdf.len());
|
||||||
|
pdf.extend_from_slice(format!("{} 0 obj\n{object}\nendobj\n", index + 1).as_bytes());
|
||||||
|
}
|
||||||
|
|
||||||
|
let xref_offset = pdf.len();
|
||||||
|
pdf.extend_from_slice(
|
||||||
|
format!("xref\n0 {}\n0000000000 65535 f \n", objects.len() + 1).as_bytes(),
|
||||||
|
);
|
||||||
|
for offset in offsets {
|
||||||
|
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||||
|
}
|
||||||
|
let info = info_id.map_or_else(String::new, |id| format!(" /Info {id} 0 R"));
|
||||||
|
pdf.extend_from_slice(
|
||||||
|
format!(
|
||||||
|
"trailer\n<< /Size {} /Root 1 0 R{info} >>\nstartxref\n{xref_offset}\n%%EOF\n",
|
||||||
|
objects.len() + 1
|
||||||
|
)
|
||||||
|
.as_bytes(),
|
||||||
|
);
|
||||||
|
pdf
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn converts_text_title_and_page_markers() {
|
||||||
|
let pdf = pdf_fixture(
|
||||||
|
&[
|
||||||
|
"BT /F1 12 Tf 72 720 Td (First page text) Tj 0 -18 Td (More first page text) Tj 0 -18 \
|
||||||
|
Td (End first page) Tj ET",
|
||||||
|
"BT /F1 12 Tf 72 720 Td (Second page text) Tj 0 -18 Td (More second page text) Tj 0 \
|
||||||
|
-18 Td (End second page) Tj ET",
|
||||||
|
],
|
||||||
|
Some("Fixture Title"),
|
||||||
|
);
|
||||||
|
|
||||||
|
let result = convert_pdf(&pdf).expect("fixture PDF should convert");
|
||||||
|
|
||||||
|
assert_eq!(result.title.as_deref(), Some("Fixture Title"));
|
||||||
|
assert_eq!(result.page_count, 2);
|
||||||
|
assert!(result.markdown.contains("First page text"), "{}", result.markdown);
|
||||||
|
assert!(result.markdown.contains("Second page text"), "{}", result.markdown);
|
||||||
|
assert!(result.markdown.contains("<!-- Page 1 -->"), "{}", result.markdown);
|
||||||
|
assert!(result.markdown.contains("<!-- Page 2 -->"), "{}", result.markdown);
|
||||||
|
assert!(!result.has_encoding_issues);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn reports_empty_pages_as_needing_ocr() {
|
||||||
|
let pdf = pdf_fixture(&[""], None);
|
||||||
|
|
||||||
|
let result = convert_pdf(&pdf).expect("empty-page PDF should still convert");
|
||||||
|
|
||||||
|
assert_eq!(result.page_count, 1);
|
||||||
|
assert_eq!(result.pages_needing_ocr, vec![1]);
|
||||||
|
}
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn prefixes_malformed_pdf_errors() {
|
||||||
|
let error = convert_pdf(b"not a PDF")
|
||||||
|
.err()
|
||||||
|
.expect("malformed input should fail");
|
||||||
|
|
||||||
|
assert!(
|
||||||
|
error.reason.starts_with("PDF conversion failed:"),
|
||||||
|
"unexpected error: {}",
|
||||||
|
error.reason
|
||||||
|
);
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -1738,10 +1738,6 @@
|
|||||||
url = "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz";
|
url = "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz";
|
||||||
hash = "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==";
|
hash = "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==";
|
||||||
};
|
};
|
||||||
"mupdf@1.28.0" = fetchurl {
|
|
||||||
url = "https://registry.npmjs.org/mupdf/-/mupdf-1.28.0.tgz";
|
|
||||||
hash = "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg==";
|
|
||||||
};
|
|
||||||
"mute-stream@3.0.0" = fetchurl {
|
"mute-stream@3.0.0" = fetchurl {
|
||||||
url = "https://registry.npmjs.org/mute-stream/-/mute-stream-3.0.0.tgz";
|
url = "https://registry.npmjs.org/mute-stream/-/mute-stream-3.0.0.tgz";
|
||||||
hash = "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw==";
|
hash = "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw==";
|
||||||
|
|||||||
@@ -61,7 +61,6 @@
|
|||||||
"ghostty-web": "^0.4.0",
|
"ghostty-web": "^0.4.0",
|
||||||
"lint-staged": "^17.0.8",
|
"lint-staged": "^17.0.8",
|
||||||
"lucide-react": "^1.24.0",
|
"lucide-react": "^1.24.0",
|
||||||
"mupdf": "^1.28.0",
|
|
||||||
"onnxruntime-node": "1.26.0",
|
"onnxruntime-node": "1.26.0",
|
||||||
"postcss": "^8.5.16",
|
"postcss": "^8.5.16",
|
||||||
"prettier": "^3.9.5",
|
"prettier": "^3.9.5",
|
||||||
@@ -170,8 +169,6 @@
|
|||||||
"gen:nix": "bun scripts/gen-nix-bun.ts",
|
"gen:nix": "bun scripts/gen-nix-bun.ts",
|
||||||
"gen:tool-views": "bun --cwd=packages/collab-web run gen:tool-views",
|
"gen:tool-views": "bun --cwd=packages/collab-web run gen:tool-views",
|
||||||
"gen:bundle": "bun --cwd=packages/coding-agent run gen:bundle",
|
"gen:bundle": "bun --cwd=packages/coding-agent run gen:bundle",
|
||||||
"gen:mupdf": "bun --cwd=packages/coding-agent run gen:mupdf",
|
|
||||||
"gen:mupdf:reset": "bun --cwd=packages/coding-agent run gen:mupdf:reset",
|
|
||||||
"gen:native": "bun --cwd=packages/natives run gen:native",
|
"gen:native": "bun --cwd=packages/natives run gen:native",
|
||||||
"gen:native:reset": "bun --cwd=packages/natives run gen:native:reset",
|
"gen:native:reset": "bun --cwd=packages/natives run gen:native:reset",
|
||||||
"check-spoofed-versions": "bun scripts/check-spoofed-versions.ts"
|
"check-spoofed-versions": "bun scripts/check-spoofed-versions.ts"
|
||||||
|
|||||||
@@ -2,6 +2,11 @@
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Changed
|
||||||
|
|
||||||
|
- Replaced the MuPDF-WASM PDF document backend with `pdf-inspector` through `@oh-my-pi/pi-natives`, preserving cached text conversion and PDF line selectors while reporting pages that need OCR.
|
||||||
|
- Removed `read <pdf>:` image listings and `read <pdf>:<image>.png` extraction because `pdf-inspector` does not rasterize pages; these reads now direct users to the Puppeteer browser tool for rendering or to read the PDF path for extracted text.
|
||||||
|
|
||||||
## [17.3.3] - 2026-08-14
|
## [17.3.3] - 2026-08-14
|
||||||
|
|
||||||
### Fixed
|
### Fixed
|
||||||
|
|||||||
@@ -41,8 +41,6 @@
|
|||||||
"format-prompts": "bun scripts/format-prompts.ts",
|
"format-prompts": "bun scripts/format-prompts.ts",
|
||||||
"gen:tool-views": "bun --cwd=../collab-web run gen:tool-views",
|
"gen:tool-views": "bun --cwd=../collab-web run gen:tool-views",
|
||||||
"gen:bundle": "bun scripts/bundle-dist.ts",
|
"gen:bundle": "bun scripts/bundle-dist.ts",
|
||||||
"gen:mupdf": "bun scripts/embed-mupdf-wasm.ts --generate",
|
|
||||||
"gen:mupdf:reset": "bun scripts/embed-mupdf-wasm.ts --reset",
|
|
||||||
"gen:native": "bun --cwd=../natives run gen:native",
|
"gen:native": "bun --cwd=../natives run gen:native",
|
||||||
"gen:native:reset": "bun --cwd=../natives run gen:native:reset",
|
"gen:native:reset": "bun --cwd=../natives run gen:native:reset",
|
||||||
"prepack": "bun run gen:tool-views && bun run gen:bundle",
|
"prepack": "bun run gen:tool-views && bun run gen:bundle",
|
||||||
@@ -73,7 +71,6 @@
|
|||||||
"@opentelemetry/sdk-metrics": "catalog:",
|
"@opentelemetry/sdk-metrics": "catalog:",
|
||||||
"@opentelemetry/sdk-trace-base": "catalog:",
|
"@opentelemetry/sdk-trace-base": "catalog:",
|
||||||
"@opentelemetry/sdk-trace-node": "catalog:",
|
"@opentelemetry/sdk-trace-node": "catalog:",
|
||||||
"mupdf": "catalog:",
|
|
||||||
"puppeteer-core": "catalog:"
|
"puppeteer-core": "catalog:"
|
||||||
},
|
},
|
||||||
"optionalDependencies": {
|
"optionalDependencies": {
|
||||||
|
|||||||
@@ -88,7 +88,6 @@ async function main(): Promise<void> {
|
|||||||
["bun", "--cwd=../natives", "run", "gen:native"],
|
["bun", "--cwd=../natives", "run", "gen:native"],
|
||||||
crossBuild ? { ...Bun.env, TARGET_PLATFORM: crossBuild.platform, TARGET_ARCH: crossBuild.arch } : Bun.env,
|
crossBuild ? { ...Bun.env, TARGET_PLATFORM: crossBuild.platform, TARGET_ARCH: crossBuild.arch } : Bun.env,
|
||||||
);
|
);
|
||||||
await runCommand(["bun", "run", "gen:mupdf"]);
|
|
||||||
try {
|
try {
|
||||||
await compileCodingAgent({
|
await compileCodingAgent({
|
||||||
repoRoot,
|
repoRoot,
|
||||||
@@ -104,7 +103,6 @@ async function main(): Promise<void> {
|
|||||||
await runCommand(["codesign", "--force", "--sign", "-", outputPath]);
|
await runCommand(["codesign", "--force", "--sign", "-", outputPath]);
|
||||||
}
|
}
|
||||||
} finally {
|
} finally {
|
||||||
await runCommand(["bun", "run", "gen:mupdf:reset"]);
|
|
||||||
await runCommand(["bun", "--cwd=../natives", "run", "gen:native:reset"]);
|
await runCommand(["bun", "--cwd=../natives", "run", "gen:native:reset"]);
|
||||||
}
|
}
|
||||||
} finally {
|
} finally {
|
||||||
|
|||||||
@@ -15,7 +15,6 @@ const legacyHtmlExportAssetPattern = /^(?:template-[^.]+\.(?:css|html|js)|tool-v
|
|||||||
// `omp-legacy-pi-modules` exists only in compiled binaries via the build plugin;
|
// `omp-legacy-pi-modules` exists only in compiled binaries via the build plugin;
|
||||||
// the npm bundle never executes that `isCompiledBinary()` branch.
|
// the npm bundle never executes that `isCompiledBinary()` branch.
|
||||||
const ALWAYS_EXTERNAL = [
|
const ALWAYS_EXTERNAL = [
|
||||||
"mupdf",
|
|
||||||
"@oh-my-pi/pi-natives",
|
"@oh-my-pi/pi-natives",
|
||||||
"@huggingface/transformers",
|
"@huggingface/transformers",
|
||||||
"fastembed",
|
"fastembed",
|
||||||
|
|||||||
@@ -1,67 +0,0 @@
|
|||||||
#!/usr/bin/env bun
|
|
||||||
|
|
||||||
// Embeds mupdf's `mupdf-wasm.wasm` into the compiled single-file binary.
|
|
||||||
//
|
|
||||||
// mupdf loads its wasm by reading the `mupdf-wasm.wasm` sibling of its own
|
|
||||||
// module via `new URL(..., import.meta.url)` + `readFileSync`. A `bun --compile`
|
|
||||||
// binary has no node_modules, so that read fails (`ENOENT .../mupdf-wasm.wasm`),
|
|
||||||
// and marking mupdf `--external` instead makes `bun --compile` eagerly fail to
|
|
||||||
// resolve the package at startup (the static `import * as mupdf` lives in a lazy
|
|
||||||
// chunk but is hoisted). So the binary build bundles mupdf and embeds the wasm
|
|
||||||
// bytes here, handing them to the WASM module as `$libmupdf_wasm_Module.wasmBinary`
|
|
||||||
// (see src/utils/markit.ts).
|
|
||||||
//
|
|
||||||
// `--generate` copies the wasm next to src/utils/mupdf-wasm-embed.ts and rewrites
|
|
||||||
// that module to import it via `with { type: "file" }`; `--reset` restores the
|
|
||||||
// checked-in placeholder and removes the copy. The npm `dist/cli.js` bundle never
|
|
||||||
// runs this — it keeps mupdf external and loads the wasm from node_modules.
|
|
||||||
|
|
||||||
import * as fs from "node:fs/promises";
|
|
||||||
import { createRequire } from "node:module";
|
|
||||||
import * as path from "node:path";
|
|
||||||
|
|
||||||
const utilsDir = path.join(import.meta.dir, "..", "src", "utils");
|
|
||||||
const helperPath = path.join(utilsDir, "mupdf-wasm-embed.ts");
|
|
||||||
const wasmCopyPath = path.join(utilsDir, "mupdf-wasm.wasm");
|
|
||||||
|
|
||||||
const placeholder = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
|
|
||||||
//
|
|
||||||
// Compiled single-file binaries cannot let mupdf resolve its \`mupdf-wasm.wasm\`
|
|
||||||
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
|
|
||||||
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
|
|
||||||
// wasm bytes via \`with { type: "file" }\` and copies the wasm next to it. Source
|
|
||||||
// checkouts, \`bun test\`, and the npm \`dist/cli.js\` bundle keep mupdf external and
|
|
||||||
// load the wasm from node_modules, so this placeholder returns undefined and the
|
|
||||||
// build resets back to it afterward.
|
|
||||||
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
|
|
||||||
\treturn undefined;
|
|
||||||
}
|
|
||||||
`;
|
|
||||||
|
|
||||||
const generated = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit or commit.
|
|
||||||
import { readFileSync } from "node:fs";
|
|
||||||
import wasmPath from "./mupdf-wasm.wasm" with { type: "file" };
|
|
||||||
|
|
||||||
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
|
|
||||||
\treturn readFileSync(wasmPath);
|
|
||||||
}
|
|
||||||
`;
|
|
||||||
|
|
||||||
if (process.argv.includes("--reset")) {
|
|
||||||
await Bun.write(helperPath, placeholder);
|
|
||||||
try {
|
|
||||||
await fs.unlink(wasmCopyPath);
|
|
||||||
} catch (err) {
|
|
||||||
if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err;
|
|
||||||
}
|
|
||||||
process.exit(0);
|
|
||||||
}
|
|
||||||
|
|
||||||
const wasmSource = path.join(path.dirname(createRequire(import.meta.url).resolve("mupdf")), "mupdf-wasm.wasm");
|
|
||||||
const wasmFile = Bun.file(wasmSource);
|
|
||||||
if (!(await wasmFile.exists())) {
|
|
||||||
throw new Error(`mupdf wasm not found at ${wasmSource}; run \`bun install\` first.`);
|
|
||||||
}
|
|
||||||
await Bun.write(wasmCopyPath, wasmFile);
|
|
||||||
await Bun.write(helperPath, generated);
|
|
||||||
console.log(`Embedded mupdf wasm (${wasmFile.size} bytes) into ${path.relative(process.cwd(), wasmCopyPath)}`);
|
|
||||||
@@ -1,15 +1,15 @@
|
|||||||
This directory contains an in-house document-to-markdown engine adapted from
|
Portions of this in-house document-to-markdown engine are adapted from
|
||||||
markit-ai (https://github.com/Michaelliv/markit), used under the MIT License.
|
markit-ai (https://github.com/Michaelliv/markit), used under the MIT License.
|
||||||
|
This attribution covers the shared registry/types and the DOCX, PPTX, XLSX,
|
||||||
|
and EPUB converters. The PDF converter is implemented separately and is not
|
||||||
|
derived from markit-ai.
|
||||||
|
|
||||||
Copyright (c) 2026 Michael Liv
|
Copyright (c) 2026 Michael Liv
|
||||||
|
|
||||||
Only the converters for the document formats omp supports are ported (pdf,
|
The CLI, plugin/provider, and unused converters (html, image, audio,
|
||||||
docx, pptx, xlsx, epub); the CLI, plugin/provider, and unused converters
|
plain-text, rss, github, wikipedia, csv, json, yaml, ipynb, iwork, zip, xml)
|
||||||
(html, image, audio, plain-text, rss, github, wikipedia, csv, json, yaml,
|
were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and `.rtf` have no converter
|
||||||
ipynb, iwork, zip, xml) were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and
|
and surface a conversion error.
|
||||||
`.rtf` are routed by the read/fetch tools but have no converter — they surface
|
|
||||||
a conversion error, exactly as upstream markit did. Logic is ported faithfully
|
|
||||||
so conversion output matches the upstream package.
|
|
||||||
|
|
||||||
MIT License
|
MIT License
|
||||||
|
|
||||||
|
|||||||
@@ -1,103 +0,0 @@
|
|||||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Multi-column layout detection and text box reordering.
|
|
||||||
*
|
|
||||||
* Many PDFs (legal documents, datasheets, academic papers) use two-column
|
|
||||||
* layouts. Without column detection, text boxes are ordered by Y position
|
|
||||||
* only, interleaving left and right column content.
|
|
||||||
*
|
|
||||||
* Algorithm:
|
|
||||||
* 1. Collect left edges of all text boxes on the page
|
|
||||||
* 2. Find the largest horizontal gap between consecutive left edges
|
|
||||||
* 3. If gap > MIN_GAP_RATIO of the text width and both sides have
|
|
||||||
* enough boxes → multi-column detected
|
|
||||||
* 4. Assign each text box to a column based on its center X
|
|
||||||
* 5. Return columns in reading order (left-to-right, top-to-bottom)
|
|
||||||
*
|
|
||||||
* This only detects the column structure. The caller is responsible for
|
|
||||||
* processing each column's text boxes independently (table detection,
|
|
||||||
* rendering, etc.).
|
|
||||||
*/
|
|
||||||
import type { TextBox } from "./types";
|
|
||||||
|
|
||||||
export interface ColumnLayout {
|
|
||||||
/** Number of columns detected (1 = single column, 2+ = multi-column). */
|
|
||||||
columnCount: number;
|
|
||||||
/** Text boxes grouped by column, in reading order (left to right). */
|
|
||||||
columns: TextBox[][];
|
|
||||||
/** X positions of column boundaries (between columns). */
|
|
||||||
boundaries: number[];
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Minimum gap as a fraction of the total text width to consider a column
|
|
||||||
* boundary. A two-column layout typically has ~50% gap; we use a lower
|
|
||||||
* threshold to catch asymmetric columns.
|
|
||||||
*/
|
|
||||||
const MIN_GAP_RATIO = 0.15;
|
|
||||||
/** Minimum number of text boxes on each side of the gap. */
|
|
||||||
const MIN_BOXES_PER_COLUMN = 4;
|
|
||||||
/** Minimum gap in absolute points to avoid splitting on small whitespace. */
|
|
||||||
const MIN_GAP_PTS = 40;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Detect column layout and return text boxes grouped by column.
|
|
||||||
*
|
|
||||||
* For single-column pages, returns all boxes in one group.
|
|
||||||
* For multi-column pages, returns boxes split by column in reading order.
|
|
||||||
*/
|
|
||||||
export function detectColumns(textBoxes: TextBox[]): ColumnLayout {
|
|
||||||
if (textBoxes.length < MIN_BOXES_PER_COLUMN * 2) {
|
|
||||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
|
||||||
}
|
|
||||||
// Collect unique left edges (rounded to avoid float noise)
|
|
||||||
const lefts = [...new Set(textBoxes.map(tb => Math.round(tb.bounds.left)))].sort((a, b) => a - b);
|
|
||||||
if (lefts.length < 2) {
|
|
||||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
|
||||||
}
|
|
||||||
const textXMin = lefts[0];
|
|
||||||
const textXMax = Math.max(...textBoxes.map(tb => Math.round(tb.bounds.right)));
|
|
||||||
const textWidth = textXMax - textXMin;
|
|
||||||
if (textWidth <= 0) {
|
|
||||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
|
||||||
}
|
|
||||||
// Find the largest gap between consecutive left-edge positions
|
|
||||||
let maxGap = 0;
|
|
||||||
let gapLeft = 0;
|
|
||||||
let gapRight = 0;
|
|
||||||
for (let i = 1; i < lefts.length; i++) {
|
|
||||||
const gap = lefts[i] - lefts[i - 1];
|
|
||||||
if (gap > maxGap) {
|
|
||||||
maxGap = gap;
|
|
||||||
gapLeft = lefts[i - 1];
|
|
||||||
gapRight = lefts[i];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const gapRatio = maxGap / textWidth;
|
|
||||||
if (gapRatio < MIN_GAP_RATIO || maxGap < MIN_GAP_PTS) {
|
|
||||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
|
||||||
}
|
|
||||||
// Split point is the midpoint of the gap
|
|
||||||
const splitX = (gapLeft + gapRight) / 2;
|
|
||||||
// Assign boxes to columns based on center X
|
|
||||||
const leftCol: TextBox[] = [];
|
|
||||||
const rightCol: TextBox[] = [];
|
|
||||||
for (const tb of textBoxes) {
|
|
||||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
|
||||||
if (cx < splitX) {
|
|
||||||
leftCol.push(tb);
|
|
||||||
} else {
|
|
||||||
rightCol.push(tb);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Validate both columns have enough content
|
|
||||||
if (leftCol.length < MIN_BOXES_PER_COLUMN || rightCol.length < MIN_BOXES_PER_COLUMN) {
|
|
||||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
|
||||||
}
|
|
||||||
return {
|
|
||||||
columnCount: 2,
|
|
||||||
columns: [leftCol, rightCol],
|
|
||||||
boundaries: [splitX],
|
|
||||||
};
|
|
||||||
}
|
|
||||||
@@ -1,598 +0,0 @@
|
|||||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
|
||||||
|
|
||||||
/**
|
|
||||||
* PDF content extraction using mupdf.
|
|
||||||
*
|
|
||||||
* Extracts text boxes (with position, font size, bold) and vector line
|
|
||||||
* segments (table borders) from each page. Uses mupdf's native WASM
|
|
||||||
* engine for fast parsing, and reads raw content streams for vector graphics.
|
|
||||||
*
|
|
||||||
* Coordinate system: PDF native (origin = bottom-left, Y increases upward).
|
|
||||||
*/
|
|
||||||
import type * as mupdf from "mupdf";
|
|
||||||
import type { ImageRegion, PageContent, Segment, TextBox } from "./types";
|
|
||||||
|
|
||||||
// mupdf instantiates its WASM module via a top-level await. A static
|
|
||||||
// `import * as mupdf` would pull that await into this module's init, which makes
|
|
||||||
// the whole bundled markit chunk's `__esm` init async — and bun's compiled
|
|
||||||
// bundler fails to await that init transitively through the `../markit` barrel,
|
|
||||||
// exposing the converter classes before their module-level consts initialize
|
|
||||||
// (e.g. `EXTENSIONS` reads as undefined). Importing mupdf lazily keeps the chunk
|
|
||||||
// init synchronous and also keeps the ~10MB wasm off non-PDF conversions.
|
|
||||||
let mupdfModule: typeof mupdf | undefined;
|
|
||||||
async function loadMupdf(): Promise<typeof mupdf> {
|
|
||||||
if (!mupdfModule) {
|
|
||||||
mupdfModule = await import("mupdf");
|
|
||||||
}
|
|
||||||
return mupdfModule;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** mupdf structured-text JSON bounding box (top-left origin). */
|
|
||||||
interface StextBBox {
|
|
||||||
x: number;
|
|
||||||
y: number;
|
|
||||||
w: number;
|
|
||||||
h: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Font metadata attached to a structured-text line. */
|
|
||||||
interface StextFont {
|
|
||||||
size?: number;
|
|
||||||
weight?: string;
|
|
||||||
name?: string;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A line within a text block in mupdf structured-text JSON. */
|
|
||||||
interface StextLine {
|
|
||||||
text?: string;
|
|
||||||
font?: StextFont;
|
|
||||||
bbox: StextBBox;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A block (text or image) in mupdf structured-text JSON. */
|
|
||||||
interface StextBlock {
|
|
||||||
type: string;
|
|
||||||
bbox: StextBBox;
|
|
||||||
lines: StextLine[];
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Parsed mupdf structured-text JSON for a page. */
|
|
||||||
interface StructuredTextJSON {
|
|
||||||
blocks: StextBlock[];
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A raw text fragment before merging into word/phrase boxes. */
|
|
||||||
interface RawTextItem {
|
|
||||||
text: string;
|
|
||||||
x: number;
|
|
||||||
y: number;
|
|
||||||
width: number;
|
|
||||||
height: number;
|
|
||||||
fontSize: number;
|
|
||||||
isBold: boolean;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Text extraction
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/** Y tolerance for merging text fragments on the same visual line. */
|
|
||||||
const SAME_LINE_Y_TOLERANCE = 2;
|
|
||||||
/** Max horizontal gap (pts) to merge adjacent fragments into one text box. */
|
|
||||||
const MAX_MERGE_GAP = 14;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Merge horizontally adjacent raw text items on the same visual line into
|
|
||||||
* word/phrase-level text boxes.
|
|
||||||
*/
|
|
||||||
function mergeIntoWords(raws: RawTextItem[]): RawTextItem[] {
|
|
||||||
if (raws.length === 0) return [];
|
|
||||||
// Sort by Y descending (top-first in bottom-left coords), then X ascending
|
|
||||||
const sorted = [...raws].sort((a, b) => {
|
|
||||||
const dy = b.y - a.y;
|
|
||||||
return Math.abs(dy) > SAME_LINE_Y_TOLERANCE ? dy : a.x - b.x;
|
|
||||||
});
|
|
||||||
const merged: RawTextItem[] = [];
|
|
||||||
let cur = { ...sorted[0] };
|
|
||||||
for (let i = 1; i < sorted.length; i++) {
|
|
||||||
const next = sorted[i];
|
|
||||||
const sameY = Math.abs(next.y - cur.y) <= SAME_LINE_Y_TOLERANCE;
|
|
||||||
const close = next.x <= cur.x + cur.width + MAX_MERGE_GAP;
|
|
||||||
if (sameY && close) {
|
|
||||||
const gap = next.x - (cur.x + cur.width);
|
|
||||||
const sep = gap > 1 ? " " : "";
|
|
||||||
cur.text += sep + next.text;
|
|
||||||
cur.width = next.x + next.width - cur.x;
|
|
||||||
cur.height = Math.max(cur.height, next.height);
|
|
||||||
cur.fontSize = Math.max(cur.fontSize, next.fontSize);
|
|
||||||
cur.isBold = cur.isBold || next.isBold;
|
|
||||||
} else {
|
|
||||||
merged.push(cur);
|
|
||||||
cur = { ...next };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
merged.push(cur);
|
|
||||||
return merged;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Extract text boxes from a mupdf page using structured text output.
|
|
||||||
*
|
|
||||||
* mupdf's structured text JSON uses top-left origin; we convert to
|
|
||||||
* bottom-left (standard PDF coordinates) using the page height.
|
|
||||||
*/
|
|
||||||
function extractTextBoxes(
|
|
||||||
page: mupdf.Page,
|
|
||||||
pageNumber: number,
|
|
||||||
pageHeight: number,
|
|
||||||
stext?: StructuredTextJSON,
|
|
||||||
): TextBox[] {
|
|
||||||
if (!stext) {
|
|
||||||
stext = JSON.parse(page.toStructuredText("preserve-whitespace").asJSON()) as StructuredTextJSON;
|
|
||||||
}
|
|
||||||
const raws: RawTextItem[] = [];
|
|
||||||
for (const block of stext.blocks) {
|
|
||||||
if (block.type !== "text") continue;
|
|
||||||
for (const line of block.lines) {
|
|
||||||
const text = line.text?.trim();
|
|
||||||
if (!text) continue;
|
|
||||||
const fontSize = line.font?.size ?? 0;
|
|
||||||
const weight = line.font?.weight ?? "normal";
|
|
||||||
const fontName = line.font?.name ?? "";
|
|
||||||
const isBold = weight === "bold" || /bold/i.test(fontName) || /Black|Heavy/i.test(fontName);
|
|
||||||
// mupdf bbox: {x, y, w, h} in top-left coords
|
|
||||||
// Convert to bottom-left: pdfY = pageHeight - (bbox.y + bbox.h)
|
|
||||||
const bboxY = line.bbox.y;
|
|
||||||
const bboxH = line.bbox.h;
|
|
||||||
const pdfY = pageHeight - (bboxY + bboxH);
|
|
||||||
raws.push({
|
|
||||||
text,
|
|
||||||
x: line.bbox.x,
|
|
||||||
y: pdfY,
|
|
||||||
width: line.bbox.w,
|
|
||||||
height: bboxH,
|
|
||||||
fontSize,
|
|
||||||
isBold,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const words = mergeIntoWords(raws);
|
|
||||||
return words
|
|
||||||
.map((w, i) => ({
|
|
||||||
id: `p${pageNumber}-t${i}`,
|
|
||||||
text: w.text.trim(),
|
|
||||||
pageNumber,
|
|
||||||
fontSize: w.fontSize,
|
|
||||||
isBold: w.isBold,
|
|
||||||
bounds: {
|
|
||||||
left: w.x,
|
|
||||||
right: w.x + w.width,
|
|
||||||
bottom: w.y,
|
|
||||||
top: w.y + w.height,
|
|
||||||
},
|
|
||||||
}))
|
|
||||||
.filter(b => b.text.length > 0);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Vector segment extraction from raw content stream
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/** Minimum aspect ratio for a filled rect to be considered a line. */
|
|
||||||
const LINE_ASPECT_THRESHOLD = 6;
|
|
||||||
/** Minimum length (pts) for a segment to count. */
|
|
||||||
const MIN_LENGTH = 2;
|
|
||||||
/** Maximum thickness (pts) for a border line (filters out filled areas). */
|
|
||||||
const MAX_THICKNESS = 3;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Convert a thin filled rectangle to a horizontal or vertical segment.
|
|
||||||
* Returns null if the rect doesn't look like a border line.
|
|
||||||
*/
|
|
||||||
function thinRectToSegment(id: string, x: number, y: number, w: number, h: number): Segment | null {
|
|
||||||
const aw = Math.abs(w);
|
|
||||||
const ah = Math.abs(h);
|
|
||||||
if (aw > ah * LINE_ASPECT_THRESHOLD && aw >= MIN_LENGTH && ah <= MAX_THICKNESS) {
|
|
||||||
// Horizontal line
|
|
||||||
const cy = y + ah / 2;
|
|
||||||
return { id, x1: x, y1: cy, x2: x + aw, y2: cy };
|
|
||||||
}
|
|
||||||
if (ah > aw * LINE_ASPECT_THRESHOLD && ah >= MIN_LENGTH && aw <= MAX_THICKNESS) {
|
|
||||||
// Vertical line
|
|
||||||
const cx = x + aw / 2;
|
|
||||||
return { id, x1: cx, y1: y, x2: cx, y2: y + ah };
|
|
||||||
}
|
|
||||||
return null;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Emit 4 edge segments from a stroked rectangle.
|
|
||||||
*/
|
|
||||||
function pushStrokedRectEdges(segments: Segment[], id: string, x: number, y: number, w: number, h: number): void {
|
|
||||||
const aw = Math.abs(w);
|
|
||||||
const ah = Math.abs(h);
|
|
||||||
const base = id;
|
|
||||||
if (aw >= MIN_LENGTH) {
|
|
||||||
segments.push({ id: `${base}-b`, x1: x, y1: y, x2: x + aw, y2: y });
|
|
||||||
segments.push({
|
|
||||||
id: `${base}-t`,
|
|
||||||
x1: x,
|
|
||||||
y1: y + ah,
|
|
||||||
x2: x + aw,
|
|
||||||
y2: y + ah,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
if (ah >= MIN_LENGTH) {
|
|
||||||
segments.push({ id: `${base}-l`, x1: x, y1: y, x2: x, y2: y + ah });
|
|
||||||
segments.push({
|
|
||||||
id: `${base}-r`,
|
|
||||||
x1: x + aw,
|
|
||||||
y1: y,
|
|
||||||
x2: x + aw,
|
|
||||||
y2: y + ah,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
const CTM_IDENTITY = [1, 0, 0, 1, 0, 0];
|
|
||||||
|
|
||||||
/** Concatenate two affine matrices: result = parent × child. */
|
|
||||||
function ctmConcat(p: number[], c: number[]): number[] {
|
|
||||||
return [
|
|
||||||
p[0] * c[0] + p[2] * c[1],
|
|
||||||
p[1] * c[0] + p[3] * c[1],
|
|
||||||
p[0] * c[2] + p[2] * c[3],
|
|
||||||
p[1] * c[2] + p[3] * c[3],
|
|
||||||
p[0] * c[4] + p[2] * c[5] + p[4],
|
|
||||||
p[1] * c[4] + p[3] * c[5] + p[5],
|
|
||||||
];
|
|
||||||
}
|
|
||||||
|
|
||||||
function ctmApply(m: number[], x: number, y: number): [number, number] {
|
|
||||||
return [m[0] * x + m[2] * y + m[4], m[1] * x + m[3] * y + m[5]];
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Content stream parsing
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/**
|
|
||||||
* Parse a PDF content stream and extract line segments from thin filled
|
|
||||||
* rectangles (re+f), stroked rectangles (re+S), and explicit lines (m/l+S).
|
|
||||||
* Tracks the CTM via q/Q/cm operators so coordinates are in page space.
|
|
||||||
*/
|
|
||||||
function extractSegmentsFromContentStream(raw: string, pageNumber: number): Segment[] {
|
|
||||||
const segments: Segment[] = [];
|
|
||||||
const tokens = tokenizeContentStream(raw);
|
|
||||||
let idx = 0;
|
|
||||||
let strokeWidth = 1.0;
|
|
||||||
// Graphics state stack (q/Q): saves CTM + strokeWidth
|
|
||||||
let ctm = [...CTM_IDENTITY];
|
|
||||||
const stateStack: Array<{ ctm: number[]; strokeWidth: number }> = [];
|
|
||||||
// State for path building (in user coordinates, pre-CTM)
|
|
||||||
let curX = 0;
|
|
||||||
let curY = 0;
|
|
||||||
let pathStartX = 0;
|
|
||||||
let pathStartY = 0;
|
|
||||||
const pendingRects: Array<{ x: number; y: number; w: number; h: number }> = [];
|
|
||||||
const pendingLines: Array<{ x1: number; y1: number; x2: number; y2: number }> = [];
|
|
||||||
function flushPath(mode: "fill" | "stroke"): void {
|
|
||||||
const sid = () => `p${pageNumber}-s${segments.length}`;
|
|
||||||
if (mode === "fill") {
|
|
||||||
for (const r of pendingRects) {
|
|
||||||
// Transform the rect corners through CTM, then check if it's a thin line
|
|
||||||
const [x0, y0] = ctmApply(ctm, r.x, r.y);
|
|
||||||
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
|
|
||||||
const seg = thinRectToSegment(
|
|
||||||
sid(),
|
|
||||||
Math.min(x0, x1),
|
|
||||||
Math.min(y0, y1),
|
|
||||||
Math.abs(x1 - x0),
|
|
||||||
Math.abs(y1 - y0),
|
|
||||||
);
|
|
||||||
if (seg) segments.push(seg);
|
|
||||||
}
|
|
||||||
} else if (mode === "stroke" && strokeWidth <= MAX_THICKNESS) {
|
|
||||||
for (const r of pendingRects) {
|
|
||||||
const [x0, y0] = ctmApply(ctm, r.x, r.y);
|
|
||||||
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
|
|
||||||
pushStrokedRectEdges(
|
|
||||||
segments,
|
|
||||||
sid(),
|
|
||||||
Math.min(x0, x1),
|
|
||||||
Math.min(y0, y1),
|
|
||||||
Math.abs(x1 - x0),
|
|
||||||
Math.abs(y1 - y0),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
for (const l of pendingLines) {
|
|
||||||
const [lx1, ly1] = ctmApply(ctm, l.x1, l.y1);
|
|
||||||
const [lx2, ly2] = ctmApply(ctm, l.x2, l.y2);
|
|
||||||
const dx = Math.abs(lx2 - lx1);
|
|
||||||
const dy = Math.abs(ly2 - ly1);
|
|
||||||
// Only keep H/V lines
|
|
||||||
if ((dx >= MIN_LENGTH && dy < 1) || (dy >= MIN_LENGTH && dx < 1)) {
|
|
||||||
segments.push({ id: sid(), x1: lx1, y1: ly1, x2: lx2, y2: ly2 });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
pendingRects.length = 0;
|
|
||||||
pendingLines.length = 0;
|
|
||||||
}
|
|
||||||
while (idx < tokens.length) {
|
|
||||||
const t = tokens[idx];
|
|
||||||
if (t === "q") {
|
|
||||||
stateStack.push({ ctm: [...ctm], strokeWidth });
|
|
||||||
} else if (t === "Q") {
|
|
||||||
const saved = stateStack.pop();
|
|
||||||
if (saved) {
|
|
||||||
ctm = saved.ctm;
|
|
||||||
strokeWidth = saved.strokeWidth;
|
|
||||||
}
|
|
||||||
} else if (t === "cm" && idx >= 6) {
|
|
||||||
const a = Number(tokens[idx - 6]);
|
|
||||||
const b = Number(tokens[idx - 5]);
|
|
||||||
const c = Number(tokens[idx - 4]);
|
|
||||||
const d = Number(tokens[idx - 3]);
|
|
||||||
const e = Number(tokens[idx - 2]);
|
|
||||||
const f = Number(tokens[idx - 1]);
|
|
||||||
ctm = ctmConcat(ctm, [a, b, c, d, e, f]);
|
|
||||||
} else if (t === "w" && idx >= 1) {
|
|
||||||
strokeWidth = Number(tokens[idx - 1]) || strokeWidth;
|
|
||||||
} else if (t === "re" && idx >= 4) {
|
|
||||||
const x = Number(tokens[idx - 4]);
|
|
||||||
const y = Number(tokens[idx - 3]);
|
|
||||||
const w = Number(tokens[idx - 2]);
|
|
||||||
const h = Number(tokens[idx - 1]);
|
|
||||||
if (Number.isFinite(x + y + w + h)) {
|
|
||||||
pendingRects.push({ x, y, w, h });
|
|
||||||
}
|
|
||||||
} else if (t === "m" && idx >= 2) {
|
|
||||||
curX = Number(tokens[idx - 2]);
|
|
||||||
curY = Number(tokens[idx - 1]);
|
|
||||||
pathStartX = curX;
|
|
||||||
pathStartY = curY;
|
|
||||||
} else if (t === "l" && idx >= 2) {
|
|
||||||
const x2 = Number(tokens[idx - 2]);
|
|
||||||
const y2 = Number(tokens[idx - 1]);
|
|
||||||
pendingLines.push({ x1: curX, y1: curY, x2, y2 });
|
|
||||||
curX = x2;
|
|
||||||
curY = y2;
|
|
||||||
} else if (t === "h") {
|
|
||||||
// closePath: line back to start
|
|
||||||
if (curX !== pathStartX || curY !== pathStartY) {
|
|
||||||
pendingLines.push({
|
|
||||||
x1: curX,
|
|
||||||
y1: curY,
|
|
||||||
x2: pathStartX,
|
|
||||||
y2: pathStartY,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
curX = pathStartX;
|
|
||||||
curY = pathStartY;
|
|
||||||
} else if (t === "f" || t === "F" || t === "f*") {
|
|
||||||
flushPath("fill");
|
|
||||||
} else if (t === "S" || t === "s") {
|
|
||||||
if (t === "s") {
|
|
||||||
// closeStroke: implicit closePath
|
|
||||||
if (curX !== pathStartX || curY !== pathStartY) {
|
|
||||||
pendingLines.push({
|
|
||||||
x1: curX,
|
|
||||||
y1: curY,
|
|
||||||
x2: pathStartX,
|
|
||||||
y2: pathStartY,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
flushPath("stroke");
|
|
||||||
} else if (t === "B" || t === "B*" || t === "b" || t === "b*") {
|
|
||||||
// fill + stroke combined
|
|
||||||
flushPath("fill");
|
|
||||||
flushPath("stroke");
|
|
||||||
} else if (t === "n") {
|
|
||||||
// end path without painting — discard
|
|
||||||
pendingRects.length = 0;
|
|
||||||
pendingLines.length = 0;
|
|
||||||
}
|
|
||||||
idx++;
|
|
||||||
}
|
|
||||||
return segments;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Fast tokenizer for PDF content streams.
|
|
||||||
* Splits on whitespace, skipping comments, string literals, and inline image payloads.
|
|
||||||
*/
|
|
||||||
function tokenizeContentStream(raw: string): string[] {
|
|
||||||
const tokens: string[] = [];
|
|
||||||
const len = raw.length;
|
|
||||||
let i = 0;
|
|
||||||
let inInlineImage = false;
|
|
||||||
while (i < len) {
|
|
||||||
const ch = raw.charCodeAt(i);
|
|
||||||
// Skip whitespace
|
|
||||||
if (ch <= 32) {
|
|
||||||
i++;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Skip comments
|
|
||||||
if (ch === 37 /* % */) {
|
|
||||||
while (i < len && raw.charCodeAt(i) !== 10) i++;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Skip string literals (...)
|
|
||||||
if (ch === 40 /* ( */) {
|
|
||||||
let depth = 1;
|
|
||||||
i++;
|
|
||||||
while (i < len && depth > 0) {
|
|
||||||
const c = raw.charCodeAt(i);
|
|
||||||
if (c === 92 /* \ */) {
|
|
||||||
i++;
|
|
||||||
} else if (c === 40) {
|
|
||||||
depth++;
|
|
||||||
} else if (c === 41) {
|
|
||||||
depth--;
|
|
||||||
}
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Skip hex strings <...>
|
|
||||||
if (ch === 60 /* < */ && i + 1 < len && raw.charCodeAt(i + 1) !== 60) {
|
|
||||||
i++;
|
|
||||||
while (i < len && raw.charCodeAt(i) !== 62) i++;
|
|
||||||
i++; // skip >
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Skip dict delimiters << >>
|
|
||||||
if (ch === 60 && i + 1 < len && raw.charCodeAt(i + 1) === 60) {
|
|
||||||
i += 2;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (ch === 62 && i + 1 < len && raw.charCodeAt(i + 1) === 62) {
|
|
||||||
i += 2;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Skip stray closing delimiters from malformed streams. They cannot start
|
|
||||||
// a token, so leaving i unchanged would spin forever.
|
|
||||||
if (ch === 41 || ch === 62) {
|
|
||||||
i++;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Regular token: read until whitespace or delimiter
|
|
||||||
const start = i;
|
|
||||||
while (i < len) {
|
|
||||||
const c = raw.charCodeAt(i);
|
|
||||||
if (c <= 32 || c === 40 || c === 41 || c === 60 || c === 62 || c === 37) break;
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
if (i > start) {
|
|
||||||
const token = raw.substring(start, i);
|
|
||||||
tokens.push(token);
|
|
||||||
if (token === "BI") {
|
|
||||||
inInlineImage = true;
|
|
||||||
} else if (token === "ID" && inInlineImage) {
|
|
||||||
while (i < len && raw.charCodeAt(i) <= 32) i++;
|
|
||||||
while (i < len) {
|
|
||||||
const c = raw.charCodeAt(i);
|
|
||||||
const prev = i === 0 ? 32 : raw.charCodeAt(i - 1);
|
|
||||||
const next = i + 2 >= len ? 32 : raw.charCodeAt(i + 2);
|
|
||||||
if (c === 69 && raw.charCodeAt(i + 1) === 73 && prev <= 32 && next <= 32) {
|
|
||||||
i += 2;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
i++;
|
|
||||||
}
|
|
||||||
inInlineImage = false;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return tokens;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Image region detection
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/** Minimum area (pts²) for an image to be considered a diagram, not an icon. */
|
|
||||||
const MIN_IMAGE_AREA = 5000;
|
|
||||||
|
|
||||||
function extractImageRegions(stext: StructuredTextJSON, pageNumber: number, pageHeight: number): ImageRegion[] {
|
|
||||||
const regions: ImageRegion[] = [];
|
|
||||||
for (const block of stext.blocks) {
|
|
||||||
if (block.type !== "image") continue;
|
|
||||||
const { x, y, w, h } = block.bbox;
|
|
||||||
if (w * h < MIN_IMAGE_AREA) continue; // skip tiny icons
|
|
||||||
// Convert Y from mupdf (top-left) to PDF (bottom-left) for ordering
|
|
||||||
const pdfTopY = pageHeight - y;
|
|
||||||
regions.push({
|
|
||||||
id: `p${pageNumber}-img${regions.length}`,
|
|
||||||
pageNumber,
|
|
||||||
bbox: { x, y, w, h },
|
|
||||||
topY: pdfTopY,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
return regions;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Public API
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/**
|
|
||||||
* Render an image region from a PDF page as a PNG buffer.
|
|
||||||
* Uses mupdf's DrawDevice to render just the cropped area at 2x resolution.
|
|
||||||
*/
|
|
||||||
export async function renderImageRegion(input: Uint8Array, region: ImageRegion): Promise<Uint8Array> {
|
|
||||||
const m = await loadMupdf();
|
|
||||||
const doc = m.Document.openDocument(input, "application/pdf");
|
|
||||||
const page = doc.loadPage(region.pageNumber - 1);
|
|
||||||
const pad = 10;
|
|
||||||
const bx = region.bbox.x - pad;
|
|
||||||
const by = region.bbox.y - pad;
|
|
||||||
const bw = region.bbox.w + 2 * pad;
|
|
||||||
const bh = region.bbox.h + 2 * pad;
|
|
||||||
const scale = 2;
|
|
||||||
const pw = Math.round(bw * scale);
|
|
||||||
const ph = Math.round(bh * scale);
|
|
||||||
const pix = new m.Pixmap(m.ColorSpace.DeviceRGB, [0, 0, pw, ph], false);
|
|
||||||
pix.clear(255);
|
|
||||||
const matrix: mupdf.Matrix = [scale, 0, 0, scale, -bx * scale, -by * scale];
|
|
||||||
const dl = page.toDisplayList();
|
|
||||||
const dev = new m.DrawDevice(matrix, pix);
|
|
||||||
dl.run(dev, m.Matrix.identity);
|
|
||||||
dev.close();
|
|
||||||
return pix.asPNG();
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Extract text boxes and vector segments from all pages of a PDF buffer.
|
|
||||||
*/
|
|
||||||
export async function extractPages(input: Uint8Array): Promise<PageContent[]> {
|
|
||||||
const m = await loadMupdf();
|
|
||||||
const doc = m.Document.openDocument(input, "application/pdf");
|
|
||||||
const pages: PageContent[] = [];
|
|
||||||
for (let i = 0; i < doc.countPages(); i++) {
|
|
||||||
const pageNumber = i + 1;
|
|
||||||
const page = doc.loadPage(i);
|
|
||||||
const bounds = page.getBounds();
|
|
||||||
const pageHeight = bounds[3] - bounds[1];
|
|
||||||
// Single structured text pass with both flags
|
|
||||||
const stext = JSON.parse(
|
|
||||||
page.toStructuredText("preserve-whitespace,preserve-images").asJSON(),
|
|
||||||
) as StructuredTextJSON;
|
|
||||||
// Extract text boxes and image regions from the same parse
|
|
||||||
const textBoxes = extractTextBoxes(page, pageNumber, pageHeight, stext);
|
|
||||||
const images = extractImageRegions(stext, pageNumber, pageHeight);
|
|
||||||
// Extract vector segments from raw content stream
|
|
||||||
let segments: Segment[] = [];
|
|
||||||
try {
|
|
||||||
const pageObj = (page as mupdf.PDFPage).getObject();
|
|
||||||
const contents = pageObj.get("Contents");
|
|
||||||
if (contents) {
|
|
||||||
let rawBytes: Uint8Array;
|
|
||||||
if (contents.isArray()) {
|
|
||||||
// Multiple content streams — concatenate
|
|
||||||
const parts: Uint8Array[] = [];
|
|
||||||
const len = contents.length ?? 0;
|
|
||||||
for (let j = 0; j < len; j++) {
|
|
||||||
const stream = contents.get(j);
|
|
||||||
if (stream?.readStream) {
|
|
||||||
parts.push(stream.readStream().asUint8Array());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const totalLen = parts.reduce((s, p) => s + p.length, 0);
|
|
||||||
rawBytes = new Uint8Array(totalLen);
|
|
||||||
let offset = 0;
|
|
||||||
for (const part of parts) {
|
|
||||||
rawBytes.set(part, offset);
|
|
||||||
offset += part.length;
|
|
||||||
}
|
|
||||||
} else {
|
|
||||||
rawBytes = contents.readStream().asUint8Array();
|
|
||||||
}
|
|
||||||
const raw = new TextDecoder().decode(rawBytes);
|
|
||||||
segments = extractSegmentsFromContentStream(raw, pageNumber);
|
|
||||||
}
|
|
||||||
} catch {
|
|
||||||
// Content stream extraction failed — proceed with text only
|
|
||||||
}
|
|
||||||
pages.push({ pageNumber, textBoxes, segments, images });
|
|
||||||
}
|
|
||||||
return pages;
|
|
||||||
}
|
|
||||||
@@ -1,780 +0,0 @@
|
|||||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Table grid detection from vector segments and text boxes.
|
|
||||||
*
|
|
||||||
* Ported from @oharato/pdf2md-ts with TypeScript types and without
|
|
||||||
* CJK-specific borderless table heuristics. The core algorithm:
|
|
||||||
*
|
|
||||||
* 1. Classify segments as horizontal or vertical lines
|
|
||||||
* 2. Group horizontal Y-lines into table groups (split by vertical gaps)
|
|
||||||
* 3. For each group:
|
|
||||||
* a. Full grid (H+V lines): build cells from grid intersections,
|
|
||||||
* place text via raycasting
|
|
||||||
* b. H-line only (no V lines): infer columns from text X positions
|
|
||||||
* 4. Prune empty rows/cols
|
|
||||||
*
|
|
||||||
* Coordinate system: PDF native (bottom-left origin, Y increases upward).
|
|
||||||
*/
|
|
||||||
import type { Segment, TableCell, TableGrid, TextBox } from "./types";
|
|
||||||
|
|
||||||
export interface GridResult {
|
|
||||||
grids: TableGrid[];
|
|
||||||
consumedIds: string[];
|
|
||||||
}
|
|
||||||
|
|
||||||
type RayDirection = "up" | "down" | "left" | "right";
|
|
||||||
|
|
||||||
interface Ray {
|
|
||||||
direction: RayDirection;
|
|
||||||
segmentId: string | null;
|
|
||||||
distance: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
interface Interval {
|
|
||||||
min: number;
|
|
||||||
max: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
function castRaysForTextBox(textBox: TextBox, segments: Segment[]): Ray[] {
|
|
||||||
const cx = (textBox.bounds.left + textBox.bounds.right) / 2;
|
|
||||||
const cy = (textBox.bounds.top + textBox.bounds.bottom) / 2;
|
|
||||||
let up: Ray = { direction: "up", segmentId: null, distance: Infinity };
|
|
||||||
let down: Ray = { direction: "down", segmentId: null, distance: Infinity };
|
|
||||||
let left: Ray = { direction: "left", segmentId: null, distance: Infinity };
|
|
||||||
let right: Ray = {
|
|
||||||
direction: "right",
|
|
||||||
segmentId: null,
|
|
||||||
distance: Infinity,
|
|
||||||
};
|
|
||||||
for (const seg of segments) {
|
|
||||||
const isH = Math.abs(seg.y1 - seg.y2) < 0.5;
|
|
||||||
const isV = Math.abs(seg.x1 - seg.x2) < 0.5;
|
|
||||||
if (isH) {
|
|
||||||
const minX = Math.min(seg.x1, seg.x2);
|
|
||||||
const maxX = Math.max(seg.x1, seg.x2);
|
|
||||||
if (cx >= minX && cx <= maxX) {
|
|
||||||
const d = seg.y1 - cy;
|
|
||||||
if (d >= 0 && d < up.distance) up = { direction: "up", segmentId: seg.id, distance: d };
|
|
||||||
const dd = cy - seg.y1;
|
|
||||||
if (dd >= 0 && dd < down.distance) down = { direction: "down", segmentId: seg.id, distance: dd };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (isV) {
|
|
||||||
const minY = Math.min(seg.y1, seg.y2);
|
|
||||||
const maxY = Math.max(seg.y1, seg.y2);
|
|
||||||
if (cy >= minY && cy <= maxY) {
|
|
||||||
const d = cx - seg.x1;
|
|
||||||
if (d >= 0 && d < left.distance) left = { direction: "left", segmentId: seg.id, distance: d };
|
|
||||||
const rd = seg.x1 - cx;
|
|
||||||
if (rd >= 0 && rd < right.distance) right = { direction: "right", segmentId: seg.id, distance: rd };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return [up, down, left, right];
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Utility
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
const AXIS_EPSILON = 0.8;
|
|
||||||
const PAGE_MARGIN = 20;
|
|
||||||
|
|
||||||
function uniqueSorted(values: number[]): number[] {
|
|
||||||
const sorted = [...values].sort((a, b) => a - b);
|
|
||||||
const result: number[] = [];
|
|
||||||
for (const v of sorted) {
|
|
||||||
if (result.length === 0 || Math.abs(result[result.length - 1] - v) > 1) result.push(v);
|
|
||||||
}
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Y-line group splitting
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
function chainCoversRange(intervals: Interval[], lowerY: number, upperY: number, eps: number): boolean {
|
|
||||||
const sorted = [...intervals].sort((a, b) => a.min - b.min);
|
|
||||||
let covered = lowerY;
|
|
||||||
for (const iv of sorted) {
|
|
||||||
if (iv.min > covered + eps) break;
|
|
||||||
if (iv.max > covered) covered = iv.max;
|
|
||||||
if (covered >= upperY - eps) return true;
|
|
||||||
}
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
function countBridgingVLineCols(upperY: number, lowerY: number, verticals: Segment[]): number {
|
|
||||||
const eps = 1.5;
|
|
||||||
const byX = new Map<number, Interval[]>();
|
|
||||||
for (const seg of verticals) {
|
|
||||||
const rx = Math.round(seg.x1);
|
|
||||||
if (!byX.has(rx)) byX.set(rx, []);
|
|
||||||
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
|
|
||||||
}
|
|
||||||
let count = 0;
|
|
||||||
for (const intervals of byX.values()) {
|
|
||||||
if (chainCoversRange(intervals, lowerY, upperY, eps)) count++;
|
|
||||||
}
|
|
||||||
return count;
|
|
||||||
}
|
|
||||||
|
|
||||||
function bridgingXSet(upperY: number, lowerY: number, verticals: Segment[]): Set<number> {
|
|
||||||
const eps = 1.5;
|
|
||||||
const xs = new Set<number>();
|
|
||||||
const byX = new Map<number, Interval[]>();
|
|
||||||
for (const seg of verticals) {
|
|
||||||
const rx = Math.round(seg.x1);
|
|
||||||
if (!byX.has(rx)) byX.set(rx, []);
|
|
||||||
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
|
|
||||||
}
|
|
||||||
for (const [rx, intervals] of byX) {
|
|
||||||
if (chainCoversRange(intervals, lowerY, upperY, eps)) xs.add(rx);
|
|
||||||
}
|
|
||||||
return xs;
|
|
||||||
}
|
|
||||||
|
|
||||||
const MIN_RICH_BRIDGING_COLS = 3;
|
|
||||||
|
|
||||||
function splitYLinesIntoGroups(yLines: number[], verticals: Segment[]): number[][] {
|
|
||||||
if (yLines.length === 0) return [];
|
|
||||||
const eps = 1.5;
|
|
||||||
const allX = verticals.map(s => Math.round(s.x1));
|
|
||||||
const globalXMin = allX.length > 0 ? Math.min(...allX) : 0;
|
|
||||||
const globalXMax = allX.length > 0 ? Math.max(...allX) : 0;
|
|
||||||
const groups: number[][] = [];
|
|
||||||
let currentGroup = [yLines[0]];
|
|
||||||
let prevBridgingCols = -1;
|
|
||||||
for (let i = 1; i < yLines.length; i++) {
|
|
||||||
const upperY = yLines[i - 1];
|
|
||||||
const lowerY = yLines[i];
|
|
||||||
const cols = countBridgingVLineCols(upperY, lowerY, verticals);
|
|
||||||
if (cols === 0) {
|
|
||||||
groups.push(currentGroup);
|
|
||||||
currentGroup = [yLines[i]];
|
|
||||||
prevBridgingCols = -1;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (prevBridgingCols >= MIN_RICH_BRIDGING_COLS && cols < MIN_RICH_BRIDGING_COLS) {
|
|
||||||
const bxs = bridgingXSet(upperY, lowerY, verticals);
|
|
||||||
const isOuterFrameOnly = [...bxs].every(
|
|
||||||
x => Math.abs(x - globalXMin) <= eps || Math.abs(x - globalXMax) <= eps,
|
|
||||||
);
|
|
||||||
if (!isOuterFrameOnly) {
|
|
||||||
groups.push(currentGroup);
|
|
||||||
currentGroup = [yLines[i - 1], yLines[i]];
|
|
||||||
prevBridgingCols = cols;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
currentGroup.push(yLines[i]);
|
|
||||||
prevBridgingCols = cols;
|
|
||||||
}
|
|
||||||
groups.push(currentGroup);
|
|
||||||
return groups;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Sub-row Y-cluster expansion
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
const Y_CLUSTER_GAP = 10;
|
|
||||||
const MIN_COLS_IN_TOP_CLUSTER = 2;
|
|
||||||
|
|
||||||
function assignToYCluster(y: number, clusters: number[]): number {
|
|
||||||
let closest = 0;
|
|
||||||
let closestDist = Math.abs(y - clusters[0]);
|
|
||||||
for (let k = 1; k < clusters.length; k++) {
|
|
||||||
const d = Math.abs(y - clusters[k]);
|
|
||||||
if (d < closestDist) {
|
|
||||||
closestDist = d;
|
|
||||||
closest = k;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return closest;
|
|
||||||
}
|
|
||||||
|
|
||||||
function expandSubRowsByYClusters(
|
|
||||||
originalRows: number,
|
|
||||||
cols: number,
|
|
||||||
cells: TableCell[],
|
|
||||||
cellBoxes: Map<TableCell, TextBox[]>,
|
|
||||||
): number {
|
|
||||||
let addedRows = 0;
|
|
||||||
for (let origRow = 0; origRow < originalRows; origRow++) {
|
|
||||||
const currentRow = origRow + addedRows;
|
|
||||||
const rowCellInfos: Array<{ cell: TableCell; col: number; boxes: TextBox[] }> = [];
|
|
||||||
for (let col = 0; col < cols; col++) {
|
|
||||||
const cell = cells.find(c => c.row === currentRow && c.col === col);
|
|
||||||
if (!cell) continue;
|
|
||||||
const boxes = cellBoxes.get(cell);
|
|
||||||
if (boxes && boxes.length > 0) rowCellInfos.push({ cell, col, boxes });
|
|
||||||
}
|
|
||||||
if (rowCellInfos.length === 0) continue;
|
|
||||||
const allMidYs = rowCellInfos.flatMap(({ boxes }) => boxes.map(b => (b.bounds.top + b.bounds.bottom) / 2));
|
|
||||||
const sortedY = [...new Set(allMidYs.map(y => Math.round(y * 10) / 10))].sort((a, b) => b - a);
|
|
||||||
const clusters = [sortedY[0]];
|
|
||||||
for (let i = 1; i < sortedY.length; i++) {
|
|
||||||
if (clusters[clusters.length - 1] - sortedY[i] > Y_CLUSTER_GAP) {
|
|
||||||
clusters.push(sortedY[i]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (clusters.length < 2) continue;
|
|
||||||
const colsInTopCluster = new Set<number>();
|
|
||||||
const totalNonEmptyCols = new Set<number>();
|
|
||||||
for (const { col, boxes } of rowCellInfos) {
|
|
||||||
totalNonEmptyCols.add(col);
|
|
||||||
if (boxes.some(b => assignToYCluster((b.bounds.top + b.bounds.bottom) / 2, clusters) === 0)) {
|
|
||||||
colsInTopCluster.add(col);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (colsInTopCluster.size < MIN_COLS_IN_TOP_CLUSTER) continue;
|
|
||||||
if (colsInTopCluster.size >= totalNonEmptyCols.size) continue;
|
|
||||||
const sparseColsHaveMultipleBoxes = rowCellInfos.some(
|
|
||||||
({ col, boxes }) => !colsInTopCluster.has(col) && boxes.length > 1,
|
|
||||||
);
|
|
||||||
if (!sparseColsHaveMultipleBoxes) continue;
|
|
||||||
const numSubRows = clusters.length;
|
|
||||||
const numNewRows = numSubRows - 1;
|
|
||||||
for (const cell of cells) {
|
|
||||||
if (cell.row > currentRow) cell.row += numNewRows;
|
|
||||||
}
|
|
||||||
for (let subRow = 1; subRow < numSubRows; subRow++) {
|
|
||||||
for (let col = 0; col < cols; col++) {
|
|
||||||
cells.push({
|
|
||||||
row: currentRow + subRow,
|
|
||||||
col,
|
|
||||||
text: "",
|
|
||||||
rowSpan: 1,
|
|
||||||
colSpan: 1,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (const { cell: origCell, col, boxes } of rowCellInfos) {
|
|
||||||
const subRowBoxGroups: TextBox[][] = Array.from({ length: numSubRows }, () => []);
|
|
||||||
for (const box of boxes) {
|
|
||||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
|
||||||
subRowBoxGroups[assignToYCluster(cy, clusters)].push(box);
|
|
||||||
}
|
|
||||||
cellBoxes.set(origCell, subRowBoxGroups[0]);
|
|
||||||
if (subRowBoxGroups[0].length === 0) cellBoxes.delete(origCell);
|
|
||||||
for (let subRow = 1; subRow < numSubRows; subRow++) {
|
|
||||||
if (subRowBoxGroups[subRow].length > 0) {
|
|
||||||
const newCell = cells.find(c => c.row === currentRow + subRow && c.col === col);
|
|
||||||
if (newCell) cellBoxes.set(newCell, subRowBoxGroups[subRow]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
addedRows += numNewRows;
|
|
||||||
}
|
|
||||||
return originalRows + addedRows;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Cross-column text box splitting
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/**
|
|
||||||
* Find which column a horizontal position falls into.
|
|
||||||
* Returns -1 if outside the grid.
|
|
||||||
*/
|
|
||||||
function findCol(x: number, xLines: number[]): number {
|
|
||||||
for (let i = 0; i < xLines.length - 1; i++) {
|
|
||||||
if (x >= xLines[i] && x <= xLines[i + 1]) return i;
|
|
||||||
}
|
|
||||||
return -1;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* When a text box spans across one or more vertical column boundaries,
|
|
||||||
* split it into multiple virtual text boxes — one per column — with the
|
|
||||||
* text divided proportionally by width.
|
|
||||||
*
|
|
||||||
* We split at word boundaries closest to the proportional split point
|
|
||||||
* so we don't chop words in half.
|
|
||||||
*/
|
|
||||||
function splitCrossColumnBoxes(textBoxes: TextBox[], xLines: number[]): TextBox[] {
|
|
||||||
const result: TextBox[] = [];
|
|
||||||
const MARGIN = 5; // allow small overlap before considering it cross-column
|
|
||||||
for (const tb of textBoxes) {
|
|
||||||
const leftCol = findCol(tb.bounds.left + MARGIN, xLines);
|
|
||||||
const rightCol = findCol(tb.bounds.right - MARGIN, xLines);
|
|
||||||
// Not spanning columns, or outside grid — keep as-is
|
|
||||||
if (leftCol < 0 || rightCol < 0 || leftCol === rightCol) {
|
|
||||||
result.push(tb);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Text box spans from leftCol to rightCol — split it
|
|
||||||
const totalWidth = tb.bounds.right - tb.bounds.left;
|
|
||||||
if (totalWidth <= 0) {
|
|
||||||
result.push(tb);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
const words = tb.text.split(/\s+/);
|
|
||||||
if (words.length <= 1) {
|
|
||||||
// Single word spanning columns — just assign to whichever col has more overlap
|
|
||||||
result.push(tb);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// For each column boundary crossing, find the best word-boundary split
|
|
||||||
let remainingWords = [...words];
|
|
||||||
let currentLeft = tb.bounds.left;
|
|
||||||
for (let col = leftCol; col <= rightCol && remainingWords.length > 0; col++) {
|
|
||||||
const colRight = col < xLines.length - 1 ? xLines[col + 1] : tb.bounds.right;
|
|
||||||
const segmentRight = Math.min(colRight, tb.bounds.right);
|
|
||||||
if (col === rightCol) {
|
|
||||||
// Last column — take all remaining words
|
|
||||||
result.push({
|
|
||||||
...tb,
|
|
||||||
id: `${tb.id}-split${col}`,
|
|
||||||
text: remainingWords.join(" "),
|
|
||||||
bounds: {
|
|
||||||
...tb.bounds,
|
|
||||||
left: currentLeft,
|
|
||||||
right: tb.bounds.right,
|
|
||||||
},
|
|
||||||
});
|
|
||||||
remainingWords = [];
|
|
||||||
} else {
|
|
||||||
// Find how many words fit in this column segment proportionally
|
|
||||||
const segmentWidth = segmentRight - currentLeft;
|
|
||||||
const fractionOfTotal = segmentWidth / totalWidth;
|
|
||||||
const approxChars = Math.round(fractionOfTotal * tb.text.length);
|
|
||||||
// Walk words to find the split closest to the proportional point
|
|
||||||
let charCount = 0;
|
|
||||||
let splitIdx = 0;
|
|
||||||
for (let w = 0; w < remainingWords.length; w++) {
|
|
||||||
const nextCount = charCount + remainingWords[w].length + (w > 0 ? 1 : 0);
|
|
||||||
if (nextCount > approxChars && splitIdx > 0) break;
|
|
||||||
charCount = nextCount;
|
|
||||||
splitIdx = w + 1;
|
|
||||||
}
|
|
||||||
if (splitIdx === 0) splitIdx = 1; // take at least one word
|
|
||||||
if (splitIdx >= remainingWords.length) {
|
|
||||||
// All remaining words fit here
|
|
||||||
result.push({
|
|
||||||
...tb,
|
|
||||||
id: `${tb.id}-split${col}`,
|
|
||||||
text: remainingWords.join(" "),
|
|
||||||
bounds: {
|
|
||||||
...tb.bounds,
|
|
||||||
left: currentLeft,
|
|
||||||
right: segmentRight,
|
|
||||||
},
|
|
||||||
});
|
|
||||||
remainingWords = [];
|
|
||||||
} else {
|
|
||||||
const partWords = remainingWords.slice(0, splitIdx);
|
|
||||||
result.push({
|
|
||||||
...tb,
|
|
||||||
id: `${tb.id}-split${col}`,
|
|
||||||
text: partWords.join(" "),
|
|
||||||
bounds: {
|
|
||||||
...tb.bounds,
|
|
||||||
left: currentLeft,
|
|
||||||
right: segmentRight,
|
|
||||||
},
|
|
||||||
});
|
|
||||||
remainingWords = remainingWords.slice(splitIdx);
|
|
||||||
currentLeft = segmentRight;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return result;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Full grid table (H + V lines)
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
function buildCells(rows: number, cols: number): TableCell[] {
|
|
||||||
const cells: TableCell[] = [];
|
|
||||||
for (let row = 0; row < rows; row++) {
|
|
||||||
for (let col = 0; col < cols; col++) {
|
|
||||||
cells.push({ row, col, text: "", rowSpan: 1, colSpan: 1 });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return cells;
|
|
||||||
}
|
|
||||||
|
|
||||||
function buildTableGrid(
|
|
||||||
pageNumber: number,
|
|
||||||
yLines: number[],
|
|
||||||
xLines: number[],
|
|
||||||
filteredSegments: Segment[],
|
|
||||||
textBoxes: TextBox[],
|
|
||||||
): { grid: TableGrid; consumedIds: string[] } {
|
|
||||||
let rows = yLines.length - 1;
|
|
||||||
const cols = xLines.length - 1;
|
|
||||||
const cells = buildCells(rows, cols);
|
|
||||||
const consumedIds: string[] = [];
|
|
||||||
const yMin = yLines[yLines.length - 1];
|
|
||||||
const yMax = yLines[0];
|
|
||||||
const xMin = xLines[0];
|
|
||||||
const xMax = xLines[xLines.length - 1];
|
|
||||||
// Split text boxes that span multiple columns before placement
|
|
||||||
const splitBoxes = splitCrossColumnBoxes(textBoxes, xLines);
|
|
||||||
// Track which split piece IDs get placed in cells, so we can consume
|
|
||||||
// the original (unsplit) text box IDs too.
|
|
||||||
const placedSplitIds = new Set<string>();
|
|
||||||
// Look for header text boxes just above the grid.
|
|
||||||
// Use the ORIGINAL (unsplit) text boxes for header detection so that
|
|
||||||
// wide paragraph text isn't falsely split into column-sized header chunks.
|
|
||||||
// Reject boxes wider than 1.5 columns — those are paragraph text, not headers.
|
|
||||||
const avgColWidth = (xMax - xMin) / cols;
|
|
||||||
const maxHeaderBoxWidth = avgColWidth * 1.5;
|
|
||||||
const headerBoxes = textBoxes.filter(tb => {
|
|
||||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
|
||||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
|
||||||
const boxWidth = tb.bounds.right - tb.bounds.left;
|
|
||||||
return cy > yMax && cy <= yMax + 20 && cx >= xMin && cx <= xMax && boxWidth <= maxHeaderBoxWidth;
|
|
||||||
});
|
|
||||||
if (headerBoxes.length > 0) {
|
|
||||||
rows += 1;
|
|
||||||
for (const cell of cells) cell.row += 1;
|
|
||||||
for (let col = 0; col < cols; col++) {
|
|
||||||
cells.push({ row: 0, col, text: "", rowSpan: 1, colSpan: 1 });
|
|
||||||
}
|
|
||||||
for (const tb of headerBoxes) {
|
|
||||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
|
||||||
const col = xLines.findIndex((lineX, idx) => {
|
|
||||||
const next = xLines[idx + 1];
|
|
||||||
return next !== undefined && cx >= lineX && cx <= next;
|
|
||||||
});
|
|
||||||
if (col >= 0 && col < cols) {
|
|
||||||
const cell = cells.find(c => c.row === 0 && c.col === col);
|
|
||||||
if (cell) {
|
|
||||||
cell.text = cell.text.length === 0 ? tb.text : `${cell.text} ${tb.text}`;
|
|
||||||
consumedIds.push(tb.id);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const cellBoxes = new Map<TableCell, TextBox[]>();
|
|
||||||
for (const tb of splitBoxes) {
|
|
||||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
|
||||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
|
||||||
if (cy < yMin || cy > yMax || cx < xMin || cx > xMax) continue;
|
|
||||||
const rays = castRaysForTextBox(tb, filteredSegments);
|
|
||||||
const rayConfidence = rays.filter(r => r.segmentId !== null).length;
|
|
||||||
let row = yLines.findIndex((lineY, idx) => {
|
|
||||||
const next = yLines[idx + 1];
|
|
||||||
return next !== undefined && cy <= lineY && cy >= next;
|
|
||||||
});
|
|
||||||
if (row < 0 || row >= (headerBoxes.length > 0 ? rows - 1 : rows)) continue;
|
|
||||||
if (headerBoxes.length > 0) row += 1;
|
|
||||||
const col = xLines.findIndex((lineX, idx) => {
|
|
||||||
const next = xLines[idx + 1];
|
|
||||||
return next !== undefined && cx >= lineX && cx <= next;
|
|
||||||
});
|
|
||||||
if (col < 0 || col >= cols) continue;
|
|
||||||
if (rayConfidence === 0) continue;
|
|
||||||
const cell = cells.find(c => c.row === row && c.col === col);
|
|
||||||
if (!cell) continue;
|
|
||||||
if (!cellBoxes.has(cell)) cellBoxes.set(cell, []);
|
|
||||||
cellBoxes.get(cell)?.push(tb);
|
|
||||||
consumedIds.push(tb.id);
|
|
||||||
if (tb.id.includes("-split")) placedSplitIds.add(tb.id);
|
|
||||||
}
|
|
||||||
rows = expandSubRowsByYClusters(rows, cols, cells, cellBoxes);
|
|
||||||
// Merge text boxes within each cell into cell text
|
|
||||||
for (const [cell, boxes] of cellBoxes.entries()) {
|
|
||||||
boxes.sort((a, b) => b.bounds.top - a.bounds.top);
|
|
||||||
const lines: string[] = [];
|
|
||||||
let currentLine: string[] = [];
|
|
||||||
let currentY = boxes[0].bounds.top;
|
|
||||||
for (const box of boxes) {
|
|
||||||
if (Math.abs(box.bounds.top - currentY) > 5) {
|
|
||||||
lines.push(currentLine.join(" "));
|
|
||||||
currentLine = [box.text];
|
|
||||||
currentY = box.bounds.top;
|
|
||||||
} else {
|
|
||||||
currentLine.push(box.text);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (currentLine.length > 0) lines.push(currentLine.join(" "));
|
|
||||||
cell.text = lines.join("<br>");
|
|
||||||
}
|
|
||||||
const grid = pruneEmptyRowsAndCols({
|
|
||||||
pageNumber,
|
|
||||||
rows,
|
|
||||||
cols,
|
|
||||||
cells,
|
|
||||||
warnings: [],
|
|
||||||
topY: yLines[0],
|
|
||||||
isBorderless: false,
|
|
||||||
});
|
|
||||||
// Also consume the original (unsplit) text box IDs when any of their
|
|
||||||
// split pieces were placed in a cell.
|
|
||||||
for (const splitId of placedSplitIds) {
|
|
||||||
const origId = splitId.replace(/-split\d+$/, "");
|
|
||||||
if (!consumedIds.includes(origId)) {
|
|
||||||
consumedIds.push(origId);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return { grid, consumedIds };
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// H-line-only table (inferred columns)
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
const COL_GAP_THRESHOLD = 20;
|
|
||||||
const HONLY_ROW_GAP = 30;
|
|
||||||
const HONLY_ROW_TOLERANCE = 8;
|
|
||||||
const MIN_TABLE_HEIGHT = 24;
|
|
||||||
const MIN_LEFT_SPREAD = 50;
|
|
||||||
|
|
||||||
function inferXLinesFromBoxes(textBoxes: TextBox[], xMin: number, xMax: number): number[] {
|
|
||||||
const centers = textBoxes.map(tb => (tb.bounds.left + tb.bounds.right) / 2).sort((a, b) => a - b);
|
|
||||||
if (centers.length === 0) return [xMin, xMax];
|
|
||||||
const boundaries = [xMin];
|
|
||||||
for (let i = 1; i < centers.length; i++) {
|
|
||||||
if (centers[i] - centers[i - 1] >= COL_GAP_THRESHOLD) {
|
|
||||||
boundaries.push((centers[i - 1] + centers[i]) / 2);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
boundaries.push(xMax);
|
|
||||||
return boundaries;
|
|
||||||
}
|
|
||||||
|
|
||||||
function buildHLineOnlyTable(
|
|
||||||
pageNumber: number,
|
|
||||||
yLines: number[],
|
|
||||||
xMin: number,
|
|
||||||
xMax: number,
|
|
||||||
textBoxes: TextBox[],
|
|
||||||
alreadyConsumed: Set<string>,
|
|
||||||
): { grid: TableGrid; consumedIds: string[] } | null {
|
|
||||||
const yMax = yLines[0];
|
|
||||||
const yMin = yLines[yLines.length - 1];
|
|
||||||
const candidates = textBoxes.filter(tb => !alreadyConsumed.has(tb.id));
|
|
||||||
const BOX_LEFT_TOLERANCE = 30;
|
|
||||||
const inRange = candidates.filter(tb => {
|
|
||||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
|
||||||
return (
|
|
||||||
tb.bounds.left >= xMin - BOX_LEFT_TOLERANCE &&
|
|
||||||
tb.bounds.right <= xMax + BOX_LEFT_TOLERANCE &&
|
|
||||||
cy >= yMin &&
|
|
||||||
cy <= yMax
|
|
||||||
);
|
|
||||||
});
|
|
||||||
// Extend downward below yMin
|
|
||||||
const belowYMin = candidates
|
|
||||||
.filter(tb => {
|
|
||||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
|
||||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
|
||||||
return cx >= xMin && cx <= xMax && cy < yMin;
|
|
||||||
})
|
|
||||||
.sort((a, b) => (b.bounds.top + b.bounds.bottom) / 2 - (a.bounds.top + a.bounds.bottom) / 2);
|
|
||||||
const extensionBoxes: TextBox[] = [];
|
|
||||||
let lastY = yMin;
|
|
||||||
for (const tb of belowYMin) {
|
|
||||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
|
||||||
if (lastY - cy > HONLY_ROW_GAP) break;
|
|
||||||
extensionBoxes.push(tb);
|
|
||||||
lastY = cy;
|
|
||||||
}
|
|
||||||
const allBoxes = [...inRange, ...extensionBoxes];
|
|
||||||
if (allBoxes.length === 0) return null;
|
|
||||||
const leftEdges = allBoxes.map(tb => tb.bounds.left);
|
|
||||||
if (Math.max(...leftEdges) - Math.min(...leftEdges) < MIN_LEFT_SPREAD) return null;
|
|
||||||
const xLines = inferXLinesFromBoxes(allBoxes, xMin, xMax);
|
|
||||||
if (xLines.length < 2) return null;
|
|
||||||
const cols = xLines.length - 1;
|
|
||||||
// Build visual rows
|
|
||||||
const visualRows: Array<{ midY: number; boxes: TextBox[] }> = [];
|
|
||||||
const sortedBoxes = [...allBoxes].sort((a, b) => {
|
|
||||||
const ya = (a.bounds.top + a.bounds.bottom) / 2;
|
|
||||||
const yb = (b.bounds.top + b.bounds.bottom) / 2;
|
|
||||||
if (Math.abs(ya - yb) > 0.5) return yb - ya;
|
|
||||||
return a.bounds.left - b.bounds.left;
|
|
||||||
});
|
|
||||||
for (const box of sortedBoxes) {
|
|
||||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
|
||||||
const last = visualRows[visualRows.length - 1];
|
|
||||||
if (last && Math.abs(last.midY - cy) <= HONLY_ROW_TOLERANCE) {
|
|
||||||
last.boxes.push(box);
|
|
||||||
} else {
|
|
||||||
visualRows.push({ midY: cy, boxes: [box] });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (visualRows.length === 0) return null;
|
|
||||||
const cells: TableCell[] = [];
|
|
||||||
const consumedIds: string[] = [];
|
|
||||||
for (let rowIdx = 0; rowIdx < visualRows.length; rowIdx++) {
|
|
||||||
const vrow = visualRows[rowIdx];
|
|
||||||
const colBoxes = new Map<number, TextBox[]>();
|
|
||||||
for (const box of vrow.boxes) {
|
|
||||||
const cx = (box.bounds.left + box.bounds.right) / 2;
|
|
||||||
const col = xLines.findIndex((lineX, idx) => {
|
|
||||||
const next = xLines[idx + 1];
|
|
||||||
return next !== undefined && cx >= lineX && cx <= next;
|
|
||||||
});
|
|
||||||
if (col >= 0 && col < cols) {
|
|
||||||
if (!colBoxes.has(col)) colBoxes.set(col, []);
|
|
||||||
colBoxes.get(col)?.push(box);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
for (let c = 0; c < cols; c++) {
|
|
||||||
const cbs = (colBoxes.get(c) ?? []).sort((a, b) => a.bounds.left - b.bounds.left);
|
|
||||||
cells.push({
|
|
||||||
row: rowIdx,
|
|
||||||
col: c,
|
|
||||||
text: cbs.map(b => b.text).join(" "),
|
|
||||||
rowSpan: 1,
|
|
||||||
colSpan: 1,
|
|
||||||
});
|
|
||||||
consumedIds.push(...cbs.map(b => b.id));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const contentTopY = visualRows.length > 0 ? visualRows[0].midY : yMax;
|
|
||||||
const grid = pruneEmptyRowsAndCols({
|
|
||||||
pageNumber,
|
|
||||||
rows: visualRows.length,
|
|
||||||
cols,
|
|
||||||
cells,
|
|
||||||
warnings: [],
|
|
||||||
topY: contentTopY,
|
|
||||||
isBorderless: false,
|
|
||||||
});
|
|
||||||
return { grid, consumedIds };
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Pruning
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
function pruneEmptyRowsAndCols(table: TableGrid): TableGrid {
|
|
||||||
const occupiedRows = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.row));
|
|
||||||
const occupiedCols = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.col));
|
|
||||||
if (occupiedRows.size === 0) return table;
|
|
||||||
const rowMap = new Map<number, number>();
|
|
||||||
let newRow = 0;
|
|
||||||
for (let r = 0; r < table.rows; r++) {
|
|
||||||
if (occupiedRows.has(r)) rowMap.set(r, newRow++);
|
|
||||||
}
|
|
||||||
const colMap = new Map<number, number>();
|
|
||||||
let newCol = 0;
|
|
||||||
for (let c = 0; c < table.cols; c++) {
|
|
||||||
if (occupiedCols.has(c)) colMap.set(c, newCol++);
|
|
||||||
}
|
|
||||||
const prunedCells = table.cells
|
|
||||||
.filter(c => occupiedRows.has(c.row) && occupiedCols.has(c.col))
|
|
||||||
.map(c => ({
|
|
||||||
...c,
|
|
||||||
row: rowMap.get(c.row) ?? c.row,
|
|
||||||
col: colMap.get(c.col) ?? c.col,
|
|
||||||
}));
|
|
||||||
return { ...table, rows: newRow, cols: newCol, cells: prunedCells };
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Diagram vs table discrimination
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/** Maximum column count for a plausible data table. */
|
|
||||||
const MAX_TABLE_COLS = 25;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Returns true if a grid looks like a vector diagram rather than a data table.
|
|
||||||
*
|
|
||||||
* Heuristics (any match → diagram):
|
|
||||||
* 1. Column count > 25 (diagrams create many X-lines from box edges)
|
|
||||||
* 2. Fill ratio < 25% (most cells empty — scattered boxes)
|
|
||||||
* 3. Fill < 50% AND duplicate text ratio > 30% (repeating labels in a
|
|
||||||
* diagram layout, e.g. "Hash", "Transaction" appearing in each column)
|
|
||||||
* 4. Fill < 50% AND cols >= 6 (moderate sparseness with wide grid)
|
|
||||||
*/
|
|
||||||
function isDiagram(grid: TableGrid): boolean {
|
|
||||||
const totalCells = grid.rows * grid.cols;
|
|
||||||
if (totalCells === 0) return true;
|
|
||||||
const filled = grid.cells.filter(c => c.text.trim().length > 0);
|
|
||||||
const fillRatio = filled.length / totalCells;
|
|
||||||
// Very high column count
|
|
||||||
if (grid.cols > MAX_TABLE_COLS) return true;
|
|
||||||
// Very sparse
|
|
||||||
if (fillRatio < 0.25) return true;
|
|
||||||
// Compute duplicate text ratio among non-trivial cells.
|
|
||||||
// Exclude short values (≤3 chars) like "—", "V", "YES", "NO" which
|
|
||||||
// naturally repeat in real data tables.
|
|
||||||
const substantive = filled.filter(c => c.text.trim().length > 3);
|
|
||||||
const uniqueTexts = new Set(substantive.map(c => c.text.trim())).size;
|
|
||||||
const dupRatio = substantive.length > 2 ? 1 - uniqueTexts / substantive.length : 0;
|
|
||||||
// Sparse + highly duplicated substantive text → repeating diagram
|
|
||||||
if (fillRatio < 0.5 && dupRatio > 0.3) return true;
|
|
||||||
// High duplication + wide grid → repeating diagram even at moderate fill
|
|
||||||
if (dupRatio > 0.4 && grid.cols >= 6) return true;
|
|
||||||
// Sparse + wide grid with no substantive text to judge
|
|
||||||
if (fillRatio < 0.4 && grid.cols >= 6) return true;
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Detect all table grids on a single page from its text boxes and segments.
|
|
||||||
*/
|
|
||||||
export function resolveTableGrids(pageNumber: number, textBoxes: TextBox[], segments: Segment[]): GridResult {
|
|
||||||
const vertical = segments.filter(s => Math.abs(s.x1 - s.x2) <= AXIS_EPSILON);
|
|
||||||
const horizontal = segments.filter(s => Math.abs(s.y1 - s.y2) <= AXIS_EPSILON);
|
|
||||||
// Filter segments to the text's visible area
|
|
||||||
const textYValues = textBoxes.flatMap(t => [t.bounds.bottom, t.bounds.top]);
|
|
||||||
const textYMin = textYValues.length > 0 ? Math.min(...textYValues) - PAGE_MARGIN : -Infinity;
|
|
||||||
const textYMax = textYValues.length > 0 ? Math.max(...textYValues) + PAGE_MARGIN : Infinity;
|
|
||||||
const textXValues = textBoxes.flatMap(t => [t.bounds.left, t.bounds.right]);
|
|
||||||
const textXMin = textXValues.length > 0 ? Math.min(...textXValues) - 100 : -Infinity;
|
|
||||||
const textXMax = textXValues.length > 0 ? Math.max(...textXValues) + 100 : Infinity;
|
|
||||||
const filteredH = horizontal.filter(
|
|
||||||
s => s.y1 >= textYMin && s.y1 <= textYMax && s.x1 <= textXMax && s.x2 >= textXMin,
|
|
||||||
);
|
|
||||||
const hMaxX2 = filteredH.length > 0 ? Math.max(...filteredH.map(s => s.x2)) : textXMax;
|
|
||||||
const vSegXMax = Math.max(textXMax, hMaxX2 + PAGE_MARGIN);
|
|
||||||
const filteredV = vertical.filter(s => {
|
|
||||||
const segMin = Math.min(s.y1, s.y2);
|
|
||||||
const segMax = Math.max(s.y1, s.y2);
|
|
||||||
return segMax >= textYMin && segMin <= textYMax && s.x1 >= textXMin && s.x1 <= vSegXMax;
|
|
||||||
});
|
|
||||||
const allYLines = uniqueSorted(filteredH.flatMap(s => [s.y1, s.y2])).sort((a, b) => b - a);
|
|
||||||
if (allYLines.length < 2) {
|
|
||||||
return { grids: [], consumedIds: [] };
|
|
||||||
}
|
|
||||||
const filteredSegments = [...filteredH, ...filteredV];
|
|
||||||
const yGroups = splitYLinesIntoGroups(allYLines, filteredV);
|
|
||||||
const grids: TableGrid[] = [];
|
|
||||||
const gridConsumedIds: string[][] = [];
|
|
||||||
// Flat set for the alreadyConsumed check in H-line-only tables
|
|
||||||
const allConsumedIds: string[] = [];
|
|
||||||
for (const yLines of yGroups) {
|
|
||||||
if (yLines.length < 2) continue;
|
|
||||||
const yMin = yLines[yLines.length - 1];
|
|
||||||
const yMax = yLines[0];
|
|
||||||
const groupVerticals = filteredV.filter(s => {
|
|
||||||
const segMin = Math.min(s.y1, s.y2);
|
|
||||||
const segMax = Math.max(s.y1, s.y2);
|
|
||||||
return segMin < yMax - 1.5 && segMax > yMin + 1.5;
|
|
||||||
});
|
|
||||||
const groupXLines = uniqueSorted(groupVerticals.flatMap(s => [s.x1, s.x2]));
|
|
||||||
if (groupXLines.length < 2) {
|
|
||||||
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
|
|
||||||
const groupHoriz = filteredH.filter(s => s.y1 >= yMin - 1.5 && s.y1 <= yMax + 1.5);
|
|
||||||
if (groupHoriz.length === 0) continue;
|
|
||||||
const hxMin = Math.min(...groupHoriz.map(s => s.x1));
|
|
||||||
const hxMax = Math.max(...groupHoriz.map(s => s.x2));
|
|
||||||
const result = buildHLineOnlyTable(pageNumber, yLines, hxMin, hxMax, textBoxes, new Set(allConsumedIds));
|
|
||||||
if (result) {
|
|
||||||
grids.push(result.grid);
|
|
||||||
gridConsumedIds.push(result.consumedIds);
|
|
||||||
allConsumedIds.push(...result.consumedIds);
|
|
||||||
}
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
|
|
||||||
const result = buildTableGrid(pageNumber, yLines, groupXLines, filteredSegments, textBoxes);
|
|
||||||
grids.push(result.grid);
|
|
||||||
gridConsumedIds.push(result.consumedIds);
|
|
||||||
allConsumedIds.push(...result.consumedIds);
|
|
||||||
}
|
|
||||||
// Filter out grids that look like vector diagrams, not data tables.
|
|
||||||
// Their consumed text box IDs are released so the text becomes free text.
|
|
||||||
const filteredGrids: TableGrid[] = [];
|
|
||||||
const filteredConsumedIds: string[] = [];
|
|
||||||
for (let i = 0; i < grids.length; i++) {
|
|
||||||
if (isDiagram(grids[i])) continue;
|
|
||||||
filteredGrids.push(grids[i]);
|
|
||||||
filteredConsumedIds.push(...gridConsumedIds[i]);
|
|
||||||
}
|
|
||||||
return { grids: filteredGrids, consumedIds: filteredConsumedIds };
|
|
||||||
}
|
|
||||||
@@ -1,106 +0,0 @@
|
|||||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Running header/footer detection and removal.
|
|
||||||
*
|
|
||||||
* Many PDFs have repeated text at the top or bottom of every page:
|
|
||||||
* document titles, chapter names, page numbers, copyright notices.
|
|
||||||
* These pollute the markdown output as false headings or noise.
|
|
||||||
*
|
|
||||||
* Algorithm:
|
|
||||||
* 1. For each page, bucket text boxes by Y position (top/bottom zones)
|
|
||||||
* 2. Collect the text content at each zone across all pages
|
|
||||||
* 3. Text appearing on >20% of pages OR 8+ consecutive pages is a
|
|
||||||
* running header/footer
|
|
||||||
* 4. Remove matching text boxes before further processing
|
|
||||||
*/
|
|
||||||
import type { PageContent } from "./types";
|
|
||||||
|
|
||||||
/** Minimum number of pages to enable header/footer detection. */
|
|
||||||
const MIN_PAGES = 5;
|
|
||||||
/** Minimum Y position for top zone (from bottom of page in PDF coords). */
|
|
||||||
const TOP_ZONE_MIN_Y = 700;
|
|
||||||
/** Maximum Y position for bottom zone. */
|
|
||||||
const BOTTOM_ZONE_MAX_Y = 80;
|
|
||||||
/**
|
|
||||||
* Minimum consecutive pages a text must appear on to be considered a
|
|
||||||
* running header/footer. Catches both document-wide headers (appearing
|
|
||||||
* on every page) and chapter-specific headers (appearing on 4+ consecutive
|
|
||||||
* pages within a chapter).
|
|
||||||
*/
|
|
||||||
const MIN_CONSECUTIVE_PAGES = 8;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Detect and remove running headers and footers from all pages.
|
|
||||||
* Mutates the pages array in place, removing header/footer text boxes.
|
|
||||||
*
|
|
||||||
* Uses two strategies:
|
|
||||||
* 1. Global frequency: text appearing on > 20% of all pages
|
|
||||||
* 2. Consecutive runs: text appearing on 8+ consecutive pages
|
|
||||||
*/
|
|
||||||
export function stripHeadersFooters(pages: PageContent[]): void {
|
|
||||||
if (pages.length < MIN_PAGES) return;
|
|
||||||
// Step 1: Build per-page zone text sets
|
|
||||||
const pageZoneTexts: Set<string>[] = [];
|
|
||||||
for (const page of pages) {
|
|
||||||
const zoneTexts = new Set<string>();
|
|
||||||
for (const tb of page.textBoxes) {
|
|
||||||
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
|
|
||||||
if (midY >= TOP_ZONE_MIN_Y || midY <= BOTTOM_ZONE_MAX_Y) {
|
|
||||||
const key = tb.text.trim().replace(/\s+/g, " ");
|
|
||||||
if (key.length > 0) zoneTexts.add(key);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
pageZoneTexts.push(zoneTexts);
|
|
||||||
}
|
|
||||||
// Step 2: Count global frequency AND longest consecutive run for each text
|
|
||||||
const globalCount = new Map<string, number>();
|
|
||||||
const maxConsecutive = new Map<string, number>();
|
|
||||||
// Collect all unique zone texts
|
|
||||||
const allTexts = new Set<string>();
|
|
||||||
for (const zts of pageZoneTexts) {
|
|
||||||
for (const t of zts) allTexts.add(t);
|
|
||||||
}
|
|
||||||
for (const text of allTexts) {
|
|
||||||
let total = 0;
|
|
||||||
let consecutive = 0;
|
|
||||||
let maxRun = 0;
|
|
||||||
for (const zts of pageZoneTexts) {
|
|
||||||
if (zts.has(text)) {
|
|
||||||
total++;
|
|
||||||
consecutive++;
|
|
||||||
if (consecutive > maxRun) maxRun = consecutive;
|
|
||||||
} else {
|
|
||||||
consecutive = 0;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
globalCount.set(text, total);
|
|
||||||
maxConsecutive.set(text, maxRun);
|
|
||||||
}
|
|
||||||
// Step 3: Identify running headers/footers
|
|
||||||
const globalThreshold = Math.max(3, Math.floor(pages.length * 0.2));
|
|
||||||
const repeatedTexts = new Set<string>();
|
|
||||||
for (const text of allTexts) {
|
|
||||||
const gc = globalCount.get(text) ?? 0;
|
|
||||||
const mc = maxConsecutive.get(text) ?? 0;
|
|
||||||
// Global: appears on 20%+ of pages
|
|
||||||
if (gc >= globalThreshold) {
|
|
||||||
repeatedTexts.add(text);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
// Consecutive: appears on 8+ consecutive pages (chapter-level headers)
|
|
||||||
if (mc >= MIN_CONSECUTIVE_PAGES) {
|
|
||||||
repeatedTexts.add(text);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (repeatedTexts.size === 0) return;
|
|
||||||
// Step 4: Remove matching text boxes from each page
|
|
||||||
for (const page of pages) {
|
|
||||||
page.textBoxes = page.textBoxes.filter(tb => {
|
|
||||||
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
|
|
||||||
if (midY < TOP_ZONE_MIN_Y && midY > BOTTOM_ZONE_MAX_Y) return true;
|
|
||||||
const normalized = tb.text.trim().replace(/\s+/g, " ");
|
|
||||||
return !repeatedTexts.has(normalized);
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,48 +1,10 @@
|
|||||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
import { pdfToMarkdown } from "@oh-my-pi/pi-natives";
|
||||||
|
|
||||||
/**
|
|
||||||
* PDF to Markdown converter.
|
|
||||||
*
|
|
||||||
* Uses mupdf (native WASM) for fast PDF parsing and a custom pipeline for
|
|
||||||
* table detection via vector line extraction + raycasting.
|
|
||||||
*
|
|
||||||
* Pipeline:
|
|
||||||
* 1. Extract text boxes + vector segments + image regions per page (mupdf)
|
|
||||||
* 2. Detect column layout (single vs multi-column)
|
|
||||||
* 3. Per column: detect table grids from segments (grid detection + raycasting)
|
|
||||||
* 4. Render diagrams as PNG files (if output directory provided)
|
|
||||||
* 5. Render tables as markdown tables, free text as paragraphs/headings
|
|
||||||
*/
|
|
||||||
import * as path from "node:path";
|
|
||||||
import type { ConversionResult, Converter, StreamInfo } from "../../types";
|
import type { ConversionResult, Converter, StreamInfo } from "../../types";
|
||||||
import { detectColumns } from "./columns";
|
|
||||||
import { extractPages, renderImageRegion } from "./extract";
|
|
||||||
import { resolveTableGrids } from "./grid";
|
|
||||||
import { stripHeadersFooters } from "./headers";
|
|
||||||
import { renderPageContent } from "./render";
|
|
||||||
import type { Segment, TextBox } from "./types";
|
|
||||||
|
|
||||||
const EXTENSIONS = [".pdf"];
|
const EXTENSIONS = [".pdf"];
|
||||||
const MIMETYPES = ["application/pdf", "application/x-pdf"];
|
const MIMETYPES = ["application/pdf", "application/x-pdf"];
|
||||||
|
|
||||||
type ImageBlock = { topY: number; markdown: string };
|
/** Converts PDF buffers to Markdown through the native `pdf-inspector` bridge. */
|
||||||
|
|
||||||
/**
|
|
||||||
* Process a set of text boxes (one column or full page): run table detection,
|
|
||||||
* separate free text, and render to markdown.
|
|
||||||
*/
|
|
||||||
function processColumn(
|
|
||||||
pageNumber: number,
|
|
||||||
textBoxes: TextBox[],
|
|
||||||
segments: Segment[],
|
|
||||||
imageBlocks: ImageBlock[],
|
|
||||||
): string {
|
|
||||||
const { grids, consumedIds } = resolveTableGrids(pageNumber, textBoxes, segments);
|
|
||||||
const consumedSet = new Set(consumedIds);
|
|
||||||
const freeTextBoxes = textBoxes.filter(tb => !consumedSet.has(tb.id));
|
|
||||||
return renderPageContent(freeTextBoxes, grids, imageBlocks, textBoxes);
|
|
||||||
}
|
|
||||||
|
|
||||||
export class PdfConverter implements Converter {
|
export class PdfConverter implements Converter {
|
||||||
name = "pdf";
|
name = "pdf";
|
||||||
|
|
||||||
@@ -56,91 +18,17 @@ export class PdfConverter implements Converter {
|
|||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
|
async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||||
const pdfBytes = new Uint8Array(input);
|
const result = await pdfToMarkdown(input);
|
||||||
const pages = await extractPages(pdfBytes);
|
const notice =
|
||||||
// Remove running headers/footers before processing.
|
result.pagesNeedingOcr.length > 0
|
||||||
stripHeadersFooters(pages);
|
? `Text extraction is incomplete for PDF pages ${result.pagesNeedingOcr.join(", ")}. Use the browser tool to render those pages or OCR them.`
|
||||||
const imageDir = streamInfo.imageDir;
|
: undefined;
|
||||||
|
|
||||||
const pageMarkdowns: string[] = [];
|
const conversion: ConversionResult = {
|
||||||
for (const page of pages) {
|
markdown: notice ? [result.markdown, notice].filter(Boolean).join("\n\n") : result.markdown,
|
||||||
// Build image blocks for this page.
|
};
|
||||||
const imageBlocks: ImageBlock[] = [];
|
if (result.title !== undefined) conversion.title = result.title;
|
||||||
if (imageDir && page.images.length > 0) {
|
return conversion;
|
||||||
for (const img of page.images) {
|
|
||||||
const filename = `${img.id}.png`;
|
|
||||||
const filepath = path.join(imageDir, filename);
|
|
||||||
try {
|
|
||||||
const png = await renderImageRegion(pdfBytes, img);
|
|
||||||
await Bun.write(filepath, png);
|
|
||||||
imageBlocks.push({ topY: img.topY, markdown: `` });
|
|
||||||
} catch {
|
|
||||||
// Image rendering failed — skip.
|
|
||||||
}
|
|
||||||
}
|
|
||||||
} else if (page.images.length > 0) {
|
|
||||||
for (const img of page.images) {
|
|
||||||
imageBlocks.push({
|
|
||||||
topY: img.topY,
|
|
||||||
markdown: `<!-- image: ${img.id} (page ${img.pageNumber}, ${img.bbox.w}x${img.bbox.h}pt) -->`,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
// Detect column layout.
|
|
||||||
// If the page has vertical segments (tables), suppress column detection
|
|
||||||
// when one detected column is very narrow — that's a table's first column,
|
|
||||||
// not a page layout column.
|
|
||||||
const layout = detectColumns(page.textBoxes);
|
|
||||||
if (layout.columnCount > 1 && page.segments.some(s => Math.abs(s.x1 - s.x2) <= 0.8)) {
|
|
||||||
const pageXMin = Math.min(...page.textBoxes.map(tb => tb.bounds.left));
|
|
||||||
const pageXMax = Math.max(...page.textBoxes.map(tb => tb.bounds.right));
|
|
||||||
const pageWidth = pageXMax - pageXMin;
|
|
||||||
const minColFraction = 0.3;
|
|
||||||
const tooNarrow = layout.columns.some(col => {
|
|
||||||
const colXMin = Math.min(...col.map(tb => tb.bounds.left));
|
|
||||||
const colXMax = Math.max(...col.map(tb => tb.bounds.right));
|
|
||||||
return (colXMax - colXMin) / pageWidth < minColFraction;
|
|
||||||
});
|
|
||||||
if (tooNarrow) {
|
|
||||||
layout.columnCount = 1;
|
|
||||||
layout.columns = [page.textBoxes];
|
|
||||||
layout.boundaries = [];
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
if (layout.columnCount === 1) {
|
|
||||||
// Single column — process normally.
|
|
||||||
const md = processColumn(page.pageNumber, page.textBoxes, page.segments, imageBlocks);
|
|
||||||
if (md.length > 0) pageMarkdowns.push(md);
|
|
||||||
} else {
|
|
||||||
// Multi-column — process each column independently, then join.
|
|
||||||
const columnMarkdowns: string[] = [];
|
|
||||||
for (const colBoxes of layout.columns) {
|
|
||||||
// Filter segments to those within this column's X range.
|
|
||||||
const colXMin = Math.min(...colBoxes.map(tb => tb.bounds.left));
|
|
||||||
const colXMax = Math.max(...colBoxes.map(tb => tb.bounds.right));
|
|
||||||
const margin = 10;
|
|
||||||
const colSegments = page.segments.filter(seg => {
|
|
||||||
const segXMin = Math.min(seg.x1, seg.x2);
|
|
||||||
const segXMax = Math.max(seg.x1, seg.x2);
|
|
||||||
return segXMax >= colXMin - margin && segXMin <= colXMax + margin;
|
|
||||||
});
|
|
||||||
// Images go with the first column only (no X info to split by).
|
|
||||||
const md = processColumn(
|
|
||||||
page.pageNumber,
|
|
||||||
colBoxes,
|
|
||||||
colSegments,
|
|
||||||
columnMarkdowns.length === 0 ? imageBlocks : [],
|
|
||||||
);
|
|
||||||
if (md.length > 0) columnMarkdowns.push(md);
|
|
||||||
}
|
|
||||||
const joined = columnMarkdowns.join("\n\n");
|
|
||||||
if (joined.length > 0) pageMarkdowns.push(joined);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
return { markdown: pageMarkdowns.join("\n\n") };
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,501 +0,0 @@
|
|||||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Markdown rendering for PDF pages.
|
|
||||||
*
|
|
||||||
* Converts table grids and free text boxes into markdown, handling:
|
|
||||||
* - Table grid → markdown table (`| col | col |`)
|
|
||||||
* - Free text → paragraphs with heading detection (by font size)
|
|
||||||
* - Content ordering (top-to-bottom via Y coordinate)
|
|
||||||
* - Paragraph wrap merging (lines broken across PDF line boundaries)
|
|
||||||
* - Page number removal
|
|
||||||
*
|
|
||||||
* Ported from @oharato/pdf2md-ts, stripped of CJK/TDnet-specific logic.
|
|
||||||
*/
|
|
||||||
import type { ContentBlock, TableGrid, TextBox } from "./types";
|
|
||||||
|
|
||||||
/** A free-text line grouped from horizontally adjacent text boxes. */
|
|
||||||
interface RenderLine {
|
|
||||||
text: string;
|
|
||||||
topY: number;
|
|
||||||
fontSize: number;
|
|
||||||
isBold: boolean;
|
|
||||||
isTabular: boolean;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** A content block carrying the Y of its last wrapped line during merging. */
|
|
||||||
type WrapBlock = ContentBlock & { lastTopY: number };
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Utility
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/** Convert full-width ASCII characters (A→A, !→! etc.) to normal ASCII. */
|
|
||||||
function normalizeFullWidthAscii(text: string): string {
|
|
||||||
return text.replace(/[!-~]/g, ch => String.fromCharCode(ch.charCodeAt(0) - 0xfee0));
|
|
||||||
}
|
|
||||||
|
|
||||||
function escapePipes(text: string): string {
|
|
||||||
return normalizeFullWidthAscii(text).replaceAll("|", "\\|").replaceAll("\n", "<br>");
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Parse a markdown pipe-delimited row into cell strings. */
|
|
||||||
function parsePipeRow(line: string): string[] {
|
|
||||||
const trimmed = line.trim();
|
|
||||||
if (!trimmed.startsWith("|") || !trimmed.endsWith("|")) return [];
|
|
||||||
return trimmed
|
|
||||||
.slice(1, -1)
|
|
||||||
.split("|")
|
|
||||||
.map(cell => cell.trim());
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Table rendering
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/**
|
|
||||||
* Render a TableGrid as a markdown table.
|
|
||||||
*/
|
|
||||||
export function renderTableToMarkdown(table: TableGrid): string {
|
|
||||||
if (table.rows === 0 || table.cols === 0) return "";
|
|
||||||
const matrix = Array.from({ length: table.rows }, () => Array.from({ length: table.cols }, () => ""));
|
|
||||||
for (const cell of table.cells) {
|
|
||||||
if (cell.row < table.rows && cell.col < table.cols) {
|
|
||||||
matrix[cell.row][cell.col] = escapePipes(cell.text.trim());
|
|
||||||
}
|
|
||||||
}
|
|
||||||
const normalized = normalizeShiftedSparseColumns(matrix);
|
|
||||||
const promoted = promoteSubHeaderPrefixes(normalized);
|
|
||||||
const header = `| ${promoted[0].join(" | ")} |`;
|
|
||||||
const divider = `| ${Array.from({ length: promoted[0].length }, () => "---").join(" | ")} |`;
|
|
||||||
const body = promoted
|
|
||||||
.slice(1)
|
|
||||||
.map(row => `| ${row.join(" | ")} |`)
|
|
||||||
.join("\n");
|
|
||||||
return [header, divider, body].filter(l => l.length > 0).join("\n");
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Fix tables with ≥5 columns where sparse single-value columns are
|
|
||||||
* misaligned. Shifts those values to the adjacent dense column and
|
|
||||||
* removes the now-empty sparse columns.
|
|
||||||
*/
|
|
||||||
function normalizeShiftedSparseColumns(matrix: string[][]): string[][] {
|
|
||||||
if (matrix.length === 0 || matrix[0].length < 5) return matrix;
|
|
||||||
const _rows = matrix.length;
|
|
||||||
const cols = matrix[0].length;
|
|
||||||
const counts = Array.from({ length: cols }, (_, c) =>
|
|
||||||
matrix.reduce((n, row) => n + (row[c].trim().length > 0 ? 1 : 0), 0),
|
|
||||||
);
|
|
||||||
const denseCols = new Set(
|
|
||||||
counts
|
|
||||||
.map((count, col) => ({ count, col }))
|
|
||||||
.filter(({ col, count }) => col === 0 || count >= 2)
|
|
||||||
.map(({ col }) => col),
|
|
||||||
);
|
|
||||||
const sparseCols = counts
|
|
||||||
.map((count, col) => ({ count, col }))
|
|
||||||
.filter(({ col, count }) => col > 0 && col < cols - 1 && count === 1)
|
|
||||||
.map(({ col }) => col);
|
|
||||||
if (sparseCols.length < 2 || denseCols.size < 4) return matrix;
|
|
||||||
const moves: Array<{ from: number; to: number; row: number }> = [];
|
|
||||||
for (const from of sparseCols) {
|
|
||||||
const row = matrix.findIndex(r => r[from].trim().length > 0);
|
|
||||||
const to = from + 1;
|
|
||||||
if (row < 0) return matrix;
|
|
||||||
if (!denseCols.has(to)) return matrix;
|
|
||||||
if (matrix[row][to].trim().length > 0) return matrix;
|
|
||||||
moves.push({ from, to, row });
|
|
||||||
}
|
|
||||||
const copy = matrix.map(row => [...row]);
|
|
||||||
for (const { from, to, row } of moves) {
|
|
||||||
copy[row][to] = copy[row][to].trim().length > 0 ? `${copy[row][to]} ${copy[row][from]}` : copy[row][from];
|
|
||||||
copy[row][from] = "";
|
|
||||||
}
|
|
||||||
const keepCols = Array.from({ length: cols }, (_, c) => c).filter(c => copy.some(row => row[c].trim().length > 0));
|
|
||||||
if (keepCols.length === cols) return copy;
|
|
||||||
return copy.map(row => keepCols.map(c => row[c]));
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* When a data row has ≥2 parenthesized qualifiers in non-first columns
|
|
||||||
* (and the first column is empty), promote them into the header row.
|
|
||||||
*/
|
|
||||||
function promoteSubHeaderPrefixes(matrix: string[][]): string[][] {
|
|
||||||
if (matrix.length < 2) return matrix;
|
|
||||||
const PAREN_RE = /^\([^)]{1,40}\)$/;
|
|
||||||
const result = matrix.map(row => [...row]);
|
|
||||||
const cols = matrix[0].length;
|
|
||||||
const rowsToRemove = new Set<number>();
|
|
||||||
for (let r = 1; r < result.length; r++) {
|
|
||||||
if (rowsToRemove.has(r)) continue;
|
|
||||||
const promotable: Array<{ col: number; prefix: string; isFullCell: boolean }> = [];
|
|
||||||
for (let col = 1; col < cols; col++) {
|
|
||||||
const cell = (result[r][col] ?? "").trim();
|
|
||||||
if (!cell) continue;
|
|
||||||
const parts = cell.split("<br>");
|
|
||||||
if (parts.length === 1 && PAREN_RE.test(cell)) {
|
|
||||||
promotable.push({ col, prefix: cell, isFullCell: true });
|
|
||||||
} else if (parts.length >= 2 && PAREN_RE.test(parts[0].trim())) {
|
|
||||||
promotable.push({
|
|
||||||
col,
|
|
||||||
prefix: parts[0].trim(),
|
|
||||||
isFullCell: false,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (promotable.length < 2) continue;
|
|
||||||
if (promotable.some(p => p.isFullCell) && result[r][0].trim().length > 0) continue;
|
|
||||||
for (const { col, prefix, isFullCell } of promotable) {
|
|
||||||
result[0][col] = result[0][col].trim() ? `${result[0][col]} ${prefix}` : prefix;
|
|
||||||
if (isFullCell) {
|
|
||||||
result[r][col] = "";
|
|
||||||
} else {
|
|
||||||
const parts = result[r][col].split("<br>");
|
|
||||||
result[r][col] = parts.slice(1).join("<br>");
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (result[r].every(cell => cell.trim().length === 0)) {
|
|
||||||
rowsToRemove.add(r);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return result.filter((_, r) => !rowsToRemove.has(r));
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Free text rendering
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/** Y tolerance for grouping text boxes onto the same visual line. */
|
|
||||||
const TEXT_LINE_Y_TOLERANCE = 3;
|
|
||||||
/** Minimum X gap between adjacent boxes to mark line as tabular. */
|
|
||||||
const TABULAR_X_GAP = 30;
|
|
||||||
/**
|
|
||||||
* Minimum font size (pts) to consider when computing the modal body font.
|
|
||||||
* Tiny labels from diagrams, footnote markers, and superscripts are excluded
|
|
||||||
* so they don't skew the modal toward small sizes.
|
|
||||||
*/
|
|
||||||
const MIN_BODY_FONT_SIZE = 7;
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Compute the most frequent font size among text boxes, ignoring very small
|
|
||||||
* text that likely comes from diagrams, footnotes, or superscripts.
|
|
||||||
*/
|
|
||||||
function modalFontSize(textBoxes: TextBox[]): number {
|
|
||||||
const counts = new Map<number, number>();
|
|
||||||
for (const tb of textBoxes) {
|
|
||||||
const size = Math.round((tb.fontSize ?? 0) * 10) / 10;
|
|
||||||
if (size < MIN_BODY_FONT_SIZE) continue;
|
|
||||||
counts.set(size, (counts.get(size) ?? 0) + 1);
|
|
||||||
}
|
|
||||||
let modal = 0;
|
|
||||||
let maxCount = 0;
|
|
||||||
for (const [size, count] of counts) {
|
|
||||||
if (count > maxCount) {
|
|
||||||
maxCount = count;
|
|
||||||
modal = size;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return modal;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Group free text boxes into horizontal lines, sorted top-to-bottom. */
|
|
||||||
function groupFreeTextIntoLines(textBoxes: TextBox[]): RenderLine[] {
|
|
||||||
if (textBoxes.length === 0) return [];
|
|
||||||
const sorted = [...textBoxes].sort((a, b) => {
|
|
||||||
const ya = (a.bounds.top + a.bounds.bottom) / 2;
|
|
||||||
const yb = (b.bounds.top + b.bounds.bottom) / 2;
|
|
||||||
const dy = yb - ya;
|
|
||||||
if (Math.abs(dy) > TEXT_LINE_Y_TOLERANCE) return dy;
|
|
||||||
return a.bounds.left - b.bounds.left;
|
|
||||||
});
|
|
||||||
const lines: RenderLine[] = [];
|
|
||||||
let curParts = [sorted[0].text];
|
|
||||||
let curBoxes = [sorted[0]];
|
|
||||||
let curY = (sorted[0].bounds.top + sorted[0].bounds.bottom) / 2;
|
|
||||||
let curTopY = curY;
|
|
||||||
let curFontSize = sorted[0].fontSize;
|
|
||||||
let curIsBold = sorted[0].isBold;
|
|
||||||
const finishLine = () => {
|
|
||||||
let isTabular = false;
|
|
||||||
for (let j = 1; j < curBoxes.length; j++) {
|
|
||||||
if (curBoxes[j].bounds.left - curBoxes[j - 1].bounds.right > TABULAR_X_GAP) {
|
|
||||||
isTabular = true;
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
lines.push({
|
|
||||||
text: curParts.join(" "),
|
|
||||||
topY: curTopY,
|
|
||||||
fontSize: curFontSize,
|
|
||||||
isBold: curIsBold,
|
|
||||||
isTabular,
|
|
||||||
});
|
|
||||||
};
|
|
||||||
for (let i = 1; i < sorted.length; i++) {
|
|
||||||
const box = sorted[i];
|
|
||||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
|
||||||
if (Math.abs(cy - curY) <= TEXT_LINE_Y_TOLERANCE) {
|
|
||||||
curParts.push(box.text);
|
|
||||||
curBoxes.push(box);
|
|
||||||
curFontSize = Math.max(curFontSize, box.fontSize);
|
|
||||||
curIsBold = curIsBold || box.isBold;
|
|
||||||
} else {
|
|
||||||
finishLine();
|
|
||||||
curParts = [box.text];
|
|
||||||
curBoxes = [box];
|
|
||||||
curY = cy;
|
|
||||||
curTopY = cy;
|
|
||||||
curFontSize = box.fontSize;
|
|
||||||
curIsBold = box.isBold;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
finishLine();
|
|
||||||
return lines;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Determine markdown heading prefix based on font size relative to body. */
|
|
||||||
function headingPrefix(fontSize: number, bodyFontSize: number, isBold: boolean): string {
|
|
||||||
if (bodyFontSize <= 0) return "";
|
|
||||||
const ratio = fontSize / bodyFontSize;
|
|
||||||
// Large headings (>2x body size)
|
|
||||||
if (ratio >= 2.0) return "# ";
|
|
||||||
// Medium headings (~1.5x body size)
|
|
||||||
if (ratio >= 1.4) return "## ";
|
|
||||||
// Small headings (bold and slightly larger)
|
|
||||||
if (ratio >= 1.1 && isBold) return "### ";
|
|
||||||
return "";
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Block merging
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/** Merge consecutive blocks with the same heading prefix (wrapped headings). */
|
|
||||||
function mergeConsecutiveHeadings(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
|
|
||||||
if (blocks.length === 0) return [];
|
|
||||||
const HEADING_RE = /^(#{1,6} )/;
|
|
||||||
const maxGap = Math.max(bodyFS * 3, 30);
|
|
||||||
const merged: ContentBlock[] = [];
|
|
||||||
let cur: ContentBlock = { ...blocks[0] };
|
|
||||||
for (let i = 1; i < blocks.length; i++) {
|
|
||||||
const next = blocks[i];
|
|
||||||
const curMatch = cur.content.match(HEADING_RE);
|
|
||||||
const nextMatch = next.content.match(HEADING_RE);
|
|
||||||
const gap = cur.topY - next.topY;
|
|
||||||
if (curMatch && nextMatch && curMatch[1] === nextMatch[1] && gap <= maxGap) {
|
|
||||||
cur = {
|
|
||||||
topY: cur.topY,
|
|
||||||
content: `${cur.content} ${next.content.slice(nextMatch[1].length)}`,
|
|
||||||
isTabular: cur.isTabular || next.isTabular,
|
|
||||||
};
|
|
||||||
} else {
|
|
||||||
merged.push(cur);
|
|
||||||
cur = { ...next };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
merged.push(cur);
|
|
||||||
return merged;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Merge consecutive plain-text blocks that are wrapped lines of the same paragraph.
|
|
||||||
*/
|
|
||||||
function mergeParagraphWraps(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
|
|
||||||
if (blocks.length === 0 || bodyFS <= 0) return blocks;
|
|
||||||
const HEADING_RE = /^#{1,6} /;
|
|
||||||
const SENTENCE_END_RE = /[.!?…)\]]\s*$/;
|
|
||||||
const maxGap = bodyFS * 2.0;
|
|
||||||
const MIN_WRAP_LENGTH = 25;
|
|
||||||
const merged: ContentBlock[] = [];
|
|
||||||
let cur: WrapBlock = { ...blocks[0], lastTopY: blocks[0].topY };
|
|
||||||
for (let i = 1; i < blocks.length; i++) {
|
|
||||||
const next = blocks[i];
|
|
||||||
const curIsBody = !HEADING_RE.test(cur.content) && !cur.content.startsWith("|");
|
|
||||||
const nextIsBody = !HEADING_RE.test(next.content) && !next.content.startsWith("|");
|
|
||||||
const gap = cur.lastTopY - next.topY;
|
|
||||||
const isWrap =
|
|
||||||
curIsBody &&
|
|
||||||
nextIsBody &&
|
|
||||||
!cur.isTabular &&
|
|
||||||
!next.isTabular &&
|
|
||||||
gap > 0 &&
|
|
||||||
gap <= maxGap &&
|
|
||||||
cur.content.length > MIN_WRAP_LENGTH &&
|
|
||||||
!SENTENCE_END_RE.test(cur.content);
|
|
||||||
if (isWrap) {
|
|
||||||
cur = {
|
|
||||||
topY: cur.topY,
|
|
||||||
lastTopY: next.topY,
|
|
||||||
content: `${cur.content.trimEnd()} ${next.content.trimStart()}`,
|
|
||||||
isTabular: false,
|
|
||||||
};
|
|
||||||
} else {
|
|
||||||
merged.push({ topY: cur.topY, content: cur.content });
|
|
||||||
cur = { ...next, lastTopY: next.topY };
|
|
||||||
}
|
|
||||||
}
|
|
||||||
merged.push({ topY: cur.topY, content: cur.content });
|
|
||||||
return merged;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Remove page number blocks near the bottom of the page. */
|
|
||||||
function removePageNumbers(blocks: ContentBlock[]): ContentBlock[] {
|
|
||||||
const PAGE_NUM_RE = /^(?:#{1,6}\s*)?\d+\s*$/;
|
|
||||||
const BOTTOM_Y = 120;
|
|
||||||
return blocks.filter((block, idx) => {
|
|
||||||
const isBottom = idx >= blocks.length - 3;
|
|
||||||
const isLowY = block.topY <= BOTTOM_Y;
|
|
||||||
const isPageNum = PAGE_NUM_RE.test(block.content.trim());
|
|
||||||
return !(isBottom && isLowY && isPageNum);
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Detached first-column table reconstruction
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/**
|
|
||||||
* Fix tables where the first column was emitted as free text blocks
|
|
||||||
* around a markdown table containing only the right-side columns.
|
|
||||||
*
|
|
||||||
* Detects: a plain-text header line with (N+1) tokens above an N-column
|
|
||||||
* markdown table, plus short label lines whose count matches the table's
|
|
||||||
* logical row count. Reconstructs into a proper (N+1)-column table.
|
|
||||||
*/
|
|
||||||
function normalizeDetachedFirstColumnTables(blocks: ContentBlock[]): ContentBlock[] {
|
|
||||||
const HEADING_RE = /^#{1,6}\s/;
|
|
||||||
const isTableBlock = (text: string) => text.trimStart().startsWith("|");
|
|
||||||
const isPlainBlock = (text: string) => !HEADING_RE.test(text) && !isTableBlock(text);
|
|
||||||
const isShortLabel = (text: string) => {
|
|
||||||
const t = text.trim();
|
|
||||||
return t.length > 0 && t.length <= 40;
|
|
||||||
};
|
|
||||||
const splitTokens = (text: string) =>
|
|
||||||
text
|
|
||||||
.trim()
|
|
||||||
.split(/[ \t]+/)
|
|
||||||
.filter(Boolean);
|
|
||||||
const replacements = new Map<number, string>();
|
|
||||||
const remove = new Set<number>();
|
|
||||||
for (let tableIdx = 0; tableIdx < blocks.length; tableIdx++) {
|
|
||||||
if (remove.has(tableIdx)) continue;
|
|
||||||
const tableBlock = blocks[tableIdx];
|
|
||||||
if (!isTableBlock(tableBlock.content)) continue;
|
|
||||||
const tableLines = tableBlock.content
|
|
||||||
.split("\n")
|
|
||||||
.map(line => line.trim())
|
|
||||||
.filter(line => line.startsWith("|"));
|
|
||||||
const dataRows = tableLines
|
|
||||||
.filter(line => !/^\|\s*[-: ]+\|/.test(line))
|
|
||||||
.map(parsePipeRow)
|
|
||||||
.filter(row => row.length > 0);
|
|
||||||
if (dataRows.length === 0) continue;
|
|
||||||
const cols = dataRows[0].length;
|
|
||||||
if (cols < 2 || dataRows.some(row => row.length !== cols)) continue;
|
|
||||||
// Expand by <br> count to get logical row count
|
|
||||||
const logicalRows: string[][] = [];
|
|
||||||
for (const row of dataRows) {
|
|
||||||
const splitCells = row.map(cell => cell.split("<br>").map(p => p.trim()));
|
|
||||||
const rowSpan = Math.max(...splitCells.map(parts => parts.length));
|
|
||||||
for (let k = 0; k < rowSpan; k++) {
|
|
||||||
logicalRows.push(splitCells.map(parts => parts[k] ?? ""));
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (logicalRows.length < 2) continue;
|
|
||||||
// Find header with (cols + 1) non-numeric tokens
|
|
||||||
let headerIdx = -1;
|
|
||||||
let headerTokens: string[] = [];
|
|
||||||
for (let i = Math.max(0, tableIdx - 4); i <= tableIdx - 1; i++) {
|
|
||||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
|
||||||
if (!isPlainBlock(text)) continue;
|
|
||||||
const tokens = splitTokens(text);
|
|
||||||
if (tokens.length === cols + 1 && tokens.every(tok => !/[0-9]/.test(tok))) {
|
|
||||||
headerIdx = i;
|
|
||||||
headerTokens = tokens;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (headerIdx < 0) continue;
|
|
||||||
// Collect short label lines above/below table
|
|
||||||
const aboveLabels: Array<{ idx: number; text: string }> = [];
|
|
||||||
for (let i = tableIdx - 1; i > headerIdx; i--) {
|
|
||||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
|
||||||
if (!isPlainBlock(text) || !isShortLabel(text)) break;
|
|
||||||
aboveLabels.push({ idx: i, text });
|
|
||||||
}
|
|
||||||
aboveLabels.reverse();
|
|
||||||
const belowLabels: Array<{ idx: number; text: string }> = [];
|
|
||||||
for (let i = tableIdx + 1; i < blocks.length; i++) {
|
|
||||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
|
||||||
if (!isPlainBlock(text) || !isShortLabel(text)) break;
|
|
||||||
belowLabels.push({ idx: i, text });
|
|
||||||
}
|
|
||||||
const labels = [...aboveLabels, ...belowLabels];
|
|
||||||
if (labels.length !== logicalRows.length) continue;
|
|
||||||
// Reconstruct the full table
|
|
||||||
const normalizedLines: string[] = [];
|
|
||||||
normalizedLines.push(`| ${headerTokens.join(" | ")} |`);
|
|
||||||
normalizedLines.push(`| ${Array.from({ length: cols + 1 }, () => "---").join(" | ")} |`);
|
|
||||||
for (let r = 0; r < logicalRows.length; r++) {
|
|
||||||
normalizedLines.push(`| ${labels[r].text} | ${logicalRows[r].join(" | ")} |`);
|
|
||||||
}
|
|
||||||
replacements.set(tableIdx, normalizedLines.join("\n"));
|
|
||||||
remove.add(headerIdx);
|
|
||||||
for (const label of labels) remove.add(label.idx);
|
|
||||||
}
|
|
||||||
if (replacements.size === 0 && remove.size === 0) return blocks;
|
|
||||||
const out: ContentBlock[] = [];
|
|
||||||
for (let i = 0; i < blocks.length; i++) {
|
|
||||||
if (remove.has(i)) continue;
|
|
||||||
const replaced = replacements.get(i);
|
|
||||||
if (replaced) {
|
|
||||||
out.push({ topY: blocks[i].topY, content: replaced });
|
|
||||||
} else {
|
|
||||||
out.push(blocks[i]);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return out;
|
|
||||||
}
|
|
||||||
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
// Public API
|
|
||||||
// ---------------------------------------------------------------------------
|
|
||||||
/**
|
|
||||||
* Render one page's content: free text and tables interleaved top-to-bottom.
|
|
||||||
*/
|
|
||||||
export function renderPageContent(
|
|
||||||
freeTextBoxes: TextBox[],
|
|
||||||
tables: TableGrid[],
|
|
||||||
imageBlocks: Array<{ topY: number; markdown: string }> = [],
|
|
||||||
allTextBoxes?: TextBox[],
|
|
||||||
): string {
|
|
||||||
const blocks: ContentBlock[] = [];
|
|
||||||
// Use ALL text boxes (before table/diagram filtering) for modal font size,
|
|
||||||
// so that diagram labels released as free text don't skew the body size.
|
|
||||||
const bodyFS = modalFontSize(allTextBoxes ?? freeTextBoxes);
|
|
||||||
// Free text lines
|
|
||||||
for (const line of groupFreeTextIntoLines(freeTextBoxes)) {
|
|
||||||
const prefix = headingPrefix(line.fontSize, bodyFS, line.isBold);
|
|
||||||
blocks.push({
|
|
||||||
topY: line.topY,
|
|
||||||
content: prefix + line.text,
|
|
||||||
isTabular: prefix === "" && line.isTabular,
|
|
||||||
});
|
|
||||||
}
|
|
||||||
// Tables
|
|
||||||
for (const table of tables) {
|
|
||||||
const md = renderTableToMarkdown(table);
|
|
||||||
if (md.length > 0) {
|
|
||||||
blocks.push({ topY: table.topY, content: md });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
// Images
|
|
||||||
for (const img of imageBlocks) {
|
|
||||||
blocks.push({ topY: img.topY, content: img.markdown });
|
|
||||||
}
|
|
||||||
// Sort top-to-bottom (higher Y = higher on page = comes first)
|
|
||||||
blocks.sort((a, b) => b.topY - a.topY);
|
|
||||||
const cleaned = removePageNumbers(blocks);
|
|
||||||
const headingsMerged = mergeConsecutiveHeadings(cleaned, bodyFS);
|
|
||||||
const merged = mergeParagraphWraps(headingsMerged, bodyFS);
|
|
||||||
const normalized = normalizeDetachedFirstColumnTables(merged);
|
|
||||||
return normalized
|
|
||||||
.map(b => b.content)
|
|
||||||
.join("\n\n")
|
|
||||||
.trim();
|
|
||||||
}
|
|
||||||
@@ -1,84 +0,0 @@
|
|||||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
|
||||||
|
|
||||||
/** Bounding box in PDF coordinate space (origin = bottom-left). */
|
|
||||||
export type Bounds = {
|
|
||||||
left: number;
|
|
||||||
right: number;
|
|
||||||
/** Higher value = higher on the page. */
|
|
||||||
top: number;
|
|
||||||
bottom: number;
|
|
||||||
};
|
|
||||||
|
|
||||||
/** A text fragment with position and font metadata. */
|
|
||||||
export type TextBox = {
|
|
||||||
id: string;
|
|
||||||
text: string;
|
|
||||||
bounds: Bounds;
|
|
||||||
pageNumber: number;
|
|
||||||
/** Dominant font size in points. */
|
|
||||||
fontSize: number;
|
|
||||||
/** True if rendered bold (font name or rendering mode). */
|
|
||||||
isBold: boolean;
|
|
||||||
};
|
|
||||||
|
|
||||||
/** A horizontal or vertical line segment extracted from vector graphics. */
|
|
||||||
export type Segment = {
|
|
||||||
id: string;
|
|
||||||
x1: number;
|
|
||||||
y1: number;
|
|
||||||
x2: number;
|
|
||||||
y2: number;
|
|
||||||
};
|
|
||||||
|
|
||||||
/** A single cell in a resolved table grid. */
|
|
||||||
export type TableCell = {
|
|
||||||
row: number;
|
|
||||||
col: number;
|
|
||||||
text: string;
|
|
||||||
rowSpan: number;
|
|
||||||
colSpan: number;
|
|
||||||
};
|
|
||||||
|
|
||||||
/** A resolved table grid ready for markdown rendering. */
|
|
||||||
export type TableGrid = {
|
|
||||||
pageNumber: number;
|
|
||||||
rows: number;
|
|
||||||
cols: number;
|
|
||||||
cells: TableCell[];
|
|
||||||
warnings: string[];
|
|
||||||
/** Top Y coordinate (PDF space: larger = higher on page). */
|
|
||||||
topY: number;
|
|
||||||
/** True for tables detected without vector borders. */
|
|
||||||
isBorderless: boolean;
|
|
||||||
};
|
|
||||||
|
|
||||||
/** An image/diagram region detected on a page. */
|
|
||||||
export type ImageRegion = {
|
|
||||||
id: string;
|
|
||||||
pageNumber: number;
|
|
||||||
/** Bounding box in mupdf coordinates (top-left origin). */
|
|
||||||
bbox: {
|
|
||||||
x: number;
|
|
||||||
y: number;
|
|
||||||
w: number;
|
|
||||||
h: number;
|
|
||||||
};
|
|
||||||
/** Y position in PDF coordinates (bottom-left) for ordering. */
|
|
||||||
topY: number;
|
|
||||||
};
|
|
||||||
|
|
||||||
/** Result of extracting content from a single PDF page. */
|
|
||||||
export type PageContent = {
|
|
||||||
pageNumber: number;
|
|
||||||
textBoxes: TextBox[];
|
|
||||||
segments: Segment[];
|
|
||||||
images: ImageRegion[];
|
|
||||||
};
|
|
||||||
|
|
||||||
/** A block of rendered content (text paragraph or table). */
|
|
||||||
export type ContentBlock = {
|
|
||||||
topY: number;
|
|
||||||
content: string;
|
|
||||||
/** True if this line has wide gaps between text boxes (column headers). */
|
|
||||||
isTabular?: boolean;
|
|
||||||
};
|
|
||||||
@@ -1,250 +0,0 @@
|
|||||||
import * as fs from "node:fs/promises";
|
|
||||||
import * as os from "node:os";
|
|
||||||
import * as path from "node:path";
|
|
||||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
|
||||||
import { isEexist, isEnotempty, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
|
|
||||||
import type { ToolSession } from "../sdk";
|
|
||||||
import { loadImageInput, MAX_IMAGE_INPUT_BYTES, webpExclusionForModel } from "../utils/image-loading";
|
|
||||||
import { convertFileWithMarkit } from "../utils/markit";
|
|
||||||
import type { ReadToolDetails } from "./read";
|
|
||||||
import { prependSuffixResolutionNotice } from "./read-format";
|
|
||||||
import { isNotFoundError } from "./read-path-resolution";
|
|
||||||
import { formatBytes } from "./render-utils";
|
|
||||||
import { ToolError } from "./tool-errors";
|
|
||||||
import { toolResult } from "./tool-result";
|
|
||||||
|
|
||||||
const MAX_IMAGE_SIZE = MAX_IMAGE_INPUT_BYTES;
|
|
||||||
|
|
||||||
const PDF_IMAGE_PLACEHOLDER_RE = /<!--\s*image:\s*([^\s<>]+)(.*?)-->/g;
|
|
||||||
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
|
|
||||||
const PDF_IMAGE_MEMBER_EXTENSION_RE = /\.png$/i;
|
|
||||||
const PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH = 96;
|
|
||||||
|
|
||||||
interface PdfImageSnapshot {
|
|
||||||
directory: string;
|
|
||||||
filePath: string;
|
|
||||||
digest: string;
|
|
||||||
}
|
|
||||||
|
|
||||||
interface PdfImageExtraction {
|
|
||||||
controller: AbortController;
|
|
||||||
promise: Promise<string>;
|
|
||||||
settled: boolean;
|
|
||||||
waiters: number;
|
|
||||||
}
|
|
||||||
|
|
||||||
const pdfImageExtractions = new Map<string, PdfImageExtraction>();
|
|
||||||
|
|
||||||
function pdfImageMemberPath(pdfPath: string, imageId: string): string {
|
|
||||||
const member = PDF_IMAGE_MEMBER_EXTENSION_RE.test(imageId) ? imageId : `${imageId}.png`;
|
|
||||||
return `${pdfPath}:${member}`;
|
|
||||||
}
|
|
||||||
|
|
||||||
export function rewritePdfImagePlaceholders(markdown: string, pdfPath: string): string {
|
|
||||||
return markdown.replace(PDF_IMAGE_PLACEHOLDER_RE, (_match: string, imageId: string, metadataText: string) => {
|
|
||||||
const metadata = metadataText.trim();
|
|
||||||
const suffix = metadata.length > 0 ? ` (${metadata})` : "";
|
|
||||||
return `Image ${imageId}${suffix}: read \`${pdfImageMemberPath(pdfPath, imageId)}\``;
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
export function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; member: string } | null {
|
|
||||||
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
|
|
||||||
if (!match) return null;
|
|
||||||
const pdfPath = match[1];
|
|
||||||
const member = match[2];
|
|
||||||
if (pdfPath === undefined || member === undefined) return null;
|
|
||||||
if (member.length !== 0 && !PDF_IMAGE_MEMBER_EXTENSION_RE.test(member)) return null;
|
|
||||||
return { pdfPath, member };
|
|
||||||
}
|
|
||||||
function pdfImageCacheDir(session: ToolSession, absolutePdfPath: string, contentDigest: string): string {
|
|
||||||
const artifactsDir = session.getArtifactsDir?.();
|
|
||||||
let root = artifactsDir ?? undefined;
|
|
||||||
if (root === undefined) {
|
|
||||||
const sessionFile = session.getSessionFile();
|
|
||||||
root = sessionFile?.endsWith(".jsonl") ? sessionFile.slice(0, -6) : path.join(os.tmpdir(), "omp-read-pdf-images");
|
|
||||||
}
|
|
||||||
const basename = path
|
|
||||||
.basename(absolutePdfPath)
|
|
||||||
.replace(/[^A-Za-z0-9._-]/g, "_")
|
|
||||||
.slice(0, PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH);
|
|
||||||
const pathDigest = Bun.hash(absolutePdfPath).toString(36);
|
|
||||||
return path.join(root, "read-pdf-images", `${basename}-${pathDigest}-${contentDigest}`);
|
|
||||||
}
|
|
||||||
|
|
||||||
async function snapshotPdfSource(absolutePdfPath: string, signal?: AbortSignal): Promise<PdfImageSnapshot> {
|
|
||||||
const directory = await fs.mkdtemp(path.join(os.tmpdir(), "omp-read-pdf-"));
|
|
||||||
try {
|
|
||||||
const bytes = await untilAborted(signal, () => Bun.file(absolutePdfPath).bytes());
|
|
||||||
signal?.throwIfAborted();
|
|
||||||
const digest = new Bun.CryptoHasher("sha256").update(bytes).digest("hex");
|
|
||||||
const filePath = path.join(directory, "source.pdf");
|
|
||||||
await Bun.write(filePath, bytes);
|
|
||||||
signal?.throwIfAborted();
|
|
||||||
return { directory, filePath, digest };
|
|
||||||
} catch (error) {
|
|
||||||
await fs.rm(directory, { recursive: true, force: true });
|
|
||||||
throw error;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async function listPdfImageMembers(imageDir: string): Promise<string[]> {
|
|
||||||
try {
|
|
||||||
const entries = await fs.readdir(imageDir, { withFileTypes: true });
|
|
||||||
const members: string[] = [];
|
|
||||||
for (const entry of entries) {
|
|
||||||
if (entry.isFile() && PDF_IMAGE_MEMBER_EXTENSION_RE.test(entry.name)) members.push(entry.name);
|
|
||||||
}
|
|
||||||
return members.sort();
|
|
||||||
} catch (error) {
|
|
||||||
if (isNotFoundError(error)) return [];
|
|
||||||
throw error;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async function extractPdfImages(snapshot: PdfImageSnapshot, imageDir: string, signal: AbortSignal): Promise<string> {
|
|
||||||
const markerPath = path.join(imageDir, ".extracted");
|
|
||||||
try {
|
|
||||||
await fs.stat(markerPath);
|
|
||||||
return imageDir;
|
|
||||||
} catch (error) {
|
|
||||||
if (!isNotFoundError(error)) throw error;
|
|
||||||
}
|
|
||||||
|
|
||||||
await fs.mkdir(path.dirname(imageDir), { recursive: true });
|
|
||||||
const stagingDir = await fs.mkdtemp(`${imageDir}.tmp-`);
|
|
||||||
let published = false;
|
|
||||||
try {
|
|
||||||
const result = await convertFileWithMarkit(snapshot.filePath, signal, { imageDir: stagingDir });
|
|
||||||
if (!result.ok) {
|
|
||||||
throw new ToolError(`Cannot extract images from PDF: ${result.error ?? "conversion failed"}`);
|
|
||||||
}
|
|
||||||
await Bun.write(path.join(stagingDir, ".extracted"), "ok");
|
|
||||||
try {
|
|
||||||
await fs.rename(stagingDir, imageDir);
|
|
||||||
published = true;
|
|
||||||
} catch (error) {
|
|
||||||
if (!isEexist(error) && !isEnotempty(error)) throw error;
|
|
||||||
try {
|
|
||||||
await fs.stat(markerPath);
|
|
||||||
} catch (markerError) {
|
|
||||||
if (isNotFoundError(markerError)) throw error;
|
|
||||||
throw markerError;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return imageDir;
|
|
||||||
} finally {
|
|
||||||
if (!published) await fs.rm(stagingDir, { recursive: true, force: true });
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
function createPdfImageExtraction(snapshot: PdfImageSnapshot, imageDir: string): PdfImageExtraction {
|
|
||||||
const controller = new AbortController();
|
|
||||||
const promise = extractPdfImages(snapshot, imageDir, controller.signal).finally(() =>
|
|
||||||
fs.rm(snapshot.directory, { recursive: true, force: true }),
|
|
||||||
);
|
|
||||||
const extraction: PdfImageExtraction = { controller, promise, settled: false, waiters: 0 };
|
|
||||||
const settle = () => {
|
|
||||||
extraction.settled = true;
|
|
||||||
if (pdfImageExtractions.get(imageDir) === extraction) pdfImageExtractions.delete(imageDir);
|
|
||||||
};
|
|
||||||
void promise.then(settle, settle);
|
|
||||||
return extraction;
|
|
||||||
}
|
|
||||||
|
|
||||||
async function waitForPdfImageExtraction(
|
|
||||||
extraction: PdfImageExtraction,
|
|
||||||
signal: AbortSignal | undefined,
|
|
||||||
): Promise<string> {
|
|
||||||
extraction.waiters++;
|
|
||||||
try {
|
|
||||||
return await untilAborted(signal, extraction.promise);
|
|
||||||
} finally {
|
|
||||||
extraction.waiters--;
|
|
||||||
if (extraction.waiters === 0 && !extraction.settled) {
|
|
||||||
extraction.controller.abort();
|
|
||||||
try {
|
|
||||||
await extraction.promise;
|
|
||||||
} catch {}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
async function ensurePdfImageCache(
|
|
||||||
session: ToolSession,
|
|
||||||
absolutePdfPath: string,
|
|
||||||
signal?: AbortSignal,
|
|
||||||
): Promise<string> {
|
|
||||||
const snapshot = await snapshotPdfSource(absolutePdfPath, signal);
|
|
||||||
const imageDir = pdfImageCacheDir(session, absolutePdfPath, snapshot.digest);
|
|
||||||
const existing = pdfImageExtractions.get(imageDir);
|
|
||||||
if (existing && !existing.settled && !existing.controller.signal.aborted) {
|
|
||||||
await fs.rm(snapshot.directory, { recursive: true, force: true });
|
|
||||||
return waitForPdfImageExtraction(existing, signal);
|
|
||||||
}
|
|
||||||
|
|
||||||
const extraction = createPdfImageExtraction(snapshot, imageDir);
|
|
||||||
pdfImageExtractions.set(imageDir, extraction);
|
|
||||||
return waitForPdfImageExtraction(extraction, signal);
|
|
||||||
}
|
|
||||||
|
|
||||||
export async function readPdfImageMember(
|
|
||||||
session: ToolSession,
|
|
||||||
autoResizeImages: boolean,
|
|
||||||
absolutePdfPath: string,
|
|
||||||
pdfDisplayPath: string,
|
|
||||||
member: string,
|
|
||||||
suffixResolution: { from: string; to: string } | undefined,
|
|
||||||
signal?: AbortSignal,
|
|
||||||
): Promise<AgentToolResult<ReadToolDetails>> {
|
|
||||||
const imageDir = await ensurePdfImageCache(session, absolutePdfPath, signal);
|
|
||||||
const members = await listPdfImageMembers(imageDir);
|
|
||||||
if (member.length === 0) {
|
|
||||||
const text =
|
|
||||||
members.length === 0
|
|
||||||
? "No extractable PDF image members found."
|
|
||||||
: `Extractable PDF image members:\n${members
|
|
||||||
.map(imageMember => `- read \`${pdfDisplayPath}:${imageMember}\``)
|
|
||||||
.join("\n")}`;
|
|
||||||
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
|
|
||||||
.text(prependSuffixResolutionNotice(text, suffixResolution))
|
|
||||||
.sourcePath(absolutePdfPath)
|
|
||||||
.done();
|
|
||||||
}
|
|
||||||
|
|
||||||
if (!members.includes(member)) {
|
|
||||||
const available = members.length === 0 ? "(none)" : members.join(", ");
|
|
||||||
throw new ToolError(`PDF image member '${member}' not found. Available members: ${available}`);
|
|
||||||
}
|
|
||||||
|
|
||||||
const imagePath = path.join(imageDir, member);
|
|
||||||
const imageStat = await Bun.file(imagePath).stat();
|
|
||||||
if (imageStat.size > MAX_IMAGE_SIZE) {
|
|
||||||
const sizeStr = formatBytes(imageStat.size);
|
|
||||||
const maxStr = formatBytes(MAX_IMAGE_SIZE);
|
|
||||||
throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`);
|
|
||||||
}
|
|
||||||
const metadata = await readImageMetadata(imagePath);
|
|
||||||
const mimeType = metadata?.mimeType;
|
|
||||||
if (!mimeType) throw new ToolError(`PDF image member '${member}' is not a supported image.`);
|
|
||||||
const imageInput = await loadImageInput({
|
|
||||||
path: `${pdfDisplayPath}:${member}`,
|
|
||||||
cwd: session.cwd,
|
|
||||||
autoResize: autoResizeImages,
|
|
||||||
maxBytes: MAX_IMAGE_SIZE,
|
|
||||||
resolvedPath: imagePath,
|
|
||||||
detectedMimeType: mimeType,
|
|
||||||
excludeWebP: webpExclusionForModel(session.getActiveModel?.()),
|
|
||||||
});
|
|
||||||
if (!imageInput) {
|
|
||||||
throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`);
|
|
||||||
}
|
|
||||||
const textNote = prependSuffixResolutionNotice(imageInput.textNote, suffixResolution);
|
|
||||||
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
|
|
||||||
.content([
|
|
||||||
{ type: "text", text: textNote },
|
|
||||||
{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType },
|
|
||||||
])
|
|
||||||
.sourcePath(imageInput.resolvedPath)
|
|
||||||
.done();
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,13 @@
|
|||||||
|
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
|
||||||
|
|
||||||
|
/** Parse a former PDF image-member read without claiming normal selectors. */
|
||||||
|
export function splitUnsupportedPdfImageReadPath(readPath: string): { pdfPath: string } | null {
|
||||||
|
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
|
||||||
|
const pdfPath = match?.[1];
|
||||||
|
return pdfPath ? { pdfPath } : null;
|
||||||
|
}
|
||||||
|
|
||||||
|
/** Explain how to render a PDF now that the text backend has no rasterizer. */
|
||||||
|
export function pdfImageRenderingUnsupportedMessage(pdfPath: string): string {
|
||||||
|
return `pdf-inspector cannot render PDF images. Use the Puppeteer browser tool to render '${pdfPath}', or read '${pdfPath}' for extracted text.`;
|
||||||
|
}
|
||||||
@@ -100,7 +100,7 @@ import {
|
|||||||
isRemoteMountPath,
|
isRemoteMountPath,
|
||||||
type SuffixMatchCache,
|
type SuffixMatchCache,
|
||||||
} from "./read-path-resolution";
|
} from "./read-path-resolution";
|
||||||
import { readPdfImageMember, rewritePdfImagePlaceholders, splitPdfImageMemberReadPath } from "./read-pdf-images";
|
import { pdfImageRenderingUnsupportedMessage, splitUnsupportedPdfImageReadPath } from "./read-pdf";
|
||||||
import { isMultiRange, isRawSelector, type ParsedSelector, parseSel, selToOffsetLimit } from "./read-selector";
|
import { isMultiRange, isRawSelector, type ParsedSelector, parseSel, selToOffsetLimit } from "./read-selector";
|
||||||
import { readSqlite, resolveSqliteReadPath } from "./read-sqlite";
|
import { readSqlite, resolveSqliteReadPath } from "./read-sqlite";
|
||||||
import { isProseSummaryPath, renderSummary, routeReadThroughBridge, trySummarize } from "./read-summary";
|
import { isProseSummaryPath, renderSummary, routeReadThroughBridge, trySummarize } from "./read-summary";
|
||||||
@@ -892,7 +892,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
|||||||
|
|
||||||
// Prefer a literal filesystem match over selector interpretation so real
|
// Prefer a literal filesystem match over selector interpretation so real
|
||||||
// POSIX filenames containing selector-looking suffixes win over structured
|
// POSIX filenames containing selector-looking suffixes win over structured
|
||||||
// archive / sqlite / pdf-image dispatch. A selector promoted from local://
|
// archive / sqlite / unsupported PDF-image dispatch. A selector promoted from local://
|
||||||
// remains separate so it cannot be mistaken for part of the resolved path.
|
// remains separate so it cannot be mistaken for part of the resolved path.
|
||||||
const literalSplit =
|
const literalSplit =
|
||||||
promotedSelector === undefined
|
promotedSelector === undefined
|
||||||
@@ -925,35 +925,10 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
|||||||
return readSqlite(sqlitePath, signal);
|
return readSqlite(sqlitePath, signal);
|
||||||
}
|
}
|
||||||
|
|
||||||
const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath);
|
const unsupportedPdfImageRead =
|
||||||
if (pdfImageMemberPath) {
|
literalSplit.sel === undefined ? splitUnsupportedPdfImageReadPath(readPath) : null;
|
||||||
let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd);
|
if (unsupportedPdfImageRead && (await probeLiteralPathExists(readPath, this.session.cwd)) === "missing") {
|
||||||
let suffixResolution: { from: string; to: string } | undefined;
|
throw new ToolError(pdfImageRenderingUnsupportedMessage(unsupportedPdfImageRead.pdfPath));
|
||||||
try {
|
|
||||||
const stat = await Bun.file(absolutePdfPath).stat();
|
|
||||||
if (stat.isDirectory())
|
|
||||||
throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`);
|
|
||||||
} catch (error) {
|
|
||||||
if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error;
|
|
||||||
const suffixMatch = await findSuffixMatchCached(
|
|
||||||
this.session,
|
|
||||||
suffixCache,
|
|
||||||
pdfImageMemberPath.pdfPath,
|
|
||||||
signal,
|
|
||||||
);
|
|
||||||
if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`);
|
|
||||||
absolutePdfPath = suffixMatch.absolutePath;
|
|
||||||
suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath };
|
|
||||||
}
|
|
||||||
return readPdfImageMember(
|
|
||||||
this.session,
|
|
||||||
this.#autoResizeImages,
|
|
||||||
absolutePdfPath,
|
|
||||||
pdfImageMemberPath.pdfPath,
|
|
||||||
pdfImageMemberPath.member,
|
|
||||||
suffixResolution,
|
|
||||||
signal,
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -1102,8 +1077,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
|||||||
// Convert document via markit.
|
// Convert document via markit.
|
||||||
const result = await convertFileWithMarkit(absolutePath, signal);
|
const result = await convertFileWithMarkit(absolutePath, signal);
|
||||||
if (result.ok) {
|
if (result.ok) {
|
||||||
const renderedContent =
|
const renderedContent = result.content;
|
||||||
ext === ".pdf" ? rewritePdfImagePlaceholders(result.content, resolvedDisplayPath) : result.content;
|
|
||||||
// Route the converted markdown through the in-memory text builder
|
// Route the converted markdown through the in-memory text builder
|
||||||
// so line-range selectors (`file.pdf:50-100`, `:5-16,40-80`) and
|
// so line-range selectors (`file.pdf:50-100`, `:5-16,40-80`) and
|
||||||
// raw mode apply against the converted output. Without this,
|
// raw mode apply against the converted output. Without this,
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
import * as path from "node:path";
|
import * as path from "node:path";
|
||||||
import { logger, untilAborted } from "@oh-my-pi/pi-utils";
|
import { untilAborted } from "@oh-my-pi/pi-utils";
|
||||||
import type { ConversionResult, Markit, StreamInfo } from "../markit";
|
import type { ConversionResult, Markit, StreamInfo } from "../markit";
|
||||||
import { ToolAbortError } from "../tools/tool-errors";
|
import { ToolAbortError } from "../tools/tool-errors";
|
||||||
import {
|
import {
|
||||||
@@ -8,7 +8,6 @@ import {
|
|||||||
readMarkitConversionCache,
|
readMarkitConversionCache,
|
||||||
writeMarkitConversionCache,
|
writeMarkitConversionCache,
|
||||||
} from "./markit-cache";
|
} from "./markit-cache";
|
||||||
import { loadEmbeddedMupdfWasm } from "./mupdf-wasm-embed";
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* File extensions markit can actually convert to markdown — one per registered
|
* File extensions markit can actually convert to markdown — one per registered
|
||||||
@@ -31,53 +30,16 @@ export interface MarkitConversionResult {
|
|||||||
|
|
||||||
export interface MarkitFileConversionOptions {
|
export interface MarkitFileConversionOptions {
|
||||||
/**
|
/**
|
||||||
* Directory the PDF converter writes extracted images/diagrams into. When
|
* Directory converters may use for extracted image or diagram files. Since
|
||||||
* set, each embedded image is rendered to `<id>.png` and referenced by path
|
* those files are conversion side effects, conversions using this option
|
||||||
* in the markdown; when unset, markit emits an `<!-- image: <id> ... -->`
|
* bypass the markdown cache.
|
||||||
* placeholder comment instead.
|
|
||||||
*/
|
*/
|
||||||
imageDir?: string;
|
imageDir?: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
interface MuPdfWasmModuleConfig {
|
|
||||||
print?: (...values: unknown[]) => void;
|
|
||||||
printErr?: (...values: unknown[]) => void;
|
|
||||||
wasmBinary?: Uint8Array;
|
|
||||||
}
|
|
||||||
|
|
||||||
function logMuPdfWasmOutput(stream: "stdout" | "stderr", values: unknown[]): void {
|
|
||||||
const message = values.length === 1 && typeof values[0] === "string" ? values[0] : values.map(String).join(" ");
|
|
||||||
logger.debug("mupdf wasm output", { stream, message });
|
|
||||||
}
|
|
||||||
|
|
||||||
// `$libmupdf_wasm_Module` is declared globally (as `any`) by the mupdf package.
|
|
||||||
// Install print hooks before the WASM module initializes so its stdout/stderr
|
|
||||||
// route to the file logger instead of corrupting the TUI.
|
|
||||||
function installMuPdfWasmLogger(): void {
|
|
||||||
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
|
|
||||||
moduleConfig.print = (...values: unknown[]) => logMuPdfWasmOutput("stdout", values);
|
|
||||||
moduleConfig.printErr = (...values: unknown[]) => logMuPdfWasmOutput("stderr", values);
|
|
||||||
globalThis.$libmupdf_wasm_Module = moduleConfig;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Hand the WASM module its bytes directly when the compiled binary embedded them
|
|
||||||
// (scripts/embed-mupdf-wasm.ts); a single-file binary has no node_modules for
|
|
||||||
// mupdf to read `mupdf-wasm.wasm` from. Source/npm builds get undefined here and
|
|
||||||
// mupdf loads its own wasm. Must run before the mupdf module evaluates.
|
|
||||||
function installEmbeddedMupdfWasm(): void {
|
|
||||||
const wasmBinary = loadEmbeddedMupdfWasm();
|
|
||||||
if (!wasmBinary) return;
|
|
||||||
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
|
|
||||||
moduleConfig.wasmBinary = wasmBinary;
|
|
||||||
globalThis.$libmupdf_wasm_Module = moduleConfig;
|
|
||||||
}
|
|
||||||
|
|
||||||
installMuPdfWasmLogger();
|
|
||||||
|
|
||||||
let markit: () => Markit | Promise<Markit> = async () => {
|
let markit: () => Markit | Promise<Markit> = async () => {
|
||||||
// Lazy: keep the document engine (mammoth/mupdf) off the startup
|
// Lazy: keep the document engine off the startup import graph — it loads
|
||||||
// import graph — it loads only when a document is first converted.
|
// only when a document is first converted.
|
||||||
installEmbeddedMupdfWasm();
|
|
||||||
const promise = import("../markit").then(({ Markit }) => {
|
const promise = import("../markit").then(({ Markit }) => {
|
||||||
const instance = new Markit();
|
const instance = new Markit();
|
||||||
markit = () => instance;
|
markit = () => instance;
|
||||||
|
|||||||
@@ -1,12 +0,0 @@
|
|||||||
// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
|
|
||||||
//
|
|
||||||
// Compiled single-file binaries cannot let mupdf resolve its `mupdf-wasm.wasm`
|
|
||||||
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
|
|
||||||
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
|
|
||||||
// wasm bytes via `with { type: "file" }` and copies the wasm next to it. Source
|
|
||||||
// checkouts, `bun test`, and the npm `dist/cli.js` bundle keep mupdf external and
|
|
||||||
// load the wasm from node_modules, so this placeholder returns undefined and the
|
|
||||||
// build resets back to it afterward.
|
|
||||||
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
|
|
||||||
return undefined;
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||||
|
import * as piNatives from "@oh-my-pi/pi-natives";
|
||||||
|
import { PdfConverter } from "../src/markit/converters/pdf";
|
||||||
|
|
||||||
|
describe("PdfConverter", () => {
|
||||||
|
afterEach(() => {
|
||||||
|
vi.restoreAllMocks();
|
||||||
|
});
|
||||||
|
|
||||||
|
it("keeps accepting PDF extensions and MIME types", () => {
|
||||||
|
const converter = new PdfConverter();
|
||||||
|
|
||||||
|
expect(converter.accepts({ extension: ".pdf" })).toBe(true);
|
||||||
|
expect(converter.accepts({ mimetype: "application/pdf" })).toBe(true);
|
||||||
|
expect(converter.accepts({ mimetype: "application/pdf; charset=binary" })).toBe(true);
|
||||||
|
expect(converter.accepts({ mimetype: "application/x-pdf" })).toBe(true);
|
||||||
|
expect(converter.accepts({ extension: ".txt", mimetype: "text/plain" })).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("returns a browser and OCR notice for an image-only PDF", async () => {
|
||||||
|
vi.spyOn(piNatives, "pdfToMarkdown").mockResolvedValue({
|
||||||
|
markdown: "",
|
||||||
|
pageCount: 3,
|
||||||
|
pagesNeedingOcr: [1, 3],
|
||||||
|
hasEncodingIssues: false,
|
||||||
|
});
|
||||||
|
|
||||||
|
const result = await new PdfConverter().convert(Buffer.from("image-only pdf"), { extension: ".pdf" });
|
||||||
|
|
||||||
|
expect(result.markdown).toBe(
|
||||||
|
"Text extraction is incomplete for PDF pages 1, 3. Use the browser tool to render those pages or OCR them.",
|
||||||
|
);
|
||||||
|
expect(result.markdown.length).toBeGreaterThan(0);
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -1,447 +0,0 @@
|
|||||||
/**
|
|
||||||
* PDF image extraction: markit emits inert `<!-- image: <id> ... -->`
|
|
||||||
* placeholders for embedded PDF images. The read tool rewrites those into
|
|
||||||
* browsable `read <pdf>:<id>.png` handles, and serves the actual PNG when that
|
|
||||||
* handle is read — extracting via markit's `imageDir` into a session-artifact
|
|
||||||
* cache. These lock the rewrite, the member extraction, member validation, and
|
|
||||||
* the caching contract.
|
|
||||||
*/
|
|
||||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
|
||||||
import * as fs from "node:fs";
|
|
||||||
import * as os from "node:os";
|
|
||||||
import * as path from "node:path";
|
|
||||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
|
||||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
|
||||||
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
|
||||||
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
|
|
||||||
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
|
|
||||||
import * as piUtils from "@oh-my-pi/pi-utils";
|
|
||||||
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
|
|
||||||
|
|
||||||
// 1x1 transparent PNG — small enough to pass through image loading untouched.
|
|
||||||
const TINY_PNG = Buffer.from(
|
|
||||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg==",
|
|
||||||
"base64",
|
|
||||||
);
|
|
||||||
|
|
||||||
function makeSession(testDir: string): ToolSession {
|
|
||||||
const sessionFile = path.join(testDir, "session.jsonl");
|
|
||||||
const artifactsDir = sessionFile.slice(0, -6);
|
|
||||||
return {
|
|
||||||
cwd: testDir,
|
|
||||||
hasUI: false,
|
|
||||||
getSessionFile: () => sessionFile,
|
|
||||||
getArtifactsDir: () => artifactsDir,
|
|
||||||
getSessionSpawns: () => null,
|
|
||||||
settings: Settings.isolated({ "images.autoResize": false }),
|
|
||||||
} as unknown as ToolSession;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Spy on markit so PDF "extraction" writes the given members into imageDir. */
|
|
||||||
function mockExtraction(members: Record<string, Buffer> = { "p11-img0.png": TINY_PNG }) {
|
|
||||||
return vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_filePath: string, _signal, options) => {
|
|
||||||
if (options?.imageDir) {
|
|
||||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
|
||||||
for (const name in members) {
|
|
||||||
fs.writeFileSync(path.join(options.imageDir, name), members[name]!);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
function imageBytes(result: AgentToolResult<ReadToolDetails>): Buffer {
|
|
||||||
const image = result.content.find(content => content.type === "image");
|
|
||||||
if (image?.type !== "image") throw new Error("Expected an image result");
|
|
||||||
return Buffer.from(image.data, "base64");
|
|
||||||
}
|
|
||||||
|
|
||||||
function mockBlockedExtraction() {
|
|
||||||
const entered = Promise.withResolvers<void>();
|
|
||||||
const release = Promise.withResolvers<void>();
|
|
||||||
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_sourcePath, signal, options) => {
|
|
||||||
entered.resolve();
|
|
||||||
await release.promise;
|
|
||||||
signal?.throwIfAborted();
|
|
||||||
if (options?.imageDir) {
|
|
||||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
|
||||||
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
|
|
||||||
}
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
return { entered, release, spy };
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Resolves once `count` callers are attached as waiters on the shared PDF
|
|
||||||
* extraction. Waiter attachment is the only `untilAborted` call that receives
|
|
||||||
* a promise (source snapshots pass thunks), so counting promise arguments
|
|
||||||
* observes it. The abort tests need this barrier: aborting a caller while it
|
|
||||||
* is the sole waiter tears the extraction down and deadlocks against the
|
|
||||||
* blocked conversion mock.
|
|
||||||
*/
|
|
||||||
function extractionWaitersAttached(count: number): Promise<void> {
|
|
||||||
const attached = Promise.withResolvers<void>();
|
|
||||||
const original = piUtils.untilAborted;
|
|
||||||
let seen = 0;
|
|
||||||
vi.spyOn(piUtils, "untilAborted").mockImplementation((signal, pr) => {
|
|
||||||
if (typeof pr !== "function" && ++seen === count) attached.resolve();
|
|
||||||
return original(signal, pr);
|
|
||||||
});
|
|
||||||
return attached.promise;
|
|
||||||
}
|
|
||||||
|
|
||||||
describe("read PDF image extraction", () => {
|
|
||||||
let testDir: string;
|
|
||||||
let pdfPath: string;
|
|
||||||
beforeEach(() => {
|
|
||||||
testDir = path.join(os.tmpdir(), `read-pdf-img-${Snowflake.next()}`);
|
|
||||||
fs.mkdirSync(testDir, { recursive: true });
|
|
||||||
pdfPath = path.join(testDir, "doc.pdf");
|
|
||||||
fs.writeFileSync(pdfPath, "%PDF-stub");
|
|
||||||
});
|
|
||||||
afterEach(() => {
|
|
||||||
vi.restoreAllMocks();
|
|
||||||
removeSyncWithRetries(testDir);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rewrites image placeholders into browse handles on a full read", async () => {
|
|
||||||
const converted = [
|
|
||||||
"Heading",
|
|
||||||
"",
|
|
||||||
"<!-- image: p11-img0 (page 11, 199x124pt) -->",
|
|
||||||
"",
|
|
||||||
"<!-- image: p11-img1 (page 11, 199x54pt) -->",
|
|
||||||
"",
|
|
||||||
"Footer",
|
|
||||||
].join("\n");
|
|
||||||
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: converted });
|
|
||||||
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const result = await tool.execute("call", { path: pdfPath });
|
|
||||||
const text = result.content
|
|
||||||
.filter(c => c.type === "text")
|
|
||||||
.map(c => c.text)
|
|
||||||
.join("\n");
|
|
||||||
|
|
||||||
expect(text).not.toContain("<!-- image:");
|
|
||||||
expect(text).toContain("read `doc.pdf:p11-img0.png`");
|
|
||||||
expect(text).toContain("read `doc.pdf:p11-img1.png`");
|
|
||||||
// Page/size metadata is preserved in the handle text.
|
|
||||||
expect(text).toContain("page 11, 199x124pt");
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rewrites placeholders inside a line-range view", async () => {
|
|
||||||
const lines = Array.from({ length: 20 }, (_, i) => `pdf line ${i + 1}`);
|
|
||||||
lines[9] = "<!-- image: p3-img0 (page 3, 100x50pt) -->"; // line 10
|
|
||||||
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: lines.join("\n") });
|
|
||||||
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const result = await tool.execute("call", { path: `${pdfPath}:8-12` });
|
|
||||||
const text = result.content
|
|
||||||
.filter(c => c.type === "text")
|
|
||||||
.map(c => c.text)
|
|
||||||
.join("\n");
|
|
||||||
|
|
||||||
expect(text).not.toContain("<!-- image:");
|
|
||||||
expect(text).toContain("read `doc.pdf:p3-img0.png`");
|
|
||||||
});
|
|
||||||
|
|
||||||
it("extracts a PDF image member as an inline image block", async () => {
|
|
||||||
const spy = mockExtraction();
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
|
|
||||||
const image = result.content.find(c => c.type === "image");
|
|
||||||
expect(image).toBeDefined();
|
|
||||||
expect(image && "mimeType" in image ? image.mimeType : undefined).toBe("image/png");
|
|
||||||
const text = result.content
|
|
||||||
.filter(c => c.type === "text")
|
|
||||||
.map(c => c.text)
|
|
||||||
.join("\n");
|
|
||||||
expect(text).toContain("Read image file");
|
|
||||||
// Extraction was driven through markit with an imageDir target.
|
|
||||||
expect(spy).toHaveBeenCalledTimes(1);
|
|
||||||
expect(spy.mock.calls[0]?.[2]?.imageDir).toBeTruthy();
|
|
||||||
});
|
|
||||||
|
|
||||||
it("reuses the extraction cache across member reads", async () => {
|
|
||||||
const spy = mockExtraction();
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
// Second read is served from the `.extracted` cache, not re-converted.
|
|
||||||
expect(spy).toHaveBeenCalledTimes(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("re-extracts image members after same-path PDF replacement", async () => {
|
|
||||||
const sourceA = Buffer.from("%PDF-source-a");
|
|
||||||
const sourceB = Buffer.from("%PDF-source-b");
|
|
||||||
fs.writeFileSync(pdfPath, sourceA);
|
|
||||||
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
|
|
||||||
if (options?.imageDir) {
|
|
||||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
|
||||||
fs.writeFileSync(
|
|
||||||
path.join(options.imageDir, "p11-img0.png"),
|
|
||||||
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
|
|
||||||
const originalStat = fs.statSync(pdfPath);
|
|
||||||
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
fs.writeFileSync(pdfPath, sourceB);
|
|
||||||
fs.utimesSync(pdfPath, originalStat.atime, originalStat.mtime);
|
|
||||||
const second = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
|
|
||||||
expect(imageBytes(first).subarray(TINY_PNG.length)).toEqual(sourceA);
|
|
||||||
expect(imageBytes(second).subarray(TINY_PNG.length)).toEqual(sourceB);
|
|
||||||
expect(spy).toHaveBeenCalledTimes(2);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("converts an immutable snapshot when the source changes during extraction", async () => {
|
|
||||||
const sourceA = Buffer.from("%PDF-source-a");
|
|
||||||
const sourceB = Buffer.from("%PDF-source-b");
|
|
||||||
fs.writeFileSync(pdfPath, sourceA);
|
|
||||||
const entered = Promise.withResolvers<void>();
|
|
||||||
const release = Promise.withResolvers<void>();
|
|
||||||
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
|
|
||||||
entered.resolve();
|
|
||||||
await release.promise;
|
|
||||||
if (options?.imageDir) {
|
|
||||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
|
||||||
fs.writeFileSync(
|
|
||||||
path.join(options.imageDir, "p11-img0.png"),
|
|
||||||
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
const pending = new ReadTool(makeSession(testDir)).execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
|
|
||||||
await entered.promise;
|
|
||||||
fs.writeFileSync(pdfPath, sourceB);
|
|
||||||
release.resolve();
|
|
||||||
const result = await pending;
|
|
||||||
|
|
||||||
expect(imageBytes(result).subarray(TINY_PNG.length)).toEqual(sourceA);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("coalesces concurrent cold image extraction", async () => {
|
|
||||||
const { entered, release, spy } = mockBlockedExtraction();
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
const second = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
|
|
||||||
await entered.promise;
|
|
||||||
const conversionCount = spy.mock.calls.length;
|
|
||||||
release.resolve();
|
|
||||||
const [firstResult, secondResult] = await Promise.all([first, second]);
|
|
||||||
|
|
||||||
expect(conversionCount).toBe(1);
|
|
||||||
expect(imageBytes(firstResult)).toEqual(imageBytes(secondResult));
|
|
||||||
});
|
|
||||||
|
|
||||||
it("keeps shared extraction running when its owner aborts", async () => {
|
|
||||||
const { entered, release, spy } = mockBlockedExtraction();
|
|
||||||
const bothAttached = extractionWaitersAttached(2);
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const ownerController = new AbortController();
|
|
||||||
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, ownerController.signal);
|
|
||||||
await entered.promise;
|
|
||||||
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
await bothAttached;
|
|
||||||
|
|
||||||
ownerController.abort();
|
|
||||||
await expect(owner).rejects.toThrow(/Aborted|Cancelled/);
|
|
||||||
release.resolve();
|
|
||||||
const result = await joiner;
|
|
||||||
|
|
||||||
expect(result.content.some(content => content.type === "image")).toBe(true);
|
|
||||||
expect(spy).toHaveBeenCalledTimes(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("keeps shared extraction running when a joiner aborts", async () => {
|
|
||||||
const { entered, release, spy } = mockBlockedExtraction();
|
|
||||||
const bothAttached = extractionWaitersAttached(2);
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const joinerController = new AbortController();
|
|
||||||
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
await entered.promise;
|
|
||||||
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, joinerController.signal);
|
|
||||||
await bothAttached;
|
|
||||||
|
|
||||||
joinerController.abort();
|
|
||||||
await expect(joiner).rejects.toThrow(/Aborted|Cancelled/);
|
|
||||||
release.resolve();
|
|
||||||
const result = await owner;
|
|
||||||
|
|
||||||
expect(result.content.some(content => content.type === "image")).toBe(true);
|
|
||||||
expect(spy).toHaveBeenCalledTimes(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("cleans temporary extraction state when the only caller aborts", async () => {
|
|
||||||
const entered = Promise.withResolvers<void>();
|
|
||||||
let snapshotPath: string | undefined;
|
|
||||||
let stagingDir: string | undefined;
|
|
||||||
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, signal, options) => {
|
|
||||||
snapshotPath = sourcePath;
|
|
||||||
stagingDir = options?.imageDir;
|
|
||||||
entered.resolve();
|
|
||||||
const aborted = Promise.withResolvers<void>();
|
|
||||||
const onAbort = () => aborted.resolve();
|
|
||||||
if (signal?.aborted) onAbort();
|
|
||||||
else signal?.addEventListener("abort", onAbort, { once: true });
|
|
||||||
await aborted.promise;
|
|
||||||
signal?.removeEventListener("abort", onAbort);
|
|
||||||
signal?.throwIfAborted();
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
const controller = new AbortController();
|
|
||||||
const pending = new ReadTool(makeSession(testDir)).execute(
|
|
||||||
"call",
|
|
||||||
{ path: `${pdfPath}:p11-img0.png` },
|
|
||||||
controller.signal,
|
|
||||||
);
|
|
||||||
|
|
||||||
await entered.promise;
|
|
||||||
controller.abort();
|
|
||||||
await expect(pending).rejects.toThrow(/Aborted|Cancelled/);
|
|
||||||
if (!snapshotPath || !stagingDir) throw new Error("Expected extraction paths");
|
|
||||||
|
|
||||||
expect(fs.existsSync(path.dirname(snapshotPath))).toBe(false);
|
|
||||||
expect(fs.existsSync(stagingDir)).toBe(false);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("does not let a failed generation delete a replacement generation", async () => {
|
|
||||||
const sourceA = Buffer.from("%PDF-source-a");
|
|
||||||
const sourceB = Buffer.from("%PDF-source-b");
|
|
||||||
fs.writeFileSync(pdfPath, sourceA);
|
|
||||||
const firstEntered = Promise.withResolvers<void>();
|
|
||||||
const failFirst = Promise.withResolvers<void>();
|
|
||||||
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
|
|
||||||
const source = fs.readFileSync(sourcePath);
|
|
||||||
if (source.equals(sourceA)) {
|
|
||||||
firstEntered.resolve();
|
|
||||||
await failFirst.promise;
|
|
||||||
return { ok: false, content: "", error: "generation A failed" };
|
|
||||||
}
|
|
||||||
if (options?.imageDir) {
|
|
||||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
|
||||||
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), Buffer.concat([TINY_PNG, source]));
|
|
||||||
}
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
await firstEntered.promise;
|
|
||||||
fs.writeFileSync(pdfPath, sourceB);
|
|
||||||
|
|
||||||
const replacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
failFirst.resolve();
|
|
||||||
await expect(first).rejects.toThrow(/Cannot extract images/);
|
|
||||||
const cachedReplacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
|
|
||||||
expect(imageBytes(replacement).subarray(TINY_PNG.length)).toEqual(sourceB);
|
|
||||||
expect(imageBytes(cachedReplacement)).toEqual(imageBytes(replacement));
|
|
||||||
expect(spy).toHaveBeenCalledTimes(2);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("isolates equal-content PDFs with the same basename in different directories", async () => {
|
|
||||||
const otherDir = path.join(testDir, "other");
|
|
||||||
const otherPdfPath = path.join(otherDir, path.basename(pdfPath));
|
|
||||||
fs.mkdirSync(otherDir, { recursive: true });
|
|
||||||
fs.writeFileSync(otherPdfPath, fs.readFileSync(pdfPath));
|
|
||||||
let conversion = 0;
|
|
||||||
const spy = vi
|
|
||||||
.spyOn(markit, "convertFileWithMarkit")
|
|
||||||
.mockImplementation(async (_sourcePath, _signal, options) => {
|
|
||||||
conversion++;
|
|
||||||
if (options?.imageDir) {
|
|
||||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
|
||||||
fs.writeFileSync(
|
|
||||||
path.join(options.imageDir, "p11-img0.png"),
|
|
||||||
Buffer.concat([TINY_PNG, Buffer.from(String(conversion))]),
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
|
|
||||||
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
const second = await tool.execute("call", { path: `${otherPdfPath}:p11-img0.png` });
|
|
||||||
|
|
||||||
expect(imageBytes(first).subarray(TINY_PNG.length).toString()).toBe("1");
|
|
||||||
expect(imageBytes(second).subarray(TINY_PNG.length).toString()).toBe("2");
|
|
||||||
expect(spy).toHaveBeenCalledTimes(2);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("supports PDF basenames at the filesystem component limit", async () => {
|
|
||||||
const longPdfPath = path.join(testDir, `${"a".repeat(250)}.pdf`);
|
|
||||||
fs.writeFileSync(longPdfPath, "%PDF-stub");
|
|
||||||
mockExtraction();
|
|
||||||
|
|
||||||
const result = await new ReadTool(makeSession(testDir)).execute("call", {
|
|
||||||
path: `${longPdfPath}:p11-img0.png`,
|
|
||||||
});
|
|
||||||
|
|
||||||
expect(result.content.some(content => content.type === "image")).toBe(true);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("errors with the available members for an unknown member", async () => {
|
|
||||||
mockExtraction();
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
await expect(tool.execute("call", { path: `${pdfPath}:does-not-exist.png` })).rejects.toThrow(
|
|
||||||
/not found.*p11-img0\.png/s,
|
|
||||||
);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("rejects member traversal attempts", async () => {
|
|
||||||
mockExtraction();
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
// `../../escape.png` matches the image-member shape but is not a known
|
|
||||||
// basename, so it must be refused rather than joined into the cache path.
|
|
||||||
await expect(tool.execute("call", { path: `${pdfPath}:../../escape.png` })).rejects.toThrow(/not found/);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("lists extractable members for a trailing-colon read", async () => {
|
|
||||||
mockExtraction({ "p1-img0.png": TINY_PNG, "p2-img0.png": TINY_PNG });
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
const result = await tool.execute("call", { path: `${pdfPath}:` });
|
|
||||||
const text = result.content
|
|
||||||
.filter(c => c.type === "text")
|
|
||||||
.map(c => c.text)
|
|
||||||
.join("\n");
|
|
||||||
expect(text).toContain(`read \`${pdfPath}:p1-img0.png\``);
|
|
||||||
expect(text).toContain(`read \`${pdfPath}:p2-img0.png\``);
|
|
||||||
});
|
|
||||||
|
|
||||||
it("does not cache a failed conversion", async () => {
|
|
||||||
let failedSnapshotPath: string | undefined;
|
|
||||||
let failedImageDir: string | undefined;
|
|
||||||
const spy = vi.spyOn(markit, "convertFileWithMarkit");
|
|
||||||
spy.mockImplementationOnce(async (sourcePath, _signal, options) => {
|
|
||||||
failedSnapshotPath = sourcePath;
|
|
||||||
failedImageDir = options?.imageDir;
|
|
||||||
return { ok: false, content: "", error: "boom" };
|
|
||||||
});
|
|
||||||
const tool = new ReadTool(makeSession(testDir));
|
|
||||||
await expect(tool.execute("call", { path: `${pdfPath}:p11-img0.png` })).rejects.toThrow(/Cannot extract images/);
|
|
||||||
if (!failedSnapshotPath || !failedImageDir) throw new Error("Expected failed extraction paths");
|
|
||||||
expect(fs.existsSync(path.dirname(failedSnapshotPath))).toBe(false);
|
|
||||||
expect(fs.existsSync(path.join(failedImageDir, ".extracted"))).toBe(false);
|
|
||||||
|
|
||||||
spy.mockImplementationOnce(async (_filePath: string, _signal, options) => {
|
|
||||||
if (options?.imageDir) {
|
|
||||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
|
||||||
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
|
|
||||||
}
|
|
||||||
return { ok: true, content: "" };
|
|
||||||
});
|
|
||||||
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
|
||||||
expect(result.content.some(c => c.type === "image")).toBe(true);
|
|
||||||
expect(spy).toHaveBeenCalledTimes(2);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
@@ -0,0 +1,79 @@
|
|||||||
|
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||||
|
import * as fs from "node:fs/promises";
|
||||||
|
import * as os from "node:os";
|
||||||
|
import * as path from "node:path";
|
||||||
|
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||||
|
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||||
|
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||||
|
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
|
||||||
|
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
|
||||||
|
import { removeWithRetries } from "@oh-my-pi/pi-utils";
|
||||||
|
|
||||||
|
function makeSession(cwd: string): ToolSession {
|
||||||
|
return {
|
||||||
|
cwd,
|
||||||
|
hasUI: false,
|
||||||
|
getSessionFile: () => null,
|
||||||
|
getSessionSpawns: () => "*",
|
||||||
|
settings: Settings.isolated({ "images.autoResize": false }),
|
||||||
|
} as ToolSession;
|
||||||
|
}
|
||||||
|
|
||||||
|
function textOf(result: AgentToolResult<ReadToolDetails>): string {
|
||||||
|
return result.content
|
||||||
|
.filter(entry => entry.type === "text")
|
||||||
|
.map(entry => entry.text)
|
||||||
|
.join("\n");
|
||||||
|
}
|
||||||
|
|
||||||
|
describe("read unsupported PDF image members", () => {
|
||||||
|
let testDir: string;
|
||||||
|
let pdfPath: string;
|
||||||
|
|
||||||
|
beforeEach(async () => {
|
||||||
|
testDir = await fs.mkdtemp(path.join(os.tmpdir(), "read-pdf-image-unsupported-"));
|
||||||
|
pdfPath = path.join(testDir, "doc.pdf");
|
||||||
|
await fs.writeFile(pdfPath, `%PDF-stub-${testDir}`);
|
||||||
|
});
|
||||||
|
|
||||||
|
afterEach(async () => {
|
||||||
|
vi.restoreAllMocks();
|
||||||
|
await removeWithRetries(testDir);
|
||||||
|
});
|
||||||
|
|
||||||
|
it("directs former image listing and PNG member reads to browser rendering", async () => {
|
||||||
|
const tool = new ReadTool(makeSession(testDir));
|
||||||
|
|
||||||
|
for (const readPath of [`${pdfPath}:`, `${pdfPath}:p1-img0.png`]) {
|
||||||
|
try {
|
||||||
|
await tool.execute("read-pdf-image", { path: readPath });
|
||||||
|
throw new Error("Expected the PDF image read to fail");
|
||||||
|
} catch (error) {
|
||||||
|
expect(error).toBeInstanceOf(Error);
|
||||||
|
const message = (error as Error).message;
|
||||||
|
expect(message).toContain("pdf-inspector cannot render PDF images");
|
||||||
|
expect(message).toContain("Puppeteer browser tool");
|
||||||
|
expect(message).toContain(`read '${pdfPath}' for extracted text`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
});
|
||||||
|
|
||||||
|
it("preserves a literal filename that looks like a PDF image listing", async () => {
|
||||||
|
const literalPath = `${pdfPath}:`;
|
||||||
|
await fs.writeFile(literalPath, "literal colon path wins\n");
|
||||||
|
|
||||||
|
const result = await new ReadTool(makeSession(testDir)).execute("read-literal", { path: literalPath });
|
||||||
|
expect(textOf(result)).toContain("literal colon path wins");
|
||||||
|
});
|
||||||
|
|
||||||
|
it("routes PDF line selectors through normal document conversion", async () => {
|
||||||
|
const convert = vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({
|
||||||
|
ok: true,
|
||||||
|
content: "first line\nselected line\nthird line\n",
|
||||||
|
});
|
||||||
|
|
||||||
|
const result = await new ReadTool(makeSession(testDir)).execute("read-pdf-lines", { path: `${pdfPath}:2-2` });
|
||||||
|
expect(convert).toHaveBeenCalledTimes(1);
|
||||||
|
expect(textOf(result)).toContain("selected line");
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -105,7 +105,7 @@ describe("document conversion cache", () => {
|
|||||||
|
|
||||||
it("skips cache for imageDir conversions", async () => {
|
it("skips cache for imageDir conversions", async () => {
|
||||||
const convert = vi.spyOn(Markit.prototype, "convert").mockResolvedValue({ markdown: "image body" });
|
const convert = vi.spyOn(Markit.prototype, "convert").mockResolvedValue({ markdown: "image body" });
|
||||||
const docPath = path.join(testDir, "image-doc.pdf");
|
const docPath = path.join(testDir, "image-doc.docx");
|
||||||
await fs.writeFile(docPath, new TextEncoder().encode("image bytes"));
|
await fs.writeFile(docPath, new TextEncoder().encode("image bytes"));
|
||||||
const imageDir = path.join(testDir, "images");
|
const imageDir = path.join(testDir, "images");
|
||||||
|
|
||||||
|
|||||||
@@ -2,6 +2,10 @@
|
|||||||
|
|
||||||
## [Unreleased]
|
## [Unreleased]
|
||||||
|
|
||||||
|
### Added
|
||||||
|
|
||||||
|
- Added the async `pdfToMarkdown` native API backed by `pdf-inspector`, with page numbering, page-count, OCR-needed-page, and encoding-issue metadata.
|
||||||
|
|
||||||
### Changed
|
### Changed
|
||||||
|
|
||||||
- Docker images (`Dockerfile`, `scripts/install-tests/*.dockerfile`) build the native addon through the cargo/napi-rs backend (`OMP_NATIVE_BUILD_BACKEND=cargo`) instead of Bazel: a single fixed host target gains nothing from hermetic cross toolchains, and none of those images shipped bazelisk. `OMP_NATIVE_CARGO_PROFILE` picks the profile for that path (images use `ci`, local default stays `local`).
|
- Docker images (`Dockerfile`, `scripts/install-tests/*.dockerfile`) build the native addon through the cargo/napi-rs backend (`OMP_NATIVE_BUILD_BACKEND=cargo`) instead of Bazel: a single fixed host target gains nothing from hermetic cross toolchains, and none of those images shipped bazelisk. `OMP_NATIVE_CARGO_PROFILE` picks the profile for that path (images use `ci`, local default stays `local`).
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ Native Rust functionality via N-API.
|
|||||||
- **Audio**: Cross-platform low-latency microphone capture and gapless speaker playback
|
- **Audio**: Cross-platform low-latency microphone capture and gapless speaker playback
|
||||||
- **WebRTC**: Native Opus media, SDP offer/answer negotiation, and data-channel events for live sessions
|
- **WebRTC**: Native Opus media, SDP offer/answer negotiation, and data-channel events for live sessions
|
||||||
- **File locking**: Process-owned cross-process locks with in-memory kernel names on Linux/Windows and `flock(2)` sidecars on other Unix platforms
|
- **File locking**: Process-owned cross-process locks with in-memory kernel names on Linux/Windows and `flock(2)` sidecars on other Unix platforms
|
||||||
|
- **PDF**: In-memory PDF-to-Markdown extraction with OCR-page classification via `pdf-inspector`
|
||||||
|
|
||||||
General-purpose image processing (decode/resize/encode for files and buffers)
|
General-purpose image processing (decode/resize/encode for files and buffers)
|
||||||
lives in [`Bun.Image`](https://bun.com/docs/runtime/image) on the JS side; this
|
lives in [`Bun.Image`](https://bun.com/docs/runtime/image) on the JS side; this
|
||||||
@@ -19,7 +20,7 @@ that terminal protocol.
|
|||||||
## Usage
|
## Usage
|
||||||
|
|
||||||
```typescript
|
```typescript
|
||||||
import { grep, find, encodeSixel } from "@oh-my-pi/pi-natives";
|
import { encodeSixel, grep, pdfToMarkdown } from "@oh-my-pi/pi-natives";
|
||||||
|
|
||||||
// Grep for a pattern
|
// Grep for a pattern
|
||||||
const results = await grep({
|
const results = await grep({
|
||||||
@@ -38,6 +39,10 @@ const files = await find({
|
|||||||
|
|
||||||
// SIXEL encode for a terminal cell box (px)
|
// SIXEL encode for a terminal cell box (px)
|
||||||
const sequence = encodeSixel(pngBytes, widthPx, heightPx);
|
const sequence = encodeSixel(pngBytes, widthPx, heightPx);
|
||||||
|
|
||||||
|
// Extract PDF text and identify pages that still need OCR
|
||||||
|
const pdf = await pdfToMarkdown(pdfBytes);
|
||||||
|
console.log(pdf.markdown, pdf.pagesNeedingOcr);
|
||||||
```
|
```
|
||||||
|
|
||||||
## Building
|
## Building
|
||||||
|
|||||||
Vendored
+26
@@ -1578,6 +1578,32 @@ export interface PatchHunk {
|
|||||||
lines: Array<string>
|
lines: Array<string>
|
||||||
}
|
}
|
||||||
|
|
||||||
|
/** Markdown and inspection metadata produced from a PDF document. */
|
||||||
|
export interface PdfMarkdownResult {
|
||||||
|
/** Extracted document content in Markdown format. */
|
||||||
|
markdown: string
|
||||||
|
/** Document title from PDF metadata, when present. */
|
||||||
|
title?: string
|
||||||
|
/** Total number of pages in the document. */
|
||||||
|
pageCount: number
|
||||||
|
/** One-indexed page numbers whose content requires OCR. */
|
||||||
|
pagesNeedingOcr: Array<number>
|
||||||
|
/** Whether the document contains text encoding problems. */
|
||||||
|
hasEncodingIssues: boolean
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Convert an in-memory PDF to Markdown and return its inspection metadata.
|
||||||
|
*
|
||||||
|
* Conversion copies the typed array before dispatch so JavaScript mutation
|
||||||
|
* cannot race the native worker.
|
||||||
|
*
|
||||||
|
* # Errors
|
||||||
|
* Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
|
||||||
|
* be parsed or converted.
|
||||||
|
*/
|
||||||
|
export declare function pdfToMarkdown(input: Uint8Array): Promise<PdfMarkdownResult>
|
||||||
|
|
||||||
export interface PointerOptions {
|
export interface PointerOptions {
|
||||||
button?: string
|
button?: string
|
||||||
count?: number
|
count?: number
|
||||||
|
|||||||
@@ -69,6 +69,7 @@ export const matchesLegacySequence = nativeBindings.matchesLegacySequence;
|
|||||||
export const mmrRerankIndices = nativeBindings.mmrRerankIndices;
|
export const mmrRerankIndices = nativeBindings.mmrRerankIndices;
|
||||||
export const parseKey = nativeBindings.parseKey;
|
export const parseKey = nativeBindings.parseKey;
|
||||||
export const parseKittySequence = nativeBindings.parseKittySequence;
|
export const parseKittySequence = nativeBindings.parseKittySequence;
|
||||||
|
export const pdfToMarkdown = nativeBindings.pdfToMarkdown;
|
||||||
export const readImageFromClipboard = nativeBindings.readImageFromClipboard;
|
export const readImageFromClipboard = nativeBindings.readImageFromClipboard;
|
||||||
export const renderSnapcompactPng = nativeBindings.renderSnapcompactPng;
|
export const renderSnapcompactPng = nativeBindings.renderSnapcompactPng;
|
||||||
export const search = nativeBindings.search;
|
export const search = nativeBindings.search;
|
||||||
|
|||||||
+1
@@ -26,6 +26,7 @@ export interface DetectCompiledBinaryInput {
|
|||||||
|
|
||||||
export function detectCompiledBinary(input: DetectCompiledBinaryInput): boolean;
|
export function detectCompiledBinary(input: DetectCompiledBinaryInput): boolean;
|
||||||
|
|
||||||
|
|
||||||
export interface GetAddonFilenamesInput {
|
export interface GetAddonFilenamesInput {
|
||||||
tag: string;
|
tag: string;
|
||||||
arch: string;
|
arch: string;
|
||||||
|
|||||||
@@ -87,7 +87,6 @@ export function detectCompiledBinary({ embeddedAddon, env, importMetaUrl }) {
|
|||||||
}
|
}
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* @param {{ tag: string; arch: string; variant: "modern" | "baseline" | null | undefined }} input
|
* @param {{ tag: string; arch: string; variant: "modern" | "baseline" | null | undefined }} input
|
||||||
* @returns {string[]}
|
* @returns {string[]}
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
{
|
{
|
||||||
"name": "@oh-my-pi/pi-natives",
|
"name": "@oh-my-pi/pi-natives",
|
||||||
"version": "17.3.3",
|
"version": "17.3.3",
|
||||||
"description": "Native Rust bindings for audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API",
|
"description": "Native Rust bindings for PDF conversion, audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API",
|
||||||
"type": "module",
|
"type": "module",
|
||||||
"homepage": "https://omp.sh",
|
"homepage": "https://omp.sh",
|
||||||
"author": "Can Boluk",
|
"author": "Can Boluk",
|
||||||
|
|||||||
@@ -23,6 +23,7 @@ import {
|
|||||||
matchesKey,
|
matchesKey,
|
||||||
PtySession,
|
PtySession,
|
||||||
parseKey,
|
parseKey,
|
||||||
|
pdfToMarkdown,
|
||||||
summarizeCode,
|
summarizeCode,
|
||||||
supportsLanguage,
|
supportsLanguage,
|
||||||
truncateToWidth,
|
truncateToWidth,
|
||||||
@@ -87,6 +88,30 @@ async function createFifo(fifoPath: string) {
|
|||||||
throw new Error(await new Response(process.stderr).text());
|
throw new Error(await new Response(process.stderr).text());
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function textPdf(text: string): Uint8Array {
|
||||||
|
const stream = `BT /F1 12 Tf 72 720 Td (${text}) Tj ET`;
|
||||||
|
const objects = [
|
||||||
|
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||||
|
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||||
|
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||||
|
`<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`,
|
||||||
|
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>",
|
||||||
|
];
|
||||||
|
let document = "%PDF-1.4\n";
|
||||||
|
const offsets: number[] = [];
|
||||||
|
for (const [index, object] of objects.entries()) {
|
||||||
|
offsets.push(document.length);
|
||||||
|
document += `${index + 1} 0 obj\n${object}\nendobj\n`;
|
||||||
|
}
|
||||||
|
const xrefOffset = document.length;
|
||||||
|
document += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
|
||||||
|
for (const offset of offsets) {
|
||||||
|
document += `${offset.toString().padStart(10, "0")} 00000 n \n`;
|
||||||
|
}
|
||||||
|
document += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xrefOffset}\n%%EOF\n`;
|
||||||
|
return Buffer.from(document);
|
||||||
|
}
|
||||||
|
|
||||||
describe("pi-natives", () => {
|
describe("pi-natives", () => {
|
||||||
beforeAll(async () => {
|
beforeAll(async () => {
|
||||||
await setupFixtures();
|
await setupFixtures();
|
||||||
@@ -763,6 +788,19 @@ describe("pi-natives", () => {
|
|||||||
expect(await Bun.file(markerPath).exists()).toBe(false);
|
expect(await Bun.file(markerPath).exists()).toBe(false);
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
|
describe("pdfToMarkdown", () => {
|
||||||
|
it("isolates blocking conversion from later JavaScript buffer mutation", async () => {
|
||||||
|
const input = textPdf("Copied PDF bytes");
|
||||||
|
const conversion = pdfToMarkdown(input);
|
||||||
|
input.fill(0);
|
||||||
|
|
||||||
|
const result = await conversion;
|
||||||
|
|
||||||
|
expect(result.pageCount).toBe(1);
|
||||||
|
expect(result.markdown).toContain("Copied PDF bytes");
|
||||||
|
});
|
||||||
|
});
|
||||||
describe("htmlToMarkdown", () => {
|
describe("htmlToMarkdown", () => {
|
||||||
it("should convert basic HTML to markdown", async () => {
|
it("should convert basic HTML to markdown", async () => {
|
||||||
const html = "<h1>Hello World</h1><p>This is a paragraph.</p>";
|
const html = "<h1>Hello World</h1><p>This is a paragraph.</p>";
|
||||||
|
|||||||
@@ -158,24 +158,20 @@ async function generateBundle(): Promise<void> {
|
|||||||
if (isDryRun) {
|
if (isDryRun) {
|
||||||
console.log("DRY RUN bun run gen:stats");
|
console.log("DRY RUN bun run gen:stats");
|
||||||
console.log("DRY RUN bun --cwd=packages/collab-web run gen:tool-views");
|
console.log("DRY RUN bun --cwd=packages/collab-web run gen:tool-views");
|
||||||
console.log("DRY RUN bun run gen:mupdf");
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
await runCommand(["bun", "run", "gen:stats"], repoRoot);
|
await runCommand(["bun", "run", "gen:stats"], repoRoot);
|
||||||
await runCommand(["bun", "--cwd=packages/collab-web", "run", "gen:tool-views"], repoRoot);
|
await runCommand(["bun", "--cwd=packages/collab-web", "run", "gen:tool-views"], repoRoot);
|
||||||
await runCommand(["bun", "run", "gen:mupdf"], repoRoot);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async function resetArtifacts(): Promise<void> {
|
async function resetArtifacts(): Promise<void> {
|
||||||
if (isDryRun) {
|
if (isDryRun) {
|
||||||
console.log("DRY RUN bun run gen:native:reset");
|
console.log("DRY RUN bun run gen:native:reset");
|
||||||
console.log("DRY RUN bun run gen:stats:reset");
|
console.log("DRY RUN bun run gen:stats:reset");
|
||||||
console.log("DRY RUN bun run gen:mupdf:reset");
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
await runCommand(["bun", "run", "gen:native:reset"], repoRoot);
|
await runCommand(["bun", "run", "gen:native:reset"], repoRoot);
|
||||||
await runCommand(["bun", "run", "gen:stats:reset"], repoRoot);
|
await runCommand(["bun", "run", "gen:stats:reset"], repoRoot);
|
||||||
await runCommand(["bun", "run", "gen:mupdf:reset"], repoRoot);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
async function main(): Promise<void> {
|
async function main(): Promise<void> {
|
||||||
|
|||||||
Reference in New Issue
Block a user