feat: replaced custom mupdf wasm pipeline with native function
- Replaced the custom MuPDF-WASM PDF extraction and rendering pipeline with the new `pdfToMarkdown` native function from `@oh-my-pi/pi-natives`. - Removed legacy MuPDF extraction modules, WASM embedding scripts, and PDF image extraction tools. - Added OCR warnings and browser/text redirection for unsupported PDF image reads. - Updated native package definitions, documentation, and test suites for the new PDF inspection capability.
This commit is contained in:
@@ -162,11 +162,11 @@ jobs:
|
||||
bazelisk --bazelrc="${{ steps.cache.outputs.rc }}" test //crates/...
|
||||
# Clippy scope mirrors `cargo clippy --workspace` (libraries only, no
|
||||
# test targets) plus the strict/default split: crates with
|
||||
# `[lints] workspace = true` get the workspace policy, the vendored
|
||||
# brush-core fork is exempt (same as run-rs-task.ts's cargo excludes).
|
||||
# `[lints] workspace = true` get the workspace policy, except
|
||||
# brush-core (a vendored fork excluded from the Cargo task too).
|
||||
# pi-builtins allows every clippy group in its own manifest (ported
|
||||
# brush/uutils/jaq code) but is still held to zero rustc warnings;
|
||||
# cargo honors that via `[lints]`, bazel via the clippy-ported config.
|
||||
# Cargo honors that via `[lints]`, Bazel via the clippy-ported config.
|
||||
- name: Clippy (workspace lint policy on opted-in crates)
|
||||
run: |
|
||||
bazelisk query "kind('rust_library|rust_shared_library', //crates/pi-ast/... + //crates/pi-iso/... + //crates/pi-natives/... + //crates/pi-shell/... + //crates/pi-voice/... + //crates/pi-walker/...)" \
|
||||
@@ -411,12 +411,6 @@ jobs:
|
||||
- name: Test coding-agent native/unit bucket
|
||||
env:
|
||||
OMP_TEST_CONCURRENCY: "4"
|
||||
# The mupdf/PDF-extraction chunk measures ~7 min on burstable
|
||||
# runners under a full 8-wide fan-out; the default 600 s chunk
|
||||
# watchdog SIGKILLed it (release run 30519992654). The watchdog
|
||||
# exists to catch wedged children, not slow-but-progressing
|
||||
# chunks — give this bucket a wider budget.
|
||||
OMP_TEST_CHUNK_TIMEOUT: "1200"
|
||||
run: bun run ci:test:coding-agent:native
|
||||
|
||||
test_smoke:
|
||||
|
||||
@@ -64,7 +64,6 @@ pi-*.html
|
||||
# Generated files
|
||||
packages/coding-agent/src/export/html/tool-views.generated.js
|
||||
packages/natives/npm/
|
||||
packages/coding-agent/src/utils/mupdf-wasm.wasm
|
||||
/runs/
|
||||
python/omp-rpc/src/omp_rpc.egg-info/
|
||||
python/omp-rpc/build/
|
||||
|
||||
Generated
+139
@@ -1868,6 +1868,15 @@ version = "1.0.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d0881ea181b1df73ff77ffaaf9c7544ecc11e82fba9b5f27b262a3c73a332555"
|
||||
|
||||
[[package]]
|
||||
name = "ecb"
|
||||
version = "0.1.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1a8bfa975b1aec2145850fcaa1c6fe269a16578c44705a532ae3edc92b8881c7"
|
||||
dependencies = [
|
||||
"cipher",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ecdsa"
|
||||
version = "0.16.9"
|
||||
@@ -1979,6 +1988,29 @@ dependencies = [
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "env_filter"
|
||||
version = "2.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "900d271a03799a1ee8d1ca9b19893b48ca674a9284fefcfb85f05e74ed314217"
|
||||
dependencies = [
|
||||
"log",
|
||||
"regex",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "env_logger"
|
||||
version = "0.11.11"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "de671bd27a75a797dc9ae289ba1e77276e75e2026408aab65185384e2d5cd3f6"
|
||||
dependencies = [
|
||||
"anstream",
|
||||
"anstyle",
|
||||
"env_filter",
|
||||
"jiff",
|
||||
"log",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "equivalent"
|
||||
version = "1.0.2"
|
||||
@@ -3152,6 +3184,25 @@ dependencies = [
|
||||
"quick-error",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "include_dir"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "923d117408f1e49d914f1a379a309cffe4f18c05cf4e3d12e613a15fc81bd0dd"
|
||||
dependencies = [
|
||||
"include_dir_macros",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "include_dir_macros"
|
||||
version = "0.7.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7cab85a7ed0bd5f0e76d93846e0147172bed2e2d3f859bcc33a8d9699cad1a75"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indenter"
|
||||
version = "0.3.4"
|
||||
@@ -3640,6 +3691,37 @@ version = "0.4.33"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
|
||||
|
||||
[[package]]
|
||||
name = "lopdf"
|
||||
version = "0.42.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "25aab26d99567469098e64a02f42679f8965c6401263eefa31d8f2dcc37a221c"
|
||||
dependencies = [
|
||||
"aes",
|
||||
"bitflags 2.13.1",
|
||||
"cbc",
|
||||
"chrono",
|
||||
"ecb",
|
||||
"encoding_rs",
|
||||
"flate2",
|
||||
"getrandom 0.4.3",
|
||||
"indexmap",
|
||||
"itoa",
|
||||
"jiff",
|
||||
"log",
|
||||
"md-5",
|
||||
"nom 8.0.0",
|
||||
"rand 0.10.2",
|
||||
"rangemap",
|
||||
"rayon",
|
||||
"sha2",
|
||||
"stringprep",
|
||||
"thiserror 2.0.20",
|
||||
"time",
|
||||
"ttf-parser",
|
||||
"weezl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "lscolors"
|
||||
version = "0.21.0"
|
||||
@@ -4539,6 +4621,24 @@ dependencies = [
|
||||
"pkg-config",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pdf-inspector"
|
||||
version = "1.14.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e024ae242c514e2adf6aee186678e0eabdc2e5ecfbb2159186881b4498593cb"
|
||||
dependencies = [
|
||||
"env_logger",
|
||||
"include_dir",
|
||||
"log",
|
||||
"lopdf",
|
||||
"once_cell",
|
||||
"rayon",
|
||||
"regex",
|
||||
"thiserror 2.0.20",
|
||||
"ttf-parser",
|
||||
"unicode-normalization",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "peg"
|
||||
version = "0.8.6"
|
||||
@@ -4982,6 +5082,7 @@ dependencies = [
|
||||
"objc2-core-graphics",
|
||||
"objc2-foundation",
|
||||
"parking_lot",
|
||||
"pdf-inspector",
|
||||
"phf 0.13.1",
|
||||
"pi-ast",
|
||||
"pi-iso",
|
||||
@@ -5505,6 +5606,12 @@ dependencies = [
|
||||
"rand_core 0.10.1",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rangemap"
|
||||
version = "1.8.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a611d15b50743feb4c76b7d03edcb0e64f399c26961e4efe6975bc398be6aa3d"
|
||||
|
||||
[[package]]
|
||||
name = "rayon"
|
||||
version = "1.12.0"
|
||||
@@ -6210,6 +6317,17 @@ dependencies = [
|
||||
"quote",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "stringprep"
|
||||
version = "0.1.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7b4df3d392d81bd458a8a621b8bffbd2302a12ffe288a9d931670948749463b1"
|
||||
dependencies = [
|
||||
"unicode-bidi",
|
||||
"unicode-normalization",
|
||||
"unicode-properties",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "strsim"
|
||||
version = "0.11.1"
|
||||
@@ -7375,12 +7493,33 @@ version = "2.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dbc4bc3a9f746d862c45cb89d705aa10f187bb96c76001afab07a0d35ce60142"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-bidi"
|
||||
version = "0.3.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5c1cb5db39152898a79168971543b1cb5020dff7fe43c8dc468b0885f5e29df5"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-ident"
|
||||
version = "1.0.24"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e6e4313cd5fcd3dad5cafa179702e2b244f760991f45397d14d4ebf38247da75"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-normalization"
|
||||
version = "0.1.25"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "5fd4f6878c9cb28d874b009da9e8d183b5abc80117c40bbd187a1fde336be6e8"
|
||||
dependencies = [
|
||||
"tinyvec",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "unicode-properties"
|
||||
version = "0.1.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "7df058c713841ad818f1dc5d3fd88063241cc61f49f5fbea4b951e8cf5a8d71d"
|
||||
|
||||
[[package]]
|
||||
name = "unicode-segmentation"
|
||||
version = "1.13.3"
|
||||
|
||||
@@ -229,6 +229,7 @@ clap = { version = "4", features = ["derive"] }
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
# Text Processing & Parsing
|
||||
# ──────────────────────────────────────────────────────────────────────────────
|
||||
pdf-inspector = "1"
|
||||
regex = "1"
|
||||
similar = "3.1.0"
|
||||
unicode-segmentation = "1.13"
|
||||
|
||||
Generated
+596
-425
File diff suppressed because one or more lines are too long
@@ -105,7 +105,6 @@
|
||||
"@opentelemetry/sdk-metrics": "catalog:",
|
||||
"@opentelemetry/sdk-trace-base": "catalog:",
|
||||
"@opentelemetry/sdk-trace-node": "catalog:",
|
||||
"mupdf": "catalog:",
|
||||
"puppeteer-core": "catalog:",
|
||||
},
|
||||
"devDependencies": {
|
||||
@@ -386,7 +385,6 @@
|
||||
"ghostty-web": "^0.4.0",
|
||||
"lint-staged": "^17.0.8",
|
||||
"lucide-react": "^1.24.0",
|
||||
"mupdf": "^1.28.0",
|
||||
"onnxruntime-node": "1.26.0",
|
||||
"postcss": "^8.5.16",
|
||||
"prettier": "^3.9.5",
|
||||
@@ -1219,8 +1217,6 @@
|
||||
|
||||
"ms": ["ms@2.1.3", "", {}, "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA=="],
|
||||
|
||||
"mupdf": ["mupdf@1.28.0", "", {}, "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg=="],
|
||||
|
||||
"mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="],
|
||||
|
||||
"nanoid": ["nanoid@3.3.18", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-DTg4MJbGMWkfi6VZFdNt2/caMbQy4Ou+Op/hJQvGEWcnVfoA1QA+xzRKAzw9jD6+GVOOeYr/mIcuDSdug6F6+w=="],
|
||||
|
||||
@@ -44,6 +44,7 @@ inferno.workspace = true
|
||||
napi.workspace = true
|
||||
napi-derive.workspace = true
|
||||
parking_lot.workspace = true
|
||||
pdf-inspector.workspace = true
|
||||
phf.workspace = true
|
||||
flume.workspace = true
|
||||
pi-ast.workspace = true
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
//!
|
||||
//! # Overview
|
||||
//! High-performance primitives for clipboard access, grep, file discovery,
|
||||
//! ANSI-aware text measurement, syntax highlighting, HTML-to-Markdown
|
||||
//! ANSI-aware text measurement, syntax highlighting, HTML/PDF-to-Markdown
|
||||
//! conversion, and terminal SIXEL encoding.
|
||||
//!
|
||||
//! # Example
|
||||
@@ -15,7 +15,7 @@
|
||||
//!
|
||||
//! # Architecture
|
||||
//! ```text
|
||||
//! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/highlight/sixel/text)
|
||||
//! JS (packages/natives) -> N-API -> Rust modules (clipboard/fd/glob/grep/html/pdf/highlight/sixel/text)
|
||||
//! ```
|
||||
|
||||
#![allow(clippy::trailing_empty_array, reason = "generated by napi macro")]
|
||||
@@ -41,6 +41,8 @@ pub mod html;
|
||||
pub mod iofs;
|
||||
pub mod keys;
|
||||
pub mod live;
|
||||
/// PDF inspection and Markdown conversion.
|
||||
pub mod pdf;
|
||||
pub mod sixel;
|
||||
pub mod snapcompact;
|
||||
pub use pi_ast::language;
|
||||
|
||||
@@ -0,0 +1,165 @@
|
||||
//! PDF inspection and Markdown conversion backed by `pdf-inspector`.
|
||||
|
||||
use napi::{Result, bindgen_prelude::Uint8Array};
|
||||
use napi_derive::napi;
|
||||
use pdf_inspector::{MarkdownOptions, PdfOptions, process_pdf_mem_with_options};
|
||||
|
||||
use crate::task;
|
||||
|
||||
/// Markdown and inspection metadata produced from a PDF document.
|
||||
#[napi(object)]
|
||||
pub struct PdfMarkdownResult {
|
||||
/// Extracted document content in Markdown format.
|
||||
pub markdown: String,
|
||||
/// Document title from PDF metadata, when present.
|
||||
pub title: Option<String>,
|
||||
/// Total number of pages in the document.
|
||||
pub page_count: u32,
|
||||
/// One-indexed page numbers whose content requires OCR.
|
||||
pub pages_needing_ocr: Vec<u32>,
|
||||
/// Whether the document contains text encoding problems.
|
||||
pub has_encoding_issues: bool,
|
||||
}
|
||||
|
||||
/// Convert an in-memory PDF to Markdown and return its inspection metadata.
|
||||
///
|
||||
/// Conversion copies the typed array before dispatch so JavaScript mutation
|
||||
/// cannot race the native worker.
|
||||
///
|
||||
/// # Errors
|
||||
/// Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
|
||||
/// be parsed or converted.
|
||||
#[napi(js_name = "pdfToMarkdown")]
|
||||
pub fn pdf_to_markdown(input: Uint8Array) -> task::Promise<PdfMarkdownResult> {
|
||||
let input = input.to_vec();
|
||||
task::blocking("pdf.to_markdown", (), move |_| convert_pdf(&input))
|
||||
}
|
||||
|
||||
fn convert_pdf(input: &[u8]) -> Result<PdfMarkdownResult> {
|
||||
let options = PdfOptions::new()
|
||||
.markdown(MarkdownOptions { include_page_numbers: true, ..Default::default() });
|
||||
let converted = process_pdf_mem_with_options(input, options)
|
||||
.map_err(|error| napi::Error::from_reason(format!("PDF conversion failed: {error}")))?;
|
||||
let markdown = match converted.markdown {
|
||||
Some(markdown) => markdown,
|
||||
None if !converted.pages_needing_ocr.is_empty() => String::new(),
|
||||
None => {
|
||||
return Err(napi::Error::from_reason(
|
||||
"PDF conversion failed: converter returned no Markdown",
|
||||
));
|
||||
},
|
||||
};
|
||||
|
||||
Ok(PdfMarkdownResult {
|
||||
markdown,
|
||||
title: converted.title,
|
||||
page_count: converted.page_count,
|
||||
pages_needing_ocr: converted.pages_needing_ocr,
|
||||
has_encoding_issues: converted.has_encoding_issues,
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn pdf_fixture(page_contents: &[&str], title: Option<&str>) -> Vec<u8> {
|
||||
let font_id = 3 + page_contents.len() * 2;
|
||||
let info_id = title.map(|_| font_id + 1);
|
||||
let mut objects = Vec::with_capacity(font_id + usize::from(info_id.is_some()));
|
||||
objects.push("<< /Type /Catalog /Pages 2 0 R >>".to_string());
|
||||
|
||||
let kids = (0..page_contents.len())
|
||||
.map(|index| format!("{} 0 R", 3 + index * 2))
|
||||
.collect::<Vec<_>>()
|
||||
.join(" ");
|
||||
objects.push(format!("<< /Type /Pages /Kids [{kids}] /Count {} >>", page_contents.len()));
|
||||
|
||||
for (index, content) in page_contents.iter().enumerate() {
|
||||
let content_id = 4 + index * 2;
|
||||
objects.push(format!(
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 \
|
||||
{font_id} 0 R >> >> /Contents {content_id} 0 R >>"
|
||||
));
|
||||
objects.push(format!("<< /Length {} >>\nstream\n{content}\nendstream", content.len()));
|
||||
}
|
||||
|
||||
objects.push(
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>"
|
||||
.to_string(),
|
||||
);
|
||||
if let Some(title) = title {
|
||||
objects.push(format!("<< /Title ({title}) >>"));
|
||||
}
|
||||
|
||||
let mut pdf = b"%PDF-1.4\n".to_vec();
|
||||
let mut offsets = Vec::with_capacity(objects.len());
|
||||
for (index, object) in objects.iter().enumerate() {
|
||||
offsets.push(pdf.len());
|
||||
pdf.extend_from_slice(format!("{} 0 obj\n{object}\nendobj\n", index + 1).as_bytes());
|
||||
}
|
||||
|
||||
let xref_offset = pdf.len();
|
||||
pdf.extend_from_slice(
|
||||
format!("xref\n0 {}\n0000000000 65535 f \n", objects.len() + 1).as_bytes(),
|
||||
);
|
||||
for offset in offsets {
|
||||
pdf.extend_from_slice(format!("{offset:010} 00000 n \n").as_bytes());
|
||||
}
|
||||
let info = info_id.map_or_else(String::new, |id| format!(" /Info {id} 0 R"));
|
||||
pdf.extend_from_slice(
|
||||
format!(
|
||||
"trailer\n<< /Size {} /Root 1 0 R{info} >>\nstartxref\n{xref_offset}\n%%EOF\n",
|
||||
objects.len() + 1
|
||||
)
|
||||
.as_bytes(),
|
||||
);
|
||||
pdf
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn converts_text_title_and_page_markers() {
|
||||
let pdf = pdf_fixture(
|
||||
&[
|
||||
"BT /F1 12 Tf 72 720 Td (First page text) Tj 0 -18 Td (More first page text) Tj 0 -18 \
|
||||
Td (End first page) Tj ET",
|
||||
"BT /F1 12 Tf 72 720 Td (Second page text) Tj 0 -18 Td (More second page text) Tj 0 \
|
||||
-18 Td (End second page) Tj ET",
|
||||
],
|
||||
Some("Fixture Title"),
|
||||
);
|
||||
|
||||
let result = convert_pdf(&pdf).expect("fixture PDF should convert");
|
||||
|
||||
assert_eq!(result.title.as_deref(), Some("Fixture Title"));
|
||||
assert_eq!(result.page_count, 2);
|
||||
assert!(result.markdown.contains("First page text"), "{}", result.markdown);
|
||||
assert!(result.markdown.contains("Second page text"), "{}", result.markdown);
|
||||
assert!(result.markdown.contains("<!-- Page 1 -->"), "{}", result.markdown);
|
||||
assert!(result.markdown.contains("<!-- Page 2 -->"), "{}", result.markdown);
|
||||
assert!(!result.has_encoding_issues);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reports_empty_pages_as_needing_ocr() {
|
||||
let pdf = pdf_fixture(&[""], None);
|
||||
|
||||
let result = convert_pdf(&pdf).expect("empty-page PDF should still convert");
|
||||
|
||||
assert_eq!(result.page_count, 1);
|
||||
assert_eq!(result.pages_needing_ocr, vec![1]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefixes_malformed_pdf_errors() {
|
||||
let error = convert_pdf(b"not a PDF")
|
||||
.err()
|
||||
.expect("malformed input should fail");
|
||||
|
||||
assert!(
|
||||
error.reason.starts_with("PDF conversion failed:"),
|
||||
"unexpected error: {}",
|
||||
error.reason
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -1738,10 +1738,6 @@
|
||||
url = "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz";
|
||||
hash = "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==";
|
||||
};
|
||||
"mupdf@1.28.0" = fetchurl {
|
||||
url = "https://registry.npmjs.org/mupdf/-/mupdf-1.28.0.tgz";
|
||||
hash = "sha512-ACUnbpECaQ5JLq04pwd89lS+0IGMest5qL5tb08g9TAR7bDtfqflHEkb2Xm3o4rvC/szguLiV+WEbW9kstj8Sg==";
|
||||
};
|
||||
"mute-stream@3.0.0" = fetchurl {
|
||||
url = "https://registry.npmjs.org/mute-stream/-/mute-stream-3.0.0.tgz";
|
||||
hash = "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw==";
|
||||
|
||||
@@ -61,7 +61,6 @@
|
||||
"ghostty-web": "^0.4.0",
|
||||
"lint-staged": "^17.0.8",
|
||||
"lucide-react": "^1.24.0",
|
||||
"mupdf": "^1.28.0",
|
||||
"onnxruntime-node": "1.26.0",
|
||||
"postcss": "^8.5.16",
|
||||
"prettier": "^3.9.5",
|
||||
@@ -170,8 +169,6 @@
|
||||
"gen:nix": "bun scripts/gen-nix-bun.ts",
|
||||
"gen:tool-views": "bun --cwd=packages/collab-web run gen:tool-views",
|
||||
"gen:bundle": "bun --cwd=packages/coding-agent run gen:bundle",
|
||||
"gen:mupdf": "bun --cwd=packages/coding-agent run gen:mupdf",
|
||||
"gen:mupdf:reset": "bun --cwd=packages/coding-agent run gen:mupdf:reset",
|
||||
"gen:native": "bun --cwd=packages/natives run gen:native",
|
||||
"gen:native:reset": "bun --cwd=packages/natives run gen:native:reset",
|
||||
"check-spoofed-versions": "bun scripts/check-spoofed-versions.ts"
|
||||
|
||||
@@ -2,6 +2,11 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
|
||||
- Replaced the MuPDF-WASM PDF document backend with `pdf-inspector` through `@oh-my-pi/pi-natives`, preserving cached text conversion and PDF line selectors while reporting pages that need OCR.
|
||||
- Removed `read <pdf>:` image listings and `read <pdf>:<image>.png` extraction because `pdf-inspector` does not rasterize pages; these reads now direct users to the Puppeteer browser tool for rendering or to read the PDF path for extracted text.
|
||||
|
||||
## [17.3.3] - 2026-08-14
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -41,8 +41,6 @@
|
||||
"format-prompts": "bun scripts/format-prompts.ts",
|
||||
"gen:tool-views": "bun --cwd=../collab-web run gen:tool-views",
|
||||
"gen:bundle": "bun scripts/bundle-dist.ts",
|
||||
"gen:mupdf": "bun scripts/embed-mupdf-wasm.ts --generate",
|
||||
"gen:mupdf:reset": "bun scripts/embed-mupdf-wasm.ts --reset",
|
||||
"gen:native": "bun --cwd=../natives run gen:native",
|
||||
"gen:native:reset": "bun --cwd=../natives run gen:native:reset",
|
||||
"prepack": "bun run gen:tool-views && bun run gen:bundle",
|
||||
@@ -73,7 +71,6 @@
|
||||
"@opentelemetry/sdk-metrics": "catalog:",
|
||||
"@opentelemetry/sdk-trace-base": "catalog:",
|
||||
"@opentelemetry/sdk-trace-node": "catalog:",
|
||||
"mupdf": "catalog:",
|
||||
"puppeteer-core": "catalog:"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
|
||||
@@ -88,7 +88,6 @@ async function main(): Promise<void> {
|
||||
["bun", "--cwd=../natives", "run", "gen:native"],
|
||||
crossBuild ? { ...Bun.env, TARGET_PLATFORM: crossBuild.platform, TARGET_ARCH: crossBuild.arch } : Bun.env,
|
||||
);
|
||||
await runCommand(["bun", "run", "gen:mupdf"]);
|
||||
try {
|
||||
await compileCodingAgent({
|
||||
repoRoot,
|
||||
@@ -104,7 +103,6 @@ async function main(): Promise<void> {
|
||||
await runCommand(["codesign", "--force", "--sign", "-", outputPath]);
|
||||
}
|
||||
} finally {
|
||||
await runCommand(["bun", "run", "gen:mupdf:reset"]);
|
||||
await runCommand(["bun", "--cwd=../natives", "run", "gen:native:reset"]);
|
||||
}
|
||||
} finally {
|
||||
|
||||
@@ -15,7 +15,6 @@ const legacyHtmlExportAssetPattern = /^(?:template-[^.]+\.(?:css|html|js)|tool-v
|
||||
// `omp-legacy-pi-modules` exists only in compiled binaries via the build plugin;
|
||||
// the npm bundle never executes that `isCompiledBinary()` branch.
|
||||
const ALWAYS_EXTERNAL = [
|
||||
"mupdf",
|
||||
"@oh-my-pi/pi-natives",
|
||||
"@huggingface/transformers",
|
||||
"fastembed",
|
||||
|
||||
@@ -1,67 +0,0 @@
|
||||
#!/usr/bin/env bun
|
||||
|
||||
// Embeds mupdf's `mupdf-wasm.wasm` into the compiled single-file binary.
|
||||
//
|
||||
// mupdf loads its wasm by reading the `mupdf-wasm.wasm` sibling of its own
|
||||
// module via `new URL(..., import.meta.url)` + `readFileSync`. A `bun --compile`
|
||||
// binary has no node_modules, so that read fails (`ENOENT .../mupdf-wasm.wasm`),
|
||||
// and marking mupdf `--external` instead makes `bun --compile` eagerly fail to
|
||||
// resolve the package at startup (the static `import * as mupdf` lives in a lazy
|
||||
// chunk but is hoisted). So the binary build bundles mupdf and embeds the wasm
|
||||
// bytes here, handing them to the WASM module as `$libmupdf_wasm_Module.wasmBinary`
|
||||
// (see src/utils/markit.ts).
|
||||
//
|
||||
// `--generate` copies the wasm next to src/utils/mupdf-wasm-embed.ts and rewrites
|
||||
// that module to import it via `with { type: "file" }`; `--reset` restores the
|
||||
// checked-in placeholder and removes the copy. The npm `dist/cli.js` bundle never
|
||||
// runs this — it keeps mupdf external and loads the wasm from node_modules.
|
||||
|
||||
import * as fs from "node:fs/promises";
|
||||
import { createRequire } from "node:module";
|
||||
import * as path from "node:path";
|
||||
|
||||
const utilsDir = path.join(import.meta.dir, "..", "src", "utils");
|
||||
const helperPath = path.join(utilsDir, "mupdf-wasm-embed.ts");
|
||||
const wasmCopyPath = path.join(utilsDir, "mupdf-wasm.wasm");
|
||||
|
||||
const placeholder = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
|
||||
//
|
||||
// Compiled single-file binaries cannot let mupdf resolve its \`mupdf-wasm.wasm\`
|
||||
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
|
||||
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
|
||||
// wasm bytes via \`with { type: "file" }\` and copies the wasm next to it. Source
|
||||
// checkouts, \`bun test\`, and the npm \`dist/cli.js\` bundle keep mupdf external and
|
||||
// load the wasm from node_modules, so this placeholder returns undefined and the
|
||||
// build resets back to it afterward.
|
||||
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
|
||||
\treturn undefined;
|
||||
}
|
||||
`;
|
||||
|
||||
const generated = `// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit or commit.
|
||||
import { readFileSync } from "node:fs";
|
||||
import wasmPath from "./mupdf-wasm.wasm" with { type: "file" };
|
||||
|
||||
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
|
||||
\treturn readFileSync(wasmPath);
|
||||
}
|
||||
`;
|
||||
|
||||
if (process.argv.includes("--reset")) {
|
||||
await Bun.write(helperPath, placeholder);
|
||||
try {
|
||||
await fs.unlink(wasmCopyPath);
|
||||
} catch (err) {
|
||||
if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err;
|
||||
}
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const wasmSource = path.join(path.dirname(createRequire(import.meta.url).resolve("mupdf")), "mupdf-wasm.wasm");
|
||||
const wasmFile = Bun.file(wasmSource);
|
||||
if (!(await wasmFile.exists())) {
|
||||
throw new Error(`mupdf wasm not found at ${wasmSource}; run \`bun install\` first.`);
|
||||
}
|
||||
await Bun.write(wasmCopyPath, wasmFile);
|
||||
await Bun.write(helperPath, generated);
|
||||
console.log(`Embedded mupdf wasm (${wasmFile.size} bytes) into ${path.relative(process.cwd(), wasmCopyPath)}`);
|
||||
@@ -1,15 +1,15 @@
|
||||
This directory contains an in-house document-to-markdown engine adapted from
|
||||
Portions of this in-house document-to-markdown engine are adapted from
|
||||
markit-ai (https://github.com/Michaelliv/markit), used under the MIT License.
|
||||
This attribution covers the shared registry/types and the DOCX, PPTX, XLSX,
|
||||
and EPUB converters. The PDF converter is implemented separately and is not
|
||||
derived from markit-ai.
|
||||
|
||||
Copyright (c) 2026 Michael Liv
|
||||
|
||||
Only the converters for the document formats omp supports are ported (pdf,
|
||||
docx, pptx, xlsx, epub); the CLI, plugin/provider, and unused converters
|
||||
(html, image, audio, plain-text, rss, github, wikipedia, csv, json, yaml,
|
||||
ipynb, iwork, zip, xml) were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and
|
||||
`.rtf` are routed by the read/fetch tools but have no converter — they surface
|
||||
a conversion error, exactly as upstream markit did. Logic is ported faithfully
|
||||
so conversion output matches the upstream package.
|
||||
The CLI, plugin/provider, and unused converters (html, image, audio,
|
||||
plain-text, rss, github, wikipedia, csv, json, yaml, ipynb, iwork, zip, xml)
|
||||
were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and `.rtf` have no converter
|
||||
and surface a conversion error.
|
||||
|
||||
MIT License
|
||||
|
||||
|
||||
@@ -1,103 +0,0 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Multi-column layout detection and text box reordering.
|
||||
*
|
||||
* Many PDFs (legal documents, datasheets, academic papers) use two-column
|
||||
* layouts. Without column detection, text boxes are ordered by Y position
|
||||
* only, interleaving left and right column content.
|
||||
*
|
||||
* Algorithm:
|
||||
* 1. Collect left edges of all text boxes on the page
|
||||
* 2. Find the largest horizontal gap between consecutive left edges
|
||||
* 3. If gap > MIN_GAP_RATIO of the text width and both sides have
|
||||
* enough boxes → multi-column detected
|
||||
* 4. Assign each text box to a column based on its center X
|
||||
* 5. Return columns in reading order (left-to-right, top-to-bottom)
|
||||
*
|
||||
* This only detects the column structure. The caller is responsible for
|
||||
* processing each column's text boxes independently (table detection,
|
||||
* rendering, etc.).
|
||||
*/
|
||||
import type { TextBox } from "./types";
|
||||
|
||||
export interface ColumnLayout {
|
||||
/** Number of columns detected (1 = single column, 2+ = multi-column). */
|
||||
columnCount: number;
|
||||
/** Text boxes grouped by column, in reading order (left to right). */
|
||||
columns: TextBox[][];
|
||||
/** X positions of column boundaries (between columns). */
|
||||
boundaries: number[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimum gap as a fraction of the total text width to consider a column
|
||||
* boundary. A two-column layout typically has ~50% gap; we use a lower
|
||||
* threshold to catch asymmetric columns.
|
||||
*/
|
||||
const MIN_GAP_RATIO = 0.15;
|
||||
/** Minimum number of text boxes on each side of the gap. */
|
||||
const MIN_BOXES_PER_COLUMN = 4;
|
||||
/** Minimum gap in absolute points to avoid splitting on small whitespace. */
|
||||
const MIN_GAP_PTS = 40;
|
||||
|
||||
/**
|
||||
* Detect column layout and return text boxes grouped by column.
|
||||
*
|
||||
* For single-column pages, returns all boxes in one group.
|
||||
* For multi-column pages, returns boxes split by column in reading order.
|
||||
*/
|
||||
export function detectColumns(textBoxes: TextBox[]): ColumnLayout {
|
||||
if (textBoxes.length < MIN_BOXES_PER_COLUMN * 2) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
// Collect unique left edges (rounded to avoid float noise)
|
||||
const lefts = [...new Set(textBoxes.map(tb => Math.round(tb.bounds.left)))].sort((a, b) => a - b);
|
||||
if (lefts.length < 2) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
const textXMin = lefts[0];
|
||||
const textXMax = Math.max(...textBoxes.map(tb => Math.round(tb.bounds.right)));
|
||||
const textWidth = textXMax - textXMin;
|
||||
if (textWidth <= 0) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
// Find the largest gap between consecutive left-edge positions
|
||||
let maxGap = 0;
|
||||
let gapLeft = 0;
|
||||
let gapRight = 0;
|
||||
for (let i = 1; i < lefts.length; i++) {
|
||||
const gap = lefts[i] - lefts[i - 1];
|
||||
if (gap > maxGap) {
|
||||
maxGap = gap;
|
||||
gapLeft = lefts[i - 1];
|
||||
gapRight = lefts[i];
|
||||
}
|
||||
}
|
||||
const gapRatio = maxGap / textWidth;
|
||||
if (gapRatio < MIN_GAP_RATIO || maxGap < MIN_GAP_PTS) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
// Split point is the midpoint of the gap
|
||||
const splitX = (gapLeft + gapRight) / 2;
|
||||
// Assign boxes to columns based on center X
|
||||
const leftCol: TextBox[] = [];
|
||||
const rightCol: TextBox[] = [];
|
||||
for (const tb of textBoxes) {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
if (cx < splitX) {
|
||||
leftCol.push(tb);
|
||||
} else {
|
||||
rightCol.push(tb);
|
||||
}
|
||||
}
|
||||
// Validate both columns have enough content
|
||||
if (leftCol.length < MIN_BOXES_PER_COLUMN || rightCol.length < MIN_BOXES_PER_COLUMN) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
return {
|
||||
columnCount: 2,
|
||||
columns: [leftCol, rightCol],
|
||||
boundaries: [splitX],
|
||||
};
|
||||
}
|
||||
@@ -1,598 +0,0 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* PDF content extraction using mupdf.
|
||||
*
|
||||
* Extracts text boxes (with position, font size, bold) and vector line
|
||||
* segments (table borders) from each page. Uses mupdf's native WASM
|
||||
* engine for fast parsing, and reads raw content streams for vector graphics.
|
||||
*
|
||||
* Coordinate system: PDF native (origin = bottom-left, Y increases upward).
|
||||
*/
|
||||
import type * as mupdf from "mupdf";
|
||||
import type { ImageRegion, PageContent, Segment, TextBox } from "./types";
|
||||
|
||||
// mupdf instantiates its WASM module via a top-level await. A static
|
||||
// `import * as mupdf` would pull that await into this module's init, which makes
|
||||
// the whole bundled markit chunk's `__esm` init async — and bun's compiled
|
||||
// bundler fails to await that init transitively through the `../markit` barrel,
|
||||
// exposing the converter classes before their module-level consts initialize
|
||||
// (e.g. `EXTENSIONS` reads as undefined). Importing mupdf lazily keeps the chunk
|
||||
// init synchronous and also keeps the ~10MB wasm off non-PDF conversions.
|
||||
let mupdfModule: typeof mupdf | undefined;
|
||||
async function loadMupdf(): Promise<typeof mupdf> {
|
||||
if (!mupdfModule) {
|
||||
mupdfModule = await import("mupdf");
|
||||
}
|
||||
return mupdfModule;
|
||||
}
|
||||
|
||||
/** mupdf structured-text JSON bounding box (top-left origin). */
|
||||
interface StextBBox {
|
||||
x: number;
|
||||
y: number;
|
||||
w: number;
|
||||
h: number;
|
||||
}
|
||||
|
||||
/** Font metadata attached to a structured-text line. */
|
||||
interface StextFont {
|
||||
size?: number;
|
||||
weight?: string;
|
||||
name?: string;
|
||||
}
|
||||
|
||||
/** A line within a text block in mupdf structured-text JSON. */
|
||||
interface StextLine {
|
||||
text?: string;
|
||||
font?: StextFont;
|
||||
bbox: StextBBox;
|
||||
}
|
||||
|
||||
/** A block (text or image) in mupdf structured-text JSON. */
|
||||
interface StextBlock {
|
||||
type: string;
|
||||
bbox: StextBBox;
|
||||
lines: StextLine[];
|
||||
}
|
||||
|
||||
/** Parsed mupdf structured-text JSON for a page. */
|
||||
interface StructuredTextJSON {
|
||||
blocks: StextBlock[];
|
||||
}
|
||||
|
||||
/** A raw text fragment before merging into word/phrase boxes. */
|
||||
interface RawTextItem {
|
||||
text: string;
|
||||
x: number;
|
||||
y: number;
|
||||
width: number;
|
||||
height: number;
|
||||
fontSize: number;
|
||||
isBold: boolean;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Text extraction
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Y tolerance for merging text fragments on the same visual line. */
|
||||
const SAME_LINE_Y_TOLERANCE = 2;
|
||||
/** Max horizontal gap (pts) to merge adjacent fragments into one text box. */
|
||||
const MAX_MERGE_GAP = 14;
|
||||
|
||||
/**
|
||||
* Merge horizontally adjacent raw text items on the same visual line into
|
||||
* word/phrase-level text boxes.
|
||||
*/
|
||||
function mergeIntoWords(raws: RawTextItem[]): RawTextItem[] {
|
||||
if (raws.length === 0) return [];
|
||||
// Sort by Y descending (top-first in bottom-left coords), then X ascending
|
||||
const sorted = [...raws].sort((a, b) => {
|
||||
const dy = b.y - a.y;
|
||||
return Math.abs(dy) > SAME_LINE_Y_TOLERANCE ? dy : a.x - b.x;
|
||||
});
|
||||
const merged: RawTextItem[] = [];
|
||||
let cur = { ...sorted[0] };
|
||||
for (let i = 1; i < sorted.length; i++) {
|
||||
const next = sorted[i];
|
||||
const sameY = Math.abs(next.y - cur.y) <= SAME_LINE_Y_TOLERANCE;
|
||||
const close = next.x <= cur.x + cur.width + MAX_MERGE_GAP;
|
||||
if (sameY && close) {
|
||||
const gap = next.x - (cur.x + cur.width);
|
||||
const sep = gap > 1 ? " " : "";
|
||||
cur.text += sep + next.text;
|
||||
cur.width = next.x + next.width - cur.x;
|
||||
cur.height = Math.max(cur.height, next.height);
|
||||
cur.fontSize = Math.max(cur.fontSize, next.fontSize);
|
||||
cur.isBold = cur.isBold || next.isBold;
|
||||
} else {
|
||||
merged.push(cur);
|
||||
cur = { ...next };
|
||||
}
|
||||
}
|
||||
merged.push(cur);
|
||||
return merged;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract text boxes from a mupdf page using structured text output.
|
||||
*
|
||||
* mupdf's structured text JSON uses top-left origin; we convert to
|
||||
* bottom-left (standard PDF coordinates) using the page height.
|
||||
*/
|
||||
function extractTextBoxes(
|
||||
page: mupdf.Page,
|
||||
pageNumber: number,
|
||||
pageHeight: number,
|
||||
stext?: StructuredTextJSON,
|
||||
): TextBox[] {
|
||||
if (!stext) {
|
||||
stext = JSON.parse(page.toStructuredText("preserve-whitespace").asJSON()) as StructuredTextJSON;
|
||||
}
|
||||
const raws: RawTextItem[] = [];
|
||||
for (const block of stext.blocks) {
|
||||
if (block.type !== "text") continue;
|
||||
for (const line of block.lines) {
|
||||
const text = line.text?.trim();
|
||||
if (!text) continue;
|
||||
const fontSize = line.font?.size ?? 0;
|
||||
const weight = line.font?.weight ?? "normal";
|
||||
const fontName = line.font?.name ?? "";
|
||||
const isBold = weight === "bold" || /bold/i.test(fontName) || /Black|Heavy/i.test(fontName);
|
||||
// mupdf bbox: {x, y, w, h} in top-left coords
|
||||
// Convert to bottom-left: pdfY = pageHeight - (bbox.y + bbox.h)
|
||||
const bboxY = line.bbox.y;
|
||||
const bboxH = line.bbox.h;
|
||||
const pdfY = pageHeight - (bboxY + bboxH);
|
||||
raws.push({
|
||||
text,
|
||||
x: line.bbox.x,
|
||||
y: pdfY,
|
||||
width: line.bbox.w,
|
||||
height: bboxH,
|
||||
fontSize,
|
||||
isBold,
|
||||
});
|
||||
}
|
||||
}
|
||||
const words = mergeIntoWords(raws);
|
||||
return words
|
||||
.map((w, i) => ({
|
||||
id: `p${pageNumber}-t${i}`,
|
||||
text: w.text.trim(),
|
||||
pageNumber,
|
||||
fontSize: w.fontSize,
|
||||
isBold: w.isBold,
|
||||
bounds: {
|
||||
left: w.x,
|
||||
right: w.x + w.width,
|
||||
bottom: w.y,
|
||||
top: w.y + w.height,
|
||||
},
|
||||
}))
|
||||
.filter(b => b.text.length > 0);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Vector segment extraction from raw content stream
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Minimum aspect ratio for a filled rect to be considered a line. */
|
||||
const LINE_ASPECT_THRESHOLD = 6;
|
||||
/** Minimum length (pts) for a segment to count. */
|
||||
const MIN_LENGTH = 2;
|
||||
/** Maximum thickness (pts) for a border line (filters out filled areas). */
|
||||
const MAX_THICKNESS = 3;
|
||||
|
||||
/**
|
||||
* Convert a thin filled rectangle to a horizontal or vertical segment.
|
||||
* Returns null if the rect doesn't look like a border line.
|
||||
*/
|
||||
function thinRectToSegment(id: string, x: number, y: number, w: number, h: number): Segment | null {
|
||||
const aw = Math.abs(w);
|
||||
const ah = Math.abs(h);
|
||||
if (aw > ah * LINE_ASPECT_THRESHOLD && aw >= MIN_LENGTH && ah <= MAX_THICKNESS) {
|
||||
// Horizontal line
|
||||
const cy = y + ah / 2;
|
||||
return { id, x1: x, y1: cy, x2: x + aw, y2: cy };
|
||||
}
|
||||
if (ah > aw * LINE_ASPECT_THRESHOLD && ah >= MIN_LENGTH && aw <= MAX_THICKNESS) {
|
||||
// Vertical line
|
||||
const cx = x + aw / 2;
|
||||
return { id, x1: cx, y1: y, x2: cx, y2: y + ah };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Emit 4 edge segments from a stroked rectangle.
|
||||
*/
|
||||
function pushStrokedRectEdges(segments: Segment[], id: string, x: number, y: number, w: number, h: number): void {
|
||||
const aw = Math.abs(w);
|
||||
const ah = Math.abs(h);
|
||||
const base = id;
|
||||
if (aw >= MIN_LENGTH) {
|
||||
segments.push({ id: `${base}-b`, x1: x, y1: y, x2: x + aw, y2: y });
|
||||
segments.push({
|
||||
id: `${base}-t`,
|
||||
x1: x,
|
||||
y1: y + ah,
|
||||
x2: x + aw,
|
||||
y2: y + ah,
|
||||
});
|
||||
}
|
||||
if (ah >= MIN_LENGTH) {
|
||||
segments.push({ id: `${base}-l`, x1: x, y1: y, x2: x, y2: y + ah });
|
||||
segments.push({
|
||||
id: `${base}-r`,
|
||||
x1: x + aw,
|
||||
y1: y,
|
||||
x2: x + aw,
|
||||
y2: y + ah,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const CTM_IDENTITY = [1, 0, 0, 1, 0, 0];
|
||||
|
||||
/** Concatenate two affine matrices: result = parent × child. */
|
||||
function ctmConcat(p: number[], c: number[]): number[] {
|
||||
return [
|
||||
p[0] * c[0] + p[2] * c[1],
|
||||
p[1] * c[0] + p[3] * c[1],
|
||||
p[0] * c[2] + p[2] * c[3],
|
||||
p[1] * c[2] + p[3] * c[3],
|
||||
p[0] * c[4] + p[2] * c[5] + p[4],
|
||||
p[1] * c[4] + p[3] * c[5] + p[5],
|
||||
];
|
||||
}
|
||||
|
||||
function ctmApply(m: number[], x: number, y: number): [number, number] {
|
||||
return [m[0] * x + m[2] * y + m[4], m[1] * x + m[3] * y + m[5]];
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Content stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Parse a PDF content stream and extract line segments from thin filled
|
||||
* rectangles (re+f), stroked rectangles (re+S), and explicit lines (m/l+S).
|
||||
* Tracks the CTM via q/Q/cm operators so coordinates are in page space.
|
||||
*/
|
||||
function extractSegmentsFromContentStream(raw: string, pageNumber: number): Segment[] {
|
||||
const segments: Segment[] = [];
|
||||
const tokens = tokenizeContentStream(raw);
|
||||
let idx = 0;
|
||||
let strokeWidth = 1.0;
|
||||
// Graphics state stack (q/Q): saves CTM + strokeWidth
|
||||
let ctm = [...CTM_IDENTITY];
|
||||
const stateStack: Array<{ ctm: number[]; strokeWidth: number }> = [];
|
||||
// State for path building (in user coordinates, pre-CTM)
|
||||
let curX = 0;
|
||||
let curY = 0;
|
||||
let pathStartX = 0;
|
||||
let pathStartY = 0;
|
||||
const pendingRects: Array<{ x: number; y: number; w: number; h: number }> = [];
|
||||
const pendingLines: Array<{ x1: number; y1: number; x2: number; y2: number }> = [];
|
||||
function flushPath(mode: "fill" | "stroke"): void {
|
||||
const sid = () => `p${pageNumber}-s${segments.length}`;
|
||||
if (mode === "fill") {
|
||||
for (const r of pendingRects) {
|
||||
// Transform the rect corners through CTM, then check if it's a thin line
|
||||
const [x0, y0] = ctmApply(ctm, r.x, r.y);
|
||||
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
|
||||
const seg = thinRectToSegment(
|
||||
sid(),
|
||||
Math.min(x0, x1),
|
||||
Math.min(y0, y1),
|
||||
Math.abs(x1 - x0),
|
||||
Math.abs(y1 - y0),
|
||||
);
|
||||
if (seg) segments.push(seg);
|
||||
}
|
||||
} else if (mode === "stroke" && strokeWidth <= MAX_THICKNESS) {
|
||||
for (const r of pendingRects) {
|
||||
const [x0, y0] = ctmApply(ctm, r.x, r.y);
|
||||
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
|
||||
pushStrokedRectEdges(
|
||||
segments,
|
||||
sid(),
|
||||
Math.min(x0, x1),
|
||||
Math.min(y0, y1),
|
||||
Math.abs(x1 - x0),
|
||||
Math.abs(y1 - y0),
|
||||
);
|
||||
}
|
||||
for (const l of pendingLines) {
|
||||
const [lx1, ly1] = ctmApply(ctm, l.x1, l.y1);
|
||||
const [lx2, ly2] = ctmApply(ctm, l.x2, l.y2);
|
||||
const dx = Math.abs(lx2 - lx1);
|
||||
const dy = Math.abs(ly2 - ly1);
|
||||
// Only keep H/V lines
|
||||
if ((dx >= MIN_LENGTH && dy < 1) || (dy >= MIN_LENGTH && dx < 1)) {
|
||||
segments.push({ id: sid(), x1: lx1, y1: ly1, x2: lx2, y2: ly2 });
|
||||
}
|
||||
}
|
||||
}
|
||||
pendingRects.length = 0;
|
||||
pendingLines.length = 0;
|
||||
}
|
||||
while (idx < tokens.length) {
|
||||
const t = tokens[idx];
|
||||
if (t === "q") {
|
||||
stateStack.push({ ctm: [...ctm], strokeWidth });
|
||||
} else if (t === "Q") {
|
||||
const saved = stateStack.pop();
|
||||
if (saved) {
|
||||
ctm = saved.ctm;
|
||||
strokeWidth = saved.strokeWidth;
|
||||
}
|
||||
} else if (t === "cm" && idx >= 6) {
|
||||
const a = Number(tokens[idx - 6]);
|
||||
const b = Number(tokens[idx - 5]);
|
||||
const c = Number(tokens[idx - 4]);
|
||||
const d = Number(tokens[idx - 3]);
|
||||
const e = Number(tokens[idx - 2]);
|
||||
const f = Number(tokens[idx - 1]);
|
||||
ctm = ctmConcat(ctm, [a, b, c, d, e, f]);
|
||||
} else if (t === "w" && idx >= 1) {
|
||||
strokeWidth = Number(tokens[idx - 1]) || strokeWidth;
|
||||
} else if (t === "re" && idx >= 4) {
|
||||
const x = Number(tokens[idx - 4]);
|
||||
const y = Number(tokens[idx - 3]);
|
||||
const w = Number(tokens[idx - 2]);
|
||||
const h = Number(tokens[idx - 1]);
|
||||
if (Number.isFinite(x + y + w + h)) {
|
||||
pendingRects.push({ x, y, w, h });
|
||||
}
|
||||
} else if (t === "m" && idx >= 2) {
|
||||
curX = Number(tokens[idx - 2]);
|
||||
curY = Number(tokens[idx - 1]);
|
||||
pathStartX = curX;
|
||||
pathStartY = curY;
|
||||
} else if (t === "l" && idx >= 2) {
|
||||
const x2 = Number(tokens[idx - 2]);
|
||||
const y2 = Number(tokens[idx - 1]);
|
||||
pendingLines.push({ x1: curX, y1: curY, x2, y2 });
|
||||
curX = x2;
|
||||
curY = y2;
|
||||
} else if (t === "h") {
|
||||
// closePath: line back to start
|
||||
if (curX !== pathStartX || curY !== pathStartY) {
|
||||
pendingLines.push({
|
||||
x1: curX,
|
||||
y1: curY,
|
||||
x2: pathStartX,
|
||||
y2: pathStartY,
|
||||
});
|
||||
}
|
||||
curX = pathStartX;
|
||||
curY = pathStartY;
|
||||
} else if (t === "f" || t === "F" || t === "f*") {
|
||||
flushPath("fill");
|
||||
} else if (t === "S" || t === "s") {
|
||||
if (t === "s") {
|
||||
// closeStroke: implicit closePath
|
||||
if (curX !== pathStartX || curY !== pathStartY) {
|
||||
pendingLines.push({
|
||||
x1: curX,
|
||||
y1: curY,
|
||||
x2: pathStartX,
|
||||
y2: pathStartY,
|
||||
});
|
||||
}
|
||||
}
|
||||
flushPath("stroke");
|
||||
} else if (t === "B" || t === "B*" || t === "b" || t === "b*") {
|
||||
// fill + stroke combined
|
||||
flushPath("fill");
|
||||
flushPath("stroke");
|
||||
} else if (t === "n") {
|
||||
// end path without painting — discard
|
||||
pendingRects.length = 0;
|
||||
pendingLines.length = 0;
|
||||
}
|
||||
idx++;
|
||||
}
|
||||
return segments;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fast tokenizer for PDF content streams.
|
||||
* Splits on whitespace, skipping comments, string literals, and inline image payloads.
|
||||
*/
|
||||
function tokenizeContentStream(raw: string): string[] {
|
||||
const tokens: string[] = [];
|
||||
const len = raw.length;
|
||||
let i = 0;
|
||||
let inInlineImage = false;
|
||||
while (i < len) {
|
||||
const ch = raw.charCodeAt(i);
|
||||
// Skip whitespace
|
||||
if (ch <= 32) {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
// Skip comments
|
||||
if (ch === 37 /* % */) {
|
||||
while (i < len && raw.charCodeAt(i) !== 10) i++;
|
||||
continue;
|
||||
}
|
||||
// Skip string literals (...)
|
||||
if (ch === 40 /* ( */) {
|
||||
let depth = 1;
|
||||
i++;
|
||||
while (i < len && depth > 0) {
|
||||
const c = raw.charCodeAt(i);
|
||||
if (c === 92 /* \ */) {
|
||||
i++;
|
||||
} else if (c === 40) {
|
||||
depth++;
|
||||
} else if (c === 41) {
|
||||
depth--;
|
||||
}
|
||||
i++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
// Skip hex strings <...>
|
||||
if (ch === 60 /* < */ && i + 1 < len && raw.charCodeAt(i + 1) !== 60) {
|
||||
i++;
|
||||
while (i < len && raw.charCodeAt(i) !== 62) i++;
|
||||
i++; // skip >
|
||||
continue;
|
||||
}
|
||||
// Skip dict delimiters << >>
|
||||
if (ch === 60 && i + 1 < len && raw.charCodeAt(i + 1) === 60) {
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
if (ch === 62 && i + 1 < len && raw.charCodeAt(i + 1) === 62) {
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
// Skip stray closing delimiters from malformed streams. They cannot start
|
||||
// a token, so leaving i unchanged would spin forever.
|
||||
if (ch === 41 || ch === 62) {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
// Regular token: read until whitespace or delimiter
|
||||
const start = i;
|
||||
while (i < len) {
|
||||
const c = raw.charCodeAt(i);
|
||||
if (c <= 32 || c === 40 || c === 41 || c === 60 || c === 62 || c === 37) break;
|
||||
i++;
|
||||
}
|
||||
if (i > start) {
|
||||
const token = raw.substring(start, i);
|
||||
tokens.push(token);
|
||||
if (token === "BI") {
|
||||
inInlineImage = true;
|
||||
} else if (token === "ID" && inInlineImage) {
|
||||
while (i < len && raw.charCodeAt(i) <= 32) i++;
|
||||
while (i < len) {
|
||||
const c = raw.charCodeAt(i);
|
||||
const prev = i === 0 ? 32 : raw.charCodeAt(i - 1);
|
||||
const next = i + 2 >= len ? 32 : raw.charCodeAt(i + 2);
|
||||
if (c === 69 && raw.charCodeAt(i + 1) === 73 && prev <= 32 && next <= 32) {
|
||||
i += 2;
|
||||
break;
|
||||
}
|
||||
i++;
|
||||
}
|
||||
inInlineImage = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
return tokens;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Image region detection
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Minimum area (pts²) for an image to be considered a diagram, not an icon. */
|
||||
const MIN_IMAGE_AREA = 5000;
|
||||
|
||||
function extractImageRegions(stext: StructuredTextJSON, pageNumber: number, pageHeight: number): ImageRegion[] {
|
||||
const regions: ImageRegion[] = [];
|
||||
for (const block of stext.blocks) {
|
||||
if (block.type !== "image") continue;
|
||||
const { x, y, w, h } = block.bbox;
|
||||
if (w * h < MIN_IMAGE_AREA) continue; // skip tiny icons
|
||||
// Convert Y from mupdf (top-left) to PDF (bottom-left) for ordering
|
||||
const pdfTopY = pageHeight - y;
|
||||
regions.push({
|
||||
id: `p${pageNumber}-img${regions.length}`,
|
||||
pageNumber,
|
||||
bbox: { x, y, w, h },
|
||||
topY: pdfTopY,
|
||||
});
|
||||
}
|
||||
return regions;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Render an image region from a PDF page as a PNG buffer.
|
||||
* Uses mupdf's DrawDevice to render just the cropped area at 2x resolution.
|
||||
*/
|
||||
export async function renderImageRegion(input: Uint8Array, region: ImageRegion): Promise<Uint8Array> {
|
||||
const m = await loadMupdf();
|
||||
const doc = m.Document.openDocument(input, "application/pdf");
|
||||
const page = doc.loadPage(region.pageNumber - 1);
|
||||
const pad = 10;
|
||||
const bx = region.bbox.x - pad;
|
||||
const by = region.bbox.y - pad;
|
||||
const bw = region.bbox.w + 2 * pad;
|
||||
const bh = region.bbox.h + 2 * pad;
|
||||
const scale = 2;
|
||||
const pw = Math.round(bw * scale);
|
||||
const ph = Math.round(bh * scale);
|
||||
const pix = new m.Pixmap(m.ColorSpace.DeviceRGB, [0, 0, pw, ph], false);
|
||||
pix.clear(255);
|
||||
const matrix: mupdf.Matrix = [scale, 0, 0, scale, -bx * scale, -by * scale];
|
||||
const dl = page.toDisplayList();
|
||||
const dev = new m.DrawDevice(matrix, pix);
|
||||
dl.run(dev, m.Matrix.identity);
|
||||
dev.close();
|
||||
return pix.asPNG();
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract text boxes and vector segments from all pages of a PDF buffer.
|
||||
*/
|
||||
export async function extractPages(input: Uint8Array): Promise<PageContent[]> {
|
||||
const m = await loadMupdf();
|
||||
const doc = m.Document.openDocument(input, "application/pdf");
|
||||
const pages: PageContent[] = [];
|
||||
for (let i = 0; i < doc.countPages(); i++) {
|
||||
const pageNumber = i + 1;
|
||||
const page = doc.loadPage(i);
|
||||
const bounds = page.getBounds();
|
||||
const pageHeight = bounds[3] - bounds[1];
|
||||
// Single structured text pass with both flags
|
||||
const stext = JSON.parse(
|
||||
page.toStructuredText("preserve-whitespace,preserve-images").asJSON(),
|
||||
) as StructuredTextJSON;
|
||||
// Extract text boxes and image regions from the same parse
|
||||
const textBoxes = extractTextBoxes(page, pageNumber, pageHeight, stext);
|
||||
const images = extractImageRegions(stext, pageNumber, pageHeight);
|
||||
// Extract vector segments from raw content stream
|
||||
let segments: Segment[] = [];
|
||||
try {
|
||||
const pageObj = (page as mupdf.PDFPage).getObject();
|
||||
const contents = pageObj.get("Contents");
|
||||
if (contents) {
|
||||
let rawBytes: Uint8Array;
|
||||
if (contents.isArray()) {
|
||||
// Multiple content streams — concatenate
|
||||
const parts: Uint8Array[] = [];
|
||||
const len = contents.length ?? 0;
|
||||
for (let j = 0; j < len; j++) {
|
||||
const stream = contents.get(j);
|
||||
if (stream?.readStream) {
|
||||
parts.push(stream.readStream().asUint8Array());
|
||||
}
|
||||
}
|
||||
const totalLen = parts.reduce((s, p) => s + p.length, 0);
|
||||
rawBytes = new Uint8Array(totalLen);
|
||||
let offset = 0;
|
||||
for (const part of parts) {
|
||||
rawBytes.set(part, offset);
|
||||
offset += part.length;
|
||||
}
|
||||
} else {
|
||||
rawBytes = contents.readStream().asUint8Array();
|
||||
}
|
||||
const raw = new TextDecoder().decode(rawBytes);
|
||||
segments = extractSegmentsFromContentStream(raw, pageNumber);
|
||||
}
|
||||
} catch {
|
||||
// Content stream extraction failed — proceed with text only
|
||||
}
|
||||
pages.push({ pageNumber, textBoxes, segments, images });
|
||||
}
|
||||
return pages;
|
||||
}
|
||||
@@ -1,780 +0,0 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Table grid detection from vector segments and text boxes.
|
||||
*
|
||||
* Ported from @oharato/pdf2md-ts with TypeScript types and without
|
||||
* CJK-specific borderless table heuristics. The core algorithm:
|
||||
*
|
||||
* 1. Classify segments as horizontal or vertical lines
|
||||
* 2. Group horizontal Y-lines into table groups (split by vertical gaps)
|
||||
* 3. For each group:
|
||||
* a. Full grid (H+V lines): build cells from grid intersections,
|
||||
* place text via raycasting
|
||||
* b. H-line only (no V lines): infer columns from text X positions
|
||||
* 4. Prune empty rows/cols
|
||||
*
|
||||
* Coordinate system: PDF native (bottom-left origin, Y increases upward).
|
||||
*/
|
||||
import type { Segment, TableCell, TableGrid, TextBox } from "./types";
|
||||
|
||||
export interface GridResult {
|
||||
grids: TableGrid[];
|
||||
consumedIds: string[];
|
||||
}
|
||||
|
||||
type RayDirection = "up" | "down" | "left" | "right";
|
||||
|
||||
interface Ray {
|
||||
direction: RayDirection;
|
||||
segmentId: string | null;
|
||||
distance: number;
|
||||
}
|
||||
|
||||
interface Interval {
|
||||
min: number;
|
||||
max: number;
|
||||
}
|
||||
|
||||
function castRaysForTextBox(textBox: TextBox, segments: Segment[]): Ray[] {
|
||||
const cx = (textBox.bounds.left + textBox.bounds.right) / 2;
|
||||
const cy = (textBox.bounds.top + textBox.bounds.bottom) / 2;
|
||||
let up: Ray = { direction: "up", segmentId: null, distance: Infinity };
|
||||
let down: Ray = { direction: "down", segmentId: null, distance: Infinity };
|
||||
let left: Ray = { direction: "left", segmentId: null, distance: Infinity };
|
||||
let right: Ray = {
|
||||
direction: "right",
|
||||
segmentId: null,
|
||||
distance: Infinity,
|
||||
};
|
||||
for (const seg of segments) {
|
||||
const isH = Math.abs(seg.y1 - seg.y2) < 0.5;
|
||||
const isV = Math.abs(seg.x1 - seg.x2) < 0.5;
|
||||
if (isH) {
|
||||
const minX = Math.min(seg.x1, seg.x2);
|
||||
const maxX = Math.max(seg.x1, seg.x2);
|
||||
if (cx >= minX && cx <= maxX) {
|
||||
const d = seg.y1 - cy;
|
||||
if (d >= 0 && d < up.distance) up = { direction: "up", segmentId: seg.id, distance: d };
|
||||
const dd = cy - seg.y1;
|
||||
if (dd >= 0 && dd < down.distance) down = { direction: "down", segmentId: seg.id, distance: dd };
|
||||
}
|
||||
}
|
||||
if (isV) {
|
||||
const minY = Math.min(seg.y1, seg.y2);
|
||||
const maxY = Math.max(seg.y1, seg.y2);
|
||||
if (cy >= minY && cy <= maxY) {
|
||||
const d = cx - seg.x1;
|
||||
if (d >= 0 && d < left.distance) left = { direction: "left", segmentId: seg.id, distance: d };
|
||||
const rd = seg.x1 - cx;
|
||||
if (rd >= 0 && rd < right.distance) right = { direction: "right", segmentId: seg.id, distance: rd };
|
||||
}
|
||||
}
|
||||
}
|
||||
return [up, down, left, right];
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Utility
|
||||
// ---------------------------------------------------------------------------
|
||||
const AXIS_EPSILON = 0.8;
|
||||
const PAGE_MARGIN = 20;
|
||||
|
||||
function uniqueSorted(values: number[]): number[] {
|
||||
const sorted = [...values].sort((a, b) => a - b);
|
||||
const result: number[] = [];
|
||||
for (const v of sorted) {
|
||||
if (result.length === 0 || Math.abs(result[result.length - 1] - v) > 1) result.push(v);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Y-line group splitting
|
||||
// ---------------------------------------------------------------------------
|
||||
function chainCoversRange(intervals: Interval[], lowerY: number, upperY: number, eps: number): boolean {
|
||||
const sorted = [...intervals].sort((a, b) => a.min - b.min);
|
||||
let covered = lowerY;
|
||||
for (const iv of sorted) {
|
||||
if (iv.min > covered + eps) break;
|
||||
if (iv.max > covered) covered = iv.max;
|
||||
if (covered >= upperY - eps) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function countBridgingVLineCols(upperY: number, lowerY: number, verticals: Segment[]): number {
|
||||
const eps = 1.5;
|
||||
const byX = new Map<number, Interval[]>();
|
||||
for (const seg of verticals) {
|
||||
const rx = Math.round(seg.x1);
|
||||
if (!byX.has(rx)) byX.set(rx, []);
|
||||
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
|
||||
}
|
||||
let count = 0;
|
||||
for (const intervals of byX.values()) {
|
||||
if (chainCoversRange(intervals, lowerY, upperY, eps)) count++;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
function bridgingXSet(upperY: number, lowerY: number, verticals: Segment[]): Set<number> {
|
||||
const eps = 1.5;
|
||||
const xs = new Set<number>();
|
||||
const byX = new Map<number, Interval[]>();
|
||||
for (const seg of verticals) {
|
||||
const rx = Math.round(seg.x1);
|
||||
if (!byX.has(rx)) byX.set(rx, []);
|
||||
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
|
||||
}
|
||||
for (const [rx, intervals] of byX) {
|
||||
if (chainCoversRange(intervals, lowerY, upperY, eps)) xs.add(rx);
|
||||
}
|
||||
return xs;
|
||||
}
|
||||
|
||||
const MIN_RICH_BRIDGING_COLS = 3;
|
||||
|
||||
function splitYLinesIntoGroups(yLines: number[], verticals: Segment[]): number[][] {
|
||||
if (yLines.length === 0) return [];
|
||||
const eps = 1.5;
|
||||
const allX = verticals.map(s => Math.round(s.x1));
|
||||
const globalXMin = allX.length > 0 ? Math.min(...allX) : 0;
|
||||
const globalXMax = allX.length > 0 ? Math.max(...allX) : 0;
|
||||
const groups: number[][] = [];
|
||||
let currentGroup = [yLines[0]];
|
||||
let prevBridgingCols = -1;
|
||||
for (let i = 1; i < yLines.length; i++) {
|
||||
const upperY = yLines[i - 1];
|
||||
const lowerY = yLines[i];
|
||||
const cols = countBridgingVLineCols(upperY, lowerY, verticals);
|
||||
if (cols === 0) {
|
||||
groups.push(currentGroup);
|
||||
currentGroup = [yLines[i]];
|
||||
prevBridgingCols = -1;
|
||||
continue;
|
||||
}
|
||||
if (prevBridgingCols >= MIN_RICH_BRIDGING_COLS && cols < MIN_RICH_BRIDGING_COLS) {
|
||||
const bxs = bridgingXSet(upperY, lowerY, verticals);
|
||||
const isOuterFrameOnly = [...bxs].every(
|
||||
x => Math.abs(x - globalXMin) <= eps || Math.abs(x - globalXMax) <= eps,
|
||||
);
|
||||
if (!isOuterFrameOnly) {
|
||||
groups.push(currentGroup);
|
||||
currentGroup = [yLines[i - 1], yLines[i]];
|
||||
prevBridgingCols = cols;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
currentGroup.push(yLines[i]);
|
||||
prevBridgingCols = cols;
|
||||
}
|
||||
groups.push(currentGroup);
|
||||
return groups;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Sub-row Y-cluster expansion
|
||||
// ---------------------------------------------------------------------------
|
||||
const Y_CLUSTER_GAP = 10;
|
||||
const MIN_COLS_IN_TOP_CLUSTER = 2;
|
||||
|
||||
function assignToYCluster(y: number, clusters: number[]): number {
|
||||
let closest = 0;
|
||||
let closestDist = Math.abs(y - clusters[0]);
|
||||
for (let k = 1; k < clusters.length; k++) {
|
||||
const d = Math.abs(y - clusters[k]);
|
||||
if (d < closestDist) {
|
||||
closestDist = d;
|
||||
closest = k;
|
||||
}
|
||||
}
|
||||
return closest;
|
||||
}
|
||||
|
||||
function expandSubRowsByYClusters(
|
||||
originalRows: number,
|
||||
cols: number,
|
||||
cells: TableCell[],
|
||||
cellBoxes: Map<TableCell, TextBox[]>,
|
||||
): number {
|
||||
let addedRows = 0;
|
||||
for (let origRow = 0; origRow < originalRows; origRow++) {
|
||||
const currentRow = origRow + addedRows;
|
||||
const rowCellInfos: Array<{ cell: TableCell; col: number; boxes: TextBox[] }> = [];
|
||||
for (let col = 0; col < cols; col++) {
|
||||
const cell = cells.find(c => c.row === currentRow && c.col === col);
|
||||
if (!cell) continue;
|
||||
const boxes = cellBoxes.get(cell);
|
||||
if (boxes && boxes.length > 0) rowCellInfos.push({ cell, col, boxes });
|
||||
}
|
||||
if (rowCellInfos.length === 0) continue;
|
||||
const allMidYs = rowCellInfos.flatMap(({ boxes }) => boxes.map(b => (b.bounds.top + b.bounds.bottom) / 2));
|
||||
const sortedY = [...new Set(allMidYs.map(y => Math.round(y * 10) / 10))].sort((a, b) => b - a);
|
||||
const clusters = [sortedY[0]];
|
||||
for (let i = 1; i < sortedY.length; i++) {
|
||||
if (clusters[clusters.length - 1] - sortedY[i] > Y_CLUSTER_GAP) {
|
||||
clusters.push(sortedY[i]);
|
||||
}
|
||||
}
|
||||
if (clusters.length < 2) continue;
|
||||
const colsInTopCluster = new Set<number>();
|
||||
const totalNonEmptyCols = new Set<number>();
|
||||
for (const { col, boxes } of rowCellInfos) {
|
||||
totalNonEmptyCols.add(col);
|
||||
if (boxes.some(b => assignToYCluster((b.bounds.top + b.bounds.bottom) / 2, clusters) === 0)) {
|
||||
colsInTopCluster.add(col);
|
||||
}
|
||||
}
|
||||
if (colsInTopCluster.size < MIN_COLS_IN_TOP_CLUSTER) continue;
|
||||
if (colsInTopCluster.size >= totalNonEmptyCols.size) continue;
|
||||
const sparseColsHaveMultipleBoxes = rowCellInfos.some(
|
||||
({ col, boxes }) => !colsInTopCluster.has(col) && boxes.length > 1,
|
||||
);
|
||||
if (!sparseColsHaveMultipleBoxes) continue;
|
||||
const numSubRows = clusters.length;
|
||||
const numNewRows = numSubRows - 1;
|
||||
for (const cell of cells) {
|
||||
if (cell.row > currentRow) cell.row += numNewRows;
|
||||
}
|
||||
for (let subRow = 1; subRow < numSubRows; subRow++) {
|
||||
for (let col = 0; col < cols; col++) {
|
||||
cells.push({
|
||||
row: currentRow + subRow,
|
||||
col,
|
||||
text: "",
|
||||
rowSpan: 1,
|
||||
colSpan: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
for (const { cell: origCell, col, boxes } of rowCellInfos) {
|
||||
const subRowBoxGroups: TextBox[][] = Array.from({ length: numSubRows }, () => []);
|
||||
for (const box of boxes) {
|
||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
||||
subRowBoxGroups[assignToYCluster(cy, clusters)].push(box);
|
||||
}
|
||||
cellBoxes.set(origCell, subRowBoxGroups[0]);
|
||||
if (subRowBoxGroups[0].length === 0) cellBoxes.delete(origCell);
|
||||
for (let subRow = 1; subRow < numSubRows; subRow++) {
|
||||
if (subRowBoxGroups[subRow].length > 0) {
|
||||
const newCell = cells.find(c => c.row === currentRow + subRow && c.col === col);
|
||||
if (newCell) cellBoxes.set(newCell, subRowBoxGroups[subRow]);
|
||||
}
|
||||
}
|
||||
}
|
||||
addedRows += numNewRows;
|
||||
}
|
||||
return originalRows + addedRows;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Cross-column text box splitting
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Find which column a horizontal position falls into.
|
||||
* Returns -1 if outside the grid.
|
||||
*/
|
||||
function findCol(x: number, xLines: number[]): number {
|
||||
for (let i = 0; i < xLines.length - 1; i++) {
|
||||
if (x >= xLines[i] && x <= xLines[i + 1]) return i;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/**
|
||||
* When a text box spans across one or more vertical column boundaries,
|
||||
* split it into multiple virtual text boxes — one per column — with the
|
||||
* text divided proportionally by width.
|
||||
*
|
||||
* We split at word boundaries closest to the proportional split point
|
||||
* so we don't chop words in half.
|
||||
*/
|
||||
function splitCrossColumnBoxes(textBoxes: TextBox[], xLines: number[]): TextBox[] {
|
||||
const result: TextBox[] = [];
|
||||
const MARGIN = 5; // allow small overlap before considering it cross-column
|
||||
for (const tb of textBoxes) {
|
||||
const leftCol = findCol(tb.bounds.left + MARGIN, xLines);
|
||||
const rightCol = findCol(tb.bounds.right - MARGIN, xLines);
|
||||
// Not spanning columns, or outside grid — keep as-is
|
||||
if (leftCol < 0 || rightCol < 0 || leftCol === rightCol) {
|
||||
result.push(tb);
|
||||
continue;
|
||||
}
|
||||
// Text box spans from leftCol to rightCol — split it
|
||||
const totalWidth = tb.bounds.right - tb.bounds.left;
|
||||
if (totalWidth <= 0) {
|
||||
result.push(tb);
|
||||
continue;
|
||||
}
|
||||
const words = tb.text.split(/\s+/);
|
||||
if (words.length <= 1) {
|
||||
// Single word spanning columns — just assign to whichever col has more overlap
|
||||
result.push(tb);
|
||||
continue;
|
||||
}
|
||||
// For each column boundary crossing, find the best word-boundary split
|
||||
let remainingWords = [...words];
|
||||
let currentLeft = tb.bounds.left;
|
||||
for (let col = leftCol; col <= rightCol && remainingWords.length > 0; col++) {
|
||||
const colRight = col < xLines.length - 1 ? xLines[col + 1] : tb.bounds.right;
|
||||
const segmentRight = Math.min(colRight, tb.bounds.right);
|
||||
if (col === rightCol) {
|
||||
// Last column — take all remaining words
|
||||
result.push({
|
||||
...tb,
|
||||
id: `${tb.id}-split${col}`,
|
||||
text: remainingWords.join(" "),
|
||||
bounds: {
|
||||
...tb.bounds,
|
||||
left: currentLeft,
|
||||
right: tb.bounds.right,
|
||||
},
|
||||
});
|
||||
remainingWords = [];
|
||||
} else {
|
||||
// Find how many words fit in this column segment proportionally
|
||||
const segmentWidth = segmentRight - currentLeft;
|
||||
const fractionOfTotal = segmentWidth / totalWidth;
|
||||
const approxChars = Math.round(fractionOfTotal * tb.text.length);
|
||||
// Walk words to find the split closest to the proportional point
|
||||
let charCount = 0;
|
||||
let splitIdx = 0;
|
||||
for (let w = 0; w < remainingWords.length; w++) {
|
||||
const nextCount = charCount + remainingWords[w].length + (w > 0 ? 1 : 0);
|
||||
if (nextCount > approxChars && splitIdx > 0) break;
|
||||
charCount = nextCount;
|
||||
splitIdx = w + 1;
|
||||
}
|
||||
if (splitIdx === 0) splitIdx = 1; // take at least one word
|
||||
if (splitIdx >= remainingWords.length) {
|
||||
// All remaining words fit here
|
||||
result.push({
|
||||
...tb,
|
||||
id: `${tb.id}-split${col}`,
|
||||
text: remainingWords.join(" "),
|
||||
bounds: {
|
||||
...tb.bounds,
|
||||
left: currentLeft,
|
||||
right: segmentRight,
|
||||
},
|
||||
});
|
||||
remainingWords = [];
|
||||
} else {
|
||||
const partWords = remainingWords.slice(0, splitIdx);
|
||||
result.push({
|
||||
...tb,
|
||||
id: `${tb.id}-split${col}`,
|
||||
text: partWords.join(" "),
|
||||
bounds: {
|
||||
...tb.bounds,
|
||||
left: currentLeft,
|
||||
right: segmentRight,
|
||||
},
|
||||
});
|
||||
remainingWords = remainingWords.slice(splitIdx);
|
||||
currentLeft = segmentRight;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Full grid table (H + V lines)
|
||||
// ---------------------------------------------------------------------------
|
||||
function buildCells(rows: number, cols: number): TableCell[] {
|
||||
const cells: TableCell[] = [];
|
||||
for (let row = 0; row < rows; row++) {
|
||||
for (let col = 0; col < cols; col++) {
|
||||
cells.push({ row, col, text: "", rowSpan: 1, colSpan: 1 });
|
||||
}
|
||||
}
|
||||
return cells;
|
||||
}
|
||||
|
||||
function buildTableGrid(
|
||||
pageNumber: number,
|
||||
yLines: number[],
|
||||
xLines: number[],
|
||||
filteredSegments: Segment[],
|
||||
textBoxes: TextBox[],
|
||||
): { grid: TableGrid; consumedIds: string[] } {
|
||||
let rows = yLines.length - 1;
|
||||
const cols = xLines.length - 1;
|
||||
const cells = buildCells(rows, cols);
|
||||
const consumedIds: string[] = [];
|
||||
const yMin = yLines[yLines.length - 1];
|
||||
const yMax = yLines[0];
|
||||
const xMin = xLines[0];
|
||||
const xMax = xLines[xLines.length - 1];
|
||||
// Split text boxes that span multiple columns before placement
|
||||
const splitBoxes = splitCrossColumnBoxes(textBoxes, xLines);
|
||||
// Track which split piece IDs get placed in cells, so we can consume
|
||||
// the original (unsplit) text box IDs too.
|
||||
const placedSplitIds = new Set<string>();
|
||||
// Look for header text boxes just above the grid.
|
||||
// Use the ORIGINAL (unsplit) text boxes for header detection so that
|
||||
// wide paragraph text isn't falsely split into column-sized header chunks.
|
||||
// Reject boxes wider than 1.5 columns — those are paragraph text, not headers.
|
||||
const avgColWidth = (xMax - xMin) / cols;
|
||||
const maxHeaderBoxWidth = avgColWidth * 1.5;
|
||||
const headerBoxes = textBoxes.filter(tb => {
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const boxWidth = tb.bounds.right - tb.bounds.left;
|
||||
return cy > yMax && cy <= yMax + 20 && cx >= xMin && cx <= xMax && boxWidth <= maxHeaderBoxWidth;
|
||||
});
|
||||
if (headerBoxes.length > 0) {
|
||||
rows += 1;
|
||||
for (const cell of cells) cell.row += 1;
|
||||
for (let col = 0; col < cols; col++) {
|
||||
cells.push({ row: 0, col, text: "", rowSpan: 1, colSpan: 1 });
|
||||
}
|
||||
for (const tb of headerBoxes) {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const col = xLines.findIndex((lineX, idx) => {
|
||||
const next = xLines[idx + 1];
|
||||
return next !== undefined && cx >= lineX && cx <= next;
|
||||
});
|
||||
if (col >= 0 && col < cols) {
|
||||
const cell = cells.find(c => c.row === 0 && c.col === col);
|
||||
if (cell) {
|
||||
cell.text = cell.text.length === 0 ? tb.text : `${cell.text} ${tb.text}`;
|
||||
consumedIds.push(tb.id);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
const cellBoxes = new Map<TableCell, TextBox[]>();
|
||||
for (const tb of splitBoxes) {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (cy < yMin || cy > yMax || cx < xMin || cx > xMax) continue;
|
||||
const rays = castRaysForTextBox(tb, filteredSegments);
|
||||
const rayConfidence = rays.filter(r => r.segmentId !== null).length;
|
||||
let row = yLines.findIndex((lineY, idx) => {
|
||||
const next = yLines[idx + 1];
|
||||
return next !== undefined && cy <= lineY && cy >= next;
|
||||
});
|
||||
if (row < 0 || row >= (headerBoxes.length > 0 ? rows - 1 : rows)) continue;
|
||||
if (headerBoxes.length > 0) row += 1;
|
||||
const col = xLines.findIndex((lineX, idx) => {
|
||||
const next = xLines[idx + 1];
|
||||
return next !== undefined && cx >= lineX && cx <= next;
|
||||
});
|
||||
if (col < 0 || col >= cols) continue;
|
||||
if (rayConfidence === 0) continue;
|
||||
const cell = cells.find(c => c.row === row && c.col === col);
|
||||
if (!cell) continue;
|
||||
if (!cellBoxes.has(cell)) cellBoxes.set(cell, []);
|
||||
cellBoxes.get(cell)?.push(tb);
|
||||
consumedIds.push(tb.id);
|
||||
if (tb.id.includes("-split")) placedSplitIds.add(tb.id);
|
||||
}
|
||||
rows = expandSubRowsByYClusters(rows, cols, cells, cellBoxes);
|
||||
// Merge text boxes within each cell into cell text
|
||||
for (const [cell, boxes] of cellBoxes.entries()) {
|
||||
boxes.sort((a, b) => b.bounds.top - a.bounds.top);
|
||||
const lines: string[] = [];
|
||||
let currentLine: string[] = [];
|
||||
let currentY = boxes[0].bounds.top;
|
||||
for (const box of boxes) {
|
||||
if (Math.abs(box.bounds.top - currentY) > 5) {
|
||||
lines.push(currentLine.join(" "));
|
||||
currentLine = [box.text];
|
||||
currentY = box.bounds.top;
|
||||
} else {
|
||||
currentLine.push(box.text);
|
||||
}
|
||||
}
|
||||
if (currentLine.length > 0) lines.push(currentLine.join(" "));
|
||||
cell.text = lines.join("<br>");
|
||||
}
|
||||
const grid = pruneEmptyRowsAndCols({
|
||||
pageNumber,
|
||||
rows,
|
||||
cols,
|
||||
cells,
|
||||
warnings: [],
|
||||
topY: yLines[0],
|
||||
isBorderless: false,
|
||||
});
|
||||
// Also consume the original (unsplit) text box IDs when any of their
|
||||
// split pieces were placed in a cell.
|
||||
for (const splitId of placedSplitIds) {
|
||||
const origId = splitId.replace(/-split\d+$/, "");
|
||||
if (!consumedIds.includes(origId)) {
|
||||
consumedIds.push(origId);
|
||||
}
|
||||
}
|
||||
return { grid, consumedIds };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// H-line-only table (inferred columns)
|
||||
// ---------------------------------------------------------------------------
|
||||
const COL_GAP_THRESHOLD = 20;
|
||||
const HONLY_ROW_GAP = 30;
|
||||
const HONLY_ROW_TOLERANCE = 8;
|
||||
const MIN_TABLE_HEIGHT = 24;
|
||||
const MIN_LEFT_SPREAD = 50;
|
||||
|
||||
function inferXLinesFromBoxes(textBoxes: TextBox[], xMin: number, xMax: number): number[] {
|
||||
const centers = textBoxes.map(tb => (tb.bounds.left + tb.bounds.right) / 2).sort((a, b) => a - b);
|
||||
if (centers.length === 0) return [xMin, xMax];
|
||||
const boundaries = [xMin];
|
||||
for (let i = 1; i < centers.length; i++) {
|
||||
if (centers[i] - centers[i - 1] >= COL_GAP_THRESHOLD) {
|
||||
boundaries.push((centers[i - 1] + centers[i]) / 2);
|
||||
}
|
||||
}
|
||||
boundaries.push(xMax);
|
||||
return boundaries;
|
||||
}
|
||||
|
||||
function buildHLineOnlyTable(
|
||||
pageNumber: number,
|
||||
yLines: number[],
|
||||
xMin: number,
|
||||
xMax: number,
|
||||
textBoxes: TextBox[],
|
||||
alreadyConsumed: Set<string>,
|
||||
): { grid: TableGrid; consumedIds: string[] } | null {
|
||||
const yMax = yLines[0];
|
||||
const yMin = yLines[yLines.length - 1];
|
||||
const candidates = textBoxes.filter(tb => !alreadyConsumed.has(tb.id));
|
||||
const BOX_LEFT_TOLERANCE = 30;
|
||||
const inRange = candidates.filter(tb => {
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
return (
|
||||
tb.bounds.left >= xMin - BOX_LEFT_TOLERANCE &&
|
||||
tb.bounds.right <= xMax + BOX_LEFT_TOLERANCE &&
|
||||
cy >= yMin &&
|
||||
cy <= yMax
|
||||
);
|
||||
});
|
||||
// Extend downward below yMin
|
||||
const belowYMin = candidates
|
||||
.filter(tb => {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
return cx >= xMin && cx <= xMax && cy < yMin;
|
||||
})
|
||||
.sort((a, b) => (b.bounds.top + b.bounds.bottom) / 2 - (a.bounds.top + a.bounds.bottom) / 2);
|
||||
const extensionBoxes: TextBox[] = [];
|
||||
let lastY = yMin;
|
||||
for (const tb of belowYMin) {
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (lastY - cy > HONLY_ROW_GAP) break;
|
||||
extensionBoxes.push(tb);
|
||||
lastY = cy;
|
||||
}
|
||||
const allBoxes = [...inRange, ...extensionBoxes];
|
||||
if (allBoxes.length === 0) return null;
|
||||
const leftEdges = allBoxes.map(tb => tb.bounds.left);
|
||||
if (Math.max(...leftEdges) - Math.min(...leftEdges) < MIN_LEFT_SPREAD) return null;
|
||||
const xLines = inferXLinesFromBoxes(allBoxes, xMin, xMax);
|
||||
if (xLines.length < 2) return null;
|
||||
const cols = xLines.length - 1;
|
||||
// Build visual rows
|
||||
const visualRows: Array<{ midY: number; boxes: TextBox[] }> = [];
|
||||
const sortedBoxes = [...allBoxes].sort((a, b) => {
|
||||
const ya = (a.bounds.top + a.bounds.bottom) / 2;
|
||||
const yb = (b.bounds.top + b.bounds.bottom) / 2;
|
||||
if (Math.abs(ya - yb) > 0.5) return yb - ya;
|
||||
return a.bounds.left - b.bounds.left;
|
||||
});
|
||||
for (const box of sortedBoxes) {
|
||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
||||
const last = visualRows[visualRows.length - 1];
|
||||
if (last && Math.abs(last.midY - cy) <= HONLY_ROW_TOLERANCE) {
|
||||
last.boxes.push(box);
|
||||
} else {
|
||||
visualRows.push({ midY: cy, boxes: [box] });
|
||||
}
|
||||
}
|
||||
if (visualRows.length === 0) return null;
|
||||
const cells: TableCell[] = [];
|
||||
const consumedIds: string[] = [];
|
||||
for (let rowIdx = 0; rowIdx < visualRows.length; rowIdx++) {
|
||||
const vrow = visualRows[rowIdx];
|
||||
const colBoxes = new Map<number, TextBox[]>();
|
||||
for (const box of vrow.boxes) {
|
||||
const cx = (box.bounds.left + box.bounds.right) / 2;
|
||||
const col = xLines.findIndex((lineX, idx) => {
|
||||
const next = xLines[idx + 1];
|
||||
return next !== undefined && cx >= lineX && cx <= next;
|
||||
});
|
||||
if (col >= 0 && col < cols) {
|
||||
if (!colBoxes.has(col)) colBoxes.set(col, []);
|
||||
colBoxes.get(col)?.push(box);
|
||||
}
|
||||
}
|
||||
for (let c = 0; c < cols; c++) {
|
||||
const cbs = (colBoxes.get(c) ?? []).sort((a, b) => a.bounds.left - b.bounds.left);
|
||||
cells.push({
|
||||
row: rowIdx,
|
||||
col: c,
|
||||
text: cbs.map(b => b.text).join(" "),
|
||||
rowSpan: 1,
|
||||
colSpan: 1,
|
||||
});
|
||||
consumedIds.push(...cbs.map(b => b.id));
|
||||
}
|
||||
}
|
||||
const contentTopY = visualRows.length > 0 ? visualRows[0].midY : yMax;
|
||||
const grid = pruneEmptyRowsAndCols({
|
||||
pageNumber,
|
||||
rows: visualRows.length,
|
||||
cols,
|
||||
cells,
|
||||
warnings: [],
|
||||
topY: contentTopY,
|
||||
isBorderless: false,
|
||||
});
|
||||
return { grid, consumedIds };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pruning
|
||||
// ---------------------------------------------------------------------------
|
||||
function pruneEmptyRowsAndCols(table: TableGrid): TableGrid {
|
||||
const occupiedRows = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.row));
|
||||
const occupiedCols = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.col));
|
||||
if (occupiedRows.size === 0) return table;
|
||||
const rowMap = new Map<number, number>();
|
||||
let newRow = 0;
|
||||
for (let r = 0; r < table.rows; r++) {
|
||||
if (occupiedRows.has(r)) rowMap.set(r, newRow++);
|
||||
}
|
||||
const colMap = new Map<number, number>();
|
||||
let newCol = 0;
|
||||
for (let c = 0; c < table.cols; c++) {
|
||||
if (occupiedCols.has(c)) colMap.set(c, newCol++);
|
||||
}
|
||||
const prunedCells = table.cells
|
||||
.filter(c => occupiedRows.has(c.row) && occupiedCols.has(c.col))
|
||||
.map(c => ({
|
||||
...c,
|
||||
row: rowMap.get(c.row) ?? c.row,
|
||||
col: colMap.get(c.col) ?? c.col,
|
||||
}));
|
||||
return { ...table, rows: newRow, cols: newCol, cells: prunedCells };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Diagram vs table discrimination
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Maximum column count for a plausible data table. */
|
||||
const MAX_TABLE_COLS = 25;
|
||||
|
||||
/**
|
||||
* Returns true if a grid looks like a vector diagram rather than a data table.
|
||||
*
|
||||
* Heuristics (any match → diagram):
|
||||
* 1. Column count > 25 (diagrams create many X-lines from box edges)
|
||||
* 2. Fill ratio < 25% (most cells empty — scattered boxes)
|
||||
* 3. Fill < 50% AND duplicate text ratio > 30% (repeating labels in a
|
||||
* diagram layout, e.g. "Hash", "Transaction" appearing in each column)
|
||||
* 4. Fill < 50% AND cols >= 6 (moderate sparseness with wide grid)
|
||||
*/
|
||||
function isDiagram(grid: TableGrid): boolean {
|
||||
const totalCells = grid.rows * grid.cols;
|
||||
if (totalCells === 0) return true;
|
||||
const filled = grid.cells.filter(c => c.text.trim().length > 0);
|
||||
const fillRatio = filled.length / totalCells;
|
||||
// Very high column count
|
||||
if (grid.cols > MAX_TABLE_COLS) return true;
|
||||
// Very sparse
|
||||
if (fillRatio < 0.25) return true;
|
||||
// Compute duplicate text ratio among non-trivial cells.
|
||||
// Exclude short values (≤3 chars) like "—", "V", "YES", "NO" which
|
||||
// naturally repeat in real data tables.
|
||||
const substantive = filled.filter(c => c.text.trim().length > 3);
|
||||
const uniqueTexts = new Set(substantive.map(c => c.text.trim())).size;
|
||||
const dupRatio = substantive.length > 2 ? 1 - uniqueTexts / substantive.length : 0;
|
||||
// Sparse + highly duplicated substantive text → repeating diagram
|
||||
if (fillRatio < 0.5 && dupRatio > 0.3) return true;
|
||||
// High duplication + wide grid → repeating diagram even at moderate fill
|
||||
if (dupRatio > 0.4 && grid.cols >= 6) return true;
|
||||
// Sparse + wide grid with no substantive text to judge
|
||||
if (fillRatio < 0.4 && grid.cols >= 6) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect all table grids on a single page from its text boxes and segments.
|
||||
*/
|
||||
export function resolveTableGrids(pageNumber: number, textBoxes: TextBox[], segments: Segment[]): GridResult {
|
||||
const vertical = segments.filter(s => Math.abs(s.x1 - s.x2) <= AXIS_EPSILON);
|
||||
const horizontal = segments.filter(s => Math.abs(s.y1 - s.y2) <= AXIS_EPSILON);
|
||||
// Filter segments to the text's visible area
|
||||
const textYValues = textBoxes.flatMap(t => [t.bounds.bottom, t.bounds.top]);
|
||||
const textYMin = textYValues.length > 0 ? Math.min(...textYValues) - PAGE_MARGIN : -Infinity;
|
||||
const textYMax = textYValues.length > 0 ? Math.max(...textYValues) + PAGE_MARGIN : Infinity;
|
||||
const textXValues = textBoxes.flatMap(t => [t.bounds.left, t.bounds.right]);
|
||||
const textXMin = textXValues.length > 0 ? Math.min(...textXValues) - 100 : -Infinity;
|
||||
const textXMax = textXValues.length > 0 ? Math.max(...textXValues) + 100 : Infinity;
|
||||
const filteredH = horizontal.filter(
|
||||
s => s.y1 >= textYMin && s.y1 <= textYMax && s.x1 <= textXMax && s.x2 >= textXMin,
|
||||
);
|
||||
const hMaxX2 = filteredH.length > 0 ? Math.max(...filteredH.map(s => s.x2)) : textXMax;
|
||||
const vSegXMax = Math.max(textXMax, hMaxX2 + PAGE_MARGIN);
|
||||
const filteredV = vertical.filter(s => {
|
||||
const segMin = Math.min(s.y1, s.y2);
|
||||
const segMax = Math.max(s.y1, s.y2);
|
||||
return segMax >= textYMin && segMin <= textYMax && s.x1 >= textXMin && s.x1 <= vSegXMax;
|
||||
});
|
||||
const allYLines = uniqueSorted(filteredH.flatMap(s => [s.y1, s.y2])).sort((a, b) => b - a);
|
||||
if (allYLines.length < 2) {
|
||||
return { grids: [], consumedIds: [] };
|
||||
}
|
||||
const filteredSegments = [...filteredH, ...filteredV];
|
||||
const yGroups = splitYLinesIntoGroups(allYLines, filteredV);
|
||||
const grids: TableGrid[] = [];
|
||||
const gridConsumedIds: string[][] = [];
|
||||
// Flat set for the alreadyConsumed check in H-line-only tables
|
||||
const allConsumedIds: string[] = [];
|
||||
for (const yLines of yGroups) {
|
||||
if (yLines.length < 2) continue;
|
||||
const yMin = yLines[yLines.length - 1];
|
||||
const yMax = yLines[0];
|
||||
const groupVerticals = filteredV.filter(s => {
|
||||
const segMin = Math.min(s.y1, s.y2);
|
||||
const segMax = Math.max(s.y1, s.y2);
|
||||
return segMin < yMax - 1.5 && segMax > yMin + 1.5;
|
||||
});
|
||||
const groupXLines = uniqueSorted(groupVerticals.flatMap(s => [s.x1, s.x2]));
|
||||
if (groupXLines.length < 2) {
|
||||
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
|
||||
const groupHoriz = filteredH.filter(s => s.y1 >= yMin - 1.5 && s.y1 <= yMax + 1.5);
|
||||
if (groupHoriz.length === 0) continue;
|
||||
const hxMin = Math.min(...groupHoriz.map(s => s.x1));
|
||||
const hxMax = Math.max(...groupHoriz.map(s => s.x2));
|
||||
const result = buildHLineOnlyTable(pageNumber, yLines, hxMin, hxMax, textBoxes, new Set(allConsumedIds));
|
||||
if (result) {
|
||||
grids.push(result.grid);
|
||||
gridConsumedIds.push(result.consumedIds);
|
||||
allConsumedIds.push(...result.consumedIds);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
|
||||
const result = buildTableGrid(pageNumber, yLines, groupXLines, filteredSegments, textBoxes);
|
||||
grids.push(result.grid);
|
||||
gridConsumedIds.push(result.consumedIds);
|
||||
allConsumedIds.push(...result.consumedIds);
|
||||
}
|
||||
// Filter out grids that look like vector diagrams, not data tables.
|
||||
// Their consumed text box IDs are released so the text becomes free text.
|
||||
const filteredGrids: TableGrid[] = [];
|
||||
const filteredConsumedIds: string[] = [];
|
||||
for (let i = 0; i < grids.length; i++) {
|
||||
if (isDiagram(grids[i])) continue;
|
||||
filteredGrids.push(grids[i]);
|
||||
filteredConsumedIds.push(...gridConsumedIds[i]);
|
||||
}
|
||||
return { grids: filteredGrids, consumedIds: filteredConsumedIds };
|
||||
}
|
||||
@@ -1,106 +0,0 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Running header/footer detection and removal.
|
||||
*
|
||||
* Many PDFs have repeated text at the top or bottom of every page:
|
||||
* document titles, chapter names, page numbers, copyright notices.
|
||||
* These pollute the markdown output as false headings or noise.
|
||||
*
|
||||
* Algorithm:
|
||||
* 1. For each page, bucket text boxes by Y position (top/bottom zones)
|
||||
* 2. Collect the text content at each zone across all pages
|
||||
* 3. Text appearing on >20% of pages OR 8+ consecutive pages is a
|
||||
* running header/footer
|
||||
* 4. Remove matching text boxes before further processing
|
||||
*/
|
||||
import type { PageContent } from "./types";
|
||||
|
||||
/** Minimum number of pages to enable header/footer detection. */
|
||||
const MIN_PAGES = 5;
|
||||
/** Minimum Y position for top zone (from bottom of page in PDF coords). */
|
||||
const TOP_ZONE_MIN_Y = 700;
|
||||
/** Maximum Y position for bottom zone. */
|
||||
const BOTTOM_ZONE_MAX_Y = 80;
|
||||
/**
|
||||
* Minimum consecutive pages a text must appear on to be considered a
|
||||
* running header/footer. Catches both document-wide headers (appearing
|
||||
* on every page) and chapter-specific headers (appearing on 4+ consecutive
|
||||
* pages within a chapter).
|
||||
*/
|
||||
const MIN_CONSECUTIVE_PAGES = 8;
|
||||
|
||||
/**
|
||||
* Detect and remove running headers and footers from all pages.
|
||||
* Mutates the pages array in place, removing header/footer text boxes.
|
||||
*
|
||||
* Uses two strategies:
|
||||
* 1. Global frequency: text appearing on > 20% of all pages
|
||||
* 2. Consecutive runs: text appearing on 8+ consecutive pages
|
||||
*/
|
||||
export function stripHeadersFooters(pages: PageContent[]): void {
|
||||
if (pages.length < MIN_PAGES) return;
|
||||
// Step 1: Build per-page zone text sets
|
||||
const pageZoneTexts: Set<string>[] = [];
|
||||
for (const page of pages) {
|
||||
const zoneTexts = new Set<string>();
|
||||
for (const tb of page.textBoxes) {
|
||||
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (midY >= TOP_ZONE_MIN_Y || midY <= BOTTOM_ZONE_MAX_Y) {
|
||||
const key = tb.text.trim().replace(/\s+/g, " ");
|
||||
if (key.length > 0) zoneTexts.add(key);
|
||||
}
|
||||
}
|
||||
pageZoneTexts.push(zoneTexts);
|
||||
}
|
||||
// Step 2: Count global frequency AND longest consecutive run for each text
|
||||
const globalCount = new Map<string, number>();
|
||||
const maxConsecutive = new Map<string, number>();
|
||||
// Collect all unique zone texts
|
||||
const allTexts = new Set<string>();
|
||||
for (const zts of pageZoneTexts) {
|
||||
for (const t of zts) allTexts.add(t);
|
||||
}
|
||||
for (const text of allTexts) {
|
||||
let total = 0;
|
||||
let consecutive = 0;
|
||||
let maxRun = 0;
|
||||
for (const zts of pageZoneTexts) {
|
||||
if (zts.has(text)) {
|
||||
total++;
|
||||
consecutive++;
|
||||
if (consecutive > maxRun) maxRun = consecutive;
|
||||
} else {
|
||||
consecutive = 0;
|
||||
}
|
||||
}
|
||||
globalCount.set(text, total);
|
||||
maxConsecutive.set(text, maxRun);
|
||||
}
|
||||
// Step 3: Identify running headers/footers
|
||||
const globalThreshold = Math.max(3, Math.floor(pages.length * 0.2));
|
||||
const repeatedTexts = new Set<string>();
|
||||
for (const text of allTexts) {
|
||||
const gc = globalCount.get(text) ?? 0;
|
||||
const mc = maxConsecutive.get(text) ?? 0;
|
||||
// Global: appears on 20%+ of pages
|
||||
if (gc >= globalThreshold) {
|
||||
repeatedTexts.add(text);
|
||||
continue;
|
||||
}
|
||||
// Consecutive: appears on 8+ consecutive pages (chapter-level headers)
|
||||
if (mc >= MIN_CONSECUTIVE_PAGES) {
|
||||
repeatedTexts.add(text);
|
||||
}
|
||||
}
|
||||
if (repeatedTexts.size === 0) return;
|
||||
// Step 4: Remove matching text boxes from each page
|
||||
for (const page of pages) {
|
||||
page.textBoxes = page.textBoxes.filter(tb => {
|
||||
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (midY < TOP_ZONE_MIN_Y && midY > BOTTOM_ZONE_MAX_Y) return true;
|
||||
const normalized = tb.text.trim().replace(/\s+/g, " ");
|
||||
return !repeatedTexts.has(normalized);
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -1,48 +1,10 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* PDF to Markdown converter.
|
||||
*
|
||||
* Uses mupdf (native WASM) for fast PDF parsing and a custom pipeline for
|
||||
* table detection via vector line extraction + raycasting.
|
||||
*
|
||||
* Pipeline:
|
||||
* 1. Extract text boxes + vector segments + image regions per page (mupdf)
|
||||
* 2. Detect column layout (single vs multi-column)
|
||||
* 3. Per column: detect table grids from segments (grid detection + raycasting)
|
||||
* 4. Render diagrams as PNG files (if output directory provided)
|
||||
* 5. Render tables as markdown tables, free text as paragraphs/headings
|
||||
*/
|
||||
import * as path from "node:path";
|
||||
import { pdfToMarkdown } from "@oh-my-pi/pi-natives";
|
||||
import type { ConversionResult, Converter, StreamInfo } from "../../types";
|
||||
import { detectColumns } from "./columns";
|
||||
import { extractPages, renderImageRegion } from "./extract";
|
||||
import { resolveTableGrids } from "./grid";
|
||||
import { stripHeadersFooters } from "./headers";
|
||||
import { renderPageContent } from "./render";
|
||||
import type { Segment, TextBox } from "./types";
|
||||
|
||||
const EXTENSIONS = [".pdf"];
|
||||
const MIMETYPES = ["application/pdf", "application/x-pdf"];
|
||||
|
||||
type ImageBlock = { topY: number; markdown: string };
|
||||
|
||||
/**
|
||||
* Process a set of text boxes (one column or full page): run table detection,
|
||||
* separate free text, and render to markdown.
|
||||
*/
|
||||
function processColumn(
|
||||
pageNumber: number,
|
||||
textBoxes: TextBox[],
|
||||
segments: Segment[],
|
||||
imageBlocks: ImageBlock[],
|
||||
): string {
|
||||
const { grids, consumedIds } = resolveTableGrids(pageNumber, textBoxes, segments);
|
||||
const consumedSet = new Set(consumedIds);
|
||||
const freeTextBoxes = textBoxes.filter(tb => !consumedSet.has(tb.id));
|
||||
return renderPageContent(freeTextBoxes, grids, imageBlocks, textBoxes);
|
||||
}
|
||||
|
||||
/** Converts PDF buffers to Markdown through the native `pdf-inspector` bridge. */
|
||||
export class PdfConverter implements Converter {
|
||||
name = "pdf";
|
||||
|
||||
@@ -56,91 +18,17 @@ export class PdfConverter implements Converter {
|
||||
return false;
|
||||
}
|
||||
|
||||
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const pdfBytes = new Uint8Array(input);
|
||||
const pages = await extractPages(pdfBytes);
|
||||
// Remove running headers/footers before processing.
|
||||
stripHeadersFooters(pages);
|
||||
const imageDir = streamInfo.imageDir;
|
||||
async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const result = await pdfToMarkdown(input);
|
||||
const notice =
|
||||
result.pagesNeedingOcr.length > 0
|
||||
? `Text extraction is incomplete for PDF pages ${result.pagesNeedingOcr.join(", ")}. Use the browser tool to render those pages or OCR them.`
|
||||
: undefined;
|
||||
|
||||
const pageMarkdowns: string[] = [];
|
||||
for (const page of pages) {
|
||||
// Build image blocks for this page.
|
||||
const imageBlocks: ImageBlock[] = [];
|
||||
if (imageDir && page.images.length > 0) {
|
||||
for (const img of page.images) {
|
||||
const filename = `${img.id}.png`;
|
||||
const filepath = path.join(imageDir, filename);
|
||||
try {
|
||||
const png = await renderImageRegion(pdfBytes, img);
|
||||
await Bun.write(filepath, png);
|
||||
imageBlocks.push({ topY: img.topY, markdown: `` });
|
||||
} catch {
|
||||
// Image rendering failed — skip.
|
||||
}
|
||||
}
|
||||
} else if (page.images.length > 0) {
|
||||
for (const img of page.images) {
|
||||
imageBlocks.push({
|
||||
topY: img.topY,
|
||||
markdown: `<!-- image: ${img.id} (page ${img.pageNumber}, ${img.bbox.w}x${img.bbox.h}pt) -->`,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Detect column layout.
|
||||
// If the page has vertical segments (tables), suppress column detection
|
||||
// when one detected column is very narrow — that's a table's first column,
|
||||
// not a page layout column.
|
||||
const layout = detectColumns(page.textBoxes);
|
||||
if (layout.columnCount > 1 && page.segments.some(s => Math.abs(s.x1 - s.x2) <= 0.8)) {
|
||||
const pageXMin = Math.min(...page.textBoxes.map(tb => tb.bounds.left));
|
||||
const pageXMax = Math.max(...page.textBoxes.map(tb => tb.bounds.right));
|
||||
const pageWidth = pageXMax - pageXMin;
|
||||
const minColFraction = 0.3;
|
||||
const tooNarrow = layout.columns.some(col => {
|
||||
const colXMin = Math.min(...col.map(tb => tb.bounds.left));
|
||||
const colXMax = Math.max(...col.map(tb => tb.bounds.right));
|
||||
return (colXMax - colXMin) / pageWidth < minColFraction;
|
||||
});
|
||||
if (tooNarrow) {
|
||||
layout.columnCount = 1;
|
||||
layout.columns = [page.textBoxes];
|
||||
layout.boundaries = [];
|
||||
}
|
||||
}
|
||||
|
||||
if (layout.columnCount === 1) {
|
||||
// Single column — process normally.
|
||||
const md = processColumn(page.pageNumber, page.textBoxes, page.segments, imageBlocks);
|
||||
if (md.length > 0) pageMarkdowns.push(md);
|
||||
} else {
|
||||
// Multi-column — process each column independently, then join.
|
||||
const columnMarkdowns: string[] = [];
|
||||
for (const colBoxes of layout.columns) {
|
||||
// Filter segments to those within this column's X range.
|
||||
const colXMin = Math.min(...colBoxes.map(tb => tb.bounds.left));
|
||||
const colXMax = Math.max(...colBoxes.map(tb => tb.bounds.right));
|
||||
const margin = 10;
|
||||
const colSegments = page.segments.filter(seg => {
|
||||
const segXMin = Math.min(seg.x1, seg.x2);
|
||||
const segXMax = Math.max(seg.x1, seg.x2);
|
||||
return segXMax >= colXMin - margin && segXMin <= colXMax + margin;
|
||||
});
|
||||
// Images go with the first column only (no X info to split by).
|
||||
const md = processColumn(
|
||||
page.pageNumber,
|
||||
colBoxes,
|
||||
colSegments,
|
||||
columnMarkdowns.length === 0 ? imageBlocks : [],
|
||||
);
|
||||
if (md.length > 0) columnMarkdowns.push(md);
|
||||
}
|
||||
const joined = columnMarkdowns.join("\n\n");
|
||||
if (joined.length > 0) pageMarkdowns.push(joined);
|
||||
}
|
||||
}
|
||||
|
||||
return { markdown: pageMarkdowns.join("\n\n") };
|
||||
const conversion: ConversionResult = {
|
||||
markdown: notice ? [result.markdown, notice].filter(Boolean).join("\n\n") : result.markdown,
|
||||
};
|
||||
if (result.title !== undefined) conversion.title = result.title;
|
||||
return conversion;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,501 +0,0 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Markdown rendering for PDF pages.
|
||||
*
|
||||
* Converts table grids and free text boxes into markdown, handling:
|
||||
* - Table grid → markdown table (`| col | col |`)
|
||||
* - Free text → paragraphs with heading detection (by font size)
|
||||
* - Content ordering (top-to-bottom via Y coordinate)
|
||||
* - Paragraph wrap merging (lines broken across PDF line boundaries)
|
||||
* - Page number removal
|
||||
*
|
||||
* Ported from @oharato/pdf2md-ts, stripped of CJK/TDnet-specific logic.
|
||||
*/
|
||||
import type { ContentBlock, TableGrid, TextBox } from "./types";
|
||||
|
||||
/** A free-text line grouped from horizontally adjacent text boxes. */
|
||||
interface RenderLine {
|
||||
text: string;
|
||||
topY: number;
|
||||
fontSize: number;
|
||||
isBold: boolean;
|
||||
isTabular: boolean;
|
||||
}
|
||||
|
||||
/** A content block carrying the Y of its last wrapped line during merging. */
|
||||
type WrapBlock = ContentBlock & { lastTopY: number };
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Utility
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Convert full-width ASCII characters (A→A, !→! etc.) to normal ASCII. */
|
||||
function normalizeFullWidthAscii(text: string): string {
|
||||
return text.replace(/[!-~]/g, ch => String.fromCharCode(ch.charCodeAt(0) - 0xfee0));
|
||||
}
|
||||
|
||||
function escapePipes(text: string): string {
|
||||
return normalizeFullWidthAscii(text).replaceAll("|", "\\|").replaceAll("\n", "<br>");
|
||||
}
|
||||
|
||||
/** Parse a markdown pipe-delimited row into cell strings. */
|
||||
function parsePipeRow(line: string): string[] {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed.startsWith("|") || !trimmed.endsWith("|")) return [];
|
||||
return trimmed
|
||||
.slice(1, -1)
|
||||
.split("|")
|
||||
.map(cell => cell.trim());
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Table rendering
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Render a TableGrid as a markdown table.
|
||||
*/
|
||||
export function renderTableToMarkdown(table: TableGrid): string {
|
||||
if (table.rows === 0 || table.cols === 0) return "";
|
||||
const matrix = Array.from({ length: table.rows }, () => Array.from({ length: table.cols }, () => ""));
|
||||
for (const cell of table.cells) {
|
||||
if (cell.row < table.rows && cell.col < table.cols) {
|
||||
matrix[cell.row][cell.col] = escapePipes(cell.text.trim());
|
||||
}
|
||||
}
|
||||
const normalized = normalizeShiftedSparseColumns(matrix);
|
||||
const promoted = promoteSubHeaderPrefixes(normalized);
|
||||
const header = `| ${promoted[0].join(" | ")} |`;
|
||||
const divider = `| ${Array.from({ length: promoted[0].length }, () => "---").join(" | ")} |`;
|
||||
const body = promoted
|
||||
.slice(1)
|
||||
.map(row => `| ${row.join(" | ")} |`)
|
||||
.join("\n");
|
||||
return [header, divider, body].filter(l => l.length > 0).join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
* Fix tables with ≥5 columns where sparse single-value columns are
|
||||
* misaligned. Shifts those values to the adjacent dense column and
|
||||
* removes the now-empty sparse columns.
|
||||
*/
|
||||
function normalizeShiftedSparseColumns(matrix: string[][]): string[][] {
|
||||
if (matrix.length === 0 || matrix[0].length < 5) return matrix;
|
||||
const _rows = matrix.length;
|
||||
const cols = matrix[0].length;
|
||||
const counts = Array.from({ length: cols }, (_, c) =>
|
||||
matrix.reduce((n, row) => n + (row[c].trim().length > 0 ? 1 : 0), 0),
|
||||
);
|
||||
const denseCols = new Set(
|
||||
counts
|
||||
.map((count, col) => ({ count, col }))
|
||||
.filter(({ col, count }) => col === 0 || count >= 2)
|
||||
.map(({ col }) => col),
|
||||
);
|
||||
const sparseCols = counts
|
||||
.map((count, col) => ({ count, col }))
|
||||
.filter(({ col, count }) => col > 0 && col < cols - 1 && count === 1)
|
||||
.map(({ col }) => col);
|
||||
if (sparseCols.length < 2 || denseCols.size < 4) return matrix;
|
||||
const moves: Array<{ from: number; to: number; row: number }> = [];
|
||||
for (const from of sparseCols) {
|
||||
const row = matrix.findIndex(r => r[from].trim().length > 0);
|
||||
const to = from + 1;
|
||||
if (row < 0) return matrix;
|
||||
if (!denseCols.has(to)) return matrix;
|
||||
if (matrix[row][to].trim().length > 0) return matrix;
|
||||
moves.push({ from, to, row });
|
||||
}
|
||||
const copy = matrix.map(row => [...row]);
|
||||
for (const { from, to, row } of moves) {
|
||||
copy[row][to] = copy[row][to].trim().length > 0 ? `${copy[row][to]} ${copy[row][from]}` : copy[row][from];
|
||||
copy[row][from] = "";
|
||||
}
|
||||
const keepCols = Array.from({ length: cols }, (_, c) => c).filter(c => copy.some(row => row[c].trim().length > 0));
|
||||
if (keepCols.length === cols) return copy;
|
||||
return copy.map(row => keepCols.map(c => row[c]));
|
||||
}
|
||||
|
||||
/**
|
||||
* When a data row has ≥2 parenthesized qualifiers in non-first columns
|
||||
* (and the first column is empty), promote them into the header row.
|
||||
*/
|
||||
function promoteSubHeaderPrefixes(matrix: string[][]): string[][] {
|
||||
if (matrix.length < 2) return matrix;
|
||||
const PAREN_RE = /^\([^)]{1,40}\)$/;
|
||||
const result = matrix.map(row => [...row]);
|
||||
const cols = matrix[0].length;
|
||||
const rowsToRemove = new Set<number>();
|
||||
for (let r = 1; r < result.length; r++) {
|
||||
if (rowsToRemove.has(r)) continue;
|
||||
const promotable: Array<{ col: number; prefix: string; isFullCell: boolean }> = [];
|
||||
for (let col = 1; col < cols; col++) {
|
||||
const cell = (result[r][col] ?? "").trim();
|
||||
if (!cell) continue;
|
||||
const parts = cell.split("<br>");
|
||||
if (parts.length === 1 && PAREN_RE.test(cell)) {
|
||||
promotable.push({ col, prefix: cell, isFullCell: true });
|
||||
} else if (parts.length >= 2 && PAREN_RE.test(parts[0].trim())) {
|
||||
promotable.push({
|
||||
col,
|
||||
prefix: parts[0].trim(),
|
||||
isFullCell: false,
|
||||
});
|
||||
}
|
||||
}
|
||||
if (promotable.length < 2) continue;
|
||||
if (promotable.some(p => p.isFullCell) && result[r][0].trim().length > 0) continue;
|
||||
for (const { col, prefix, isFullCell } of promotable) {
|
||||
result[0][col] = result[0][col].trim() ? `${result[0][col]} ${prefix}` : prefix;
|
||||
if (isFullCell) {
|
||||
result[r][col] = "";
|
||||
} else {
|
||||
const parts = result[r][col].split("<br>");
|
||||
result[r][col] = parts.slice(1).join("<br>");
|
||||
}
|
||||
}
|
||||
if (result[r].every(cell => cell.trim().length === 0)) {
|
||||
rowsToRemove.add(r);
|
||||
}
|
||||
}
|
||||
return result.filter((_, r) => !rowsToRemove.has(r));
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Free text rendering
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Y tolerance for grouping text boxes onto the same visual line. */
|
||||
const TEXT_LINE_Y_TOLERANCE = 3;
|
||||
/** Minimum X gap between adjacent boxes to mark line as tabular. */
|
||||
const TABULAR_X_GAP = 30;
|
||||
/**
|
||||
* Minimum font size (pts) to consider when computing the modal body font.
|
||||
* Tiny labels from diagrams, footnote markers, and superscripts are excluded
|
||||
* so they don't skew the modal toward small sizes.
|
||||
*/
|
||||
const MIN_BODY_FONT_SIZE = 7;
|
||||
|
||||
/**
|
||||
* Compute the most frequent font size among text boxes, ignoring very small
|
||||
* text that likely comes from diagrams, footnotes, or superscripts.
|
||||
*/
|
||||
function modalFontSize(textBoxes: TextBox[]): number {
|
||||
const counts = new Map<number, number>();
|
||||
for (const tb of textBoxes) {
|
||||
const size = Math.round((tb.fontSize ?? 0) * 10) / 10;
|
||||
if (size < MIN_BODY_FONT_SIZE) continue;
|
||||
counts.set(size, (counts.get(size) ?? 0) + 1);
|
||||
}
|
||||
let modal = 0;
|
||||
let maxCount = 0;
|
||||
for (const [size, count] of counts) {
|
||||
if (count > maxCount) {
|
||||
maxCount = count;
|
||||
modal = size;
|
||||
}
|
||||
}
|
||||
return modal;
|
||||
}
|
||||
|
||||
/** Group free text boxes into horizontal lines, sorted top-to-bottom. */
|
||||
function groupFreeTextIntoLines(textBoxes: TextBox[]): RenderLine[] {
|
||||
if (textBoxes.length === 0) return [];
|
||||
const sorted = [...textBoxes].sort((a, b) => {
|
||||
const ya = (a.bounds.top + a.bounds.bottom) / 2;
|
||||
const yb = (b.bounds.top + b.bounds.bottom) / 2;
|
||||
const dy = yb - ya;
|
||||
if (Math.abs(dy) > TEXT_LINE_Y_TOLERANCE) return dy;
|
||||
return a.bounds.left - b.bounds.left;
|
||||
});
|
||||
const lines: RenderLine[] = [];
|
||||
let curParts = [sorted[0].text];
|
||||
let curBoxes = [sorted[0]];
|
||||
let curY = (sorted[0].bounds.top + sorted[0].bounds.bottom) / 2;
|
||||
let curTopY = curY;
|
||||
let curFontSize = sorted[0].fontSize;
|
||||
let curIsBold = sorted[0].isBold;
|
||||
const finishLine = () => {
|
||||
let isTabular = false;
|
||||
for (let j = 1; j < curBoxes.length; j++) {
|
||||
if (curBoxes[j].bounds.left - curBoxes[j - 1].bounds.right > TABULAR_X_GAP) {
|
||||
isTabular = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
lines.push({
|
||||
text: curParts.join(" "),
|
||||
topY: curTopY,
|
||||
fontSize: curFontSize,
|
||||
isBold: curIsBold,
|
||||
isTabular,
|
||||
});
|
||||
};
|
||||
for (let i = 1; i < sorted.length; i++) {
|
||||
const box = sorted[i];
|
||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
||||
if (Math.abs(cy - curY) <= TEXT_LINE_Y_TOLERANCE) {
|
||||
curParts.push(box.text);
|
||||
curBoxes.push(box);
|
||||
curFontSize = Math.max(curFontSize, box.fontSize);
|
||||
curIsBold = curIsBold || box.isBold;
|
||||
} else {
|
||||
finishLine();
|
||||
curParts = [box.text];
|
||||
curBoxes = [box];
|
||||
curY = cy;
|
||||
curTopY = cy;
|
||||
curFontSize = box.fontSize;
|
||||
curIsBold = box.isBold;
|
||||
}
|
||||
}
|
||||
finishLine();
|
||||
return lines;
|
||||
}
|
||||
|
||||
/** Determine markdown heading prefix based on font size relative to body. */
|
||||
function headingPrefix(fontSize: number, bodyFontSize: number, isBold: boolean): string {
|
||||
if (bodyFontSize <= 0) return "";
|
||||
const ratio = fontSize / bodyFontSize;
|
||||
// Large headings (>2x body size)
|
||||
if (ratio >= 2.0) return "# ";
|
||||
// Medium headings (~1.5x body size)
|
||||
if (ratio >= 1.4) return "## ";
|
||||
// Small headings (bold and slightly larger)
|
||||
if (ratio >= 1.1 && isBold) return "### ";
|
||||
return "";
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Block merging
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Merge consecutive blocks with the same heading prefix (wrapped headings). */
|
||||
function mergeConsecutiveHeadings(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
|
||||
if (blocks.length === 0) return [];
|
||||
const HEADING_RE = /^(#{1,6} )/;
|
||||
const maxGap = Math.max(bodyFS * 3, 30);
|
||||
const merged: ContentBlock[] = [];
|
||||
let cur: ContentBlock = { ...blocks[0] };
|
||||
for (let i = 1; i < blocks.length; i++) {
|
||||
const next = blocks[i];
|
||||
const curMatch = cur.content.match(HEADING_RE);
|
||||
const nextMatch = next.content.match(HEADING_RE);
|
||||
const gap = cur.topY - next.topY;
|
||||
if (curMatch && nextMatch && curMatch[1] === nextMatch[1] && gap <= maxGap) {
|
||||
cur = {
|
||||
topY: cur.topY,
|
||||
content: `${cur.content} ${next.content.slice(nextMatch[1].length)}`,
|
||||
isTabular: cur.isTabular || next.isTabular,
|
||||
};
|
||||
} else {
|
||||
merged.push(cur);
|
||||
cur = { ...next };
|
||||
}
|
||||
}
|
||||
merged.push(cur);
|
||||
return merged;
|
||||
}
|
||||
|
||||
/**
|
||||
* Merge consecutive plain-text blocks that are wrapped lines of the same paragraph.
|
||||
*/
|
||||
function mergeParagraphWraps(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
|
||||
if (blocks.length === 0 || bodyFS <= 0) return blocks;
|
||||
const HEADING_RE = /^#{1,6} /;
|
||||
const SENTENCE_END_RE = /[.!?…)\]]\s*$/;
|
||||
const maxGap = bodyFS * 2.0;
|
||||
const MIN_WRAP_LENGTH = 25;
|
||||
const merged: ContentBlock[] = [];
|
||||
let cur: WrapBlock = { ...blocks[0], lastTopY: blocks[0].topY };
|
||||
for (let i = 1; i < blocks.length; i++) {
|
||||
const next = blocks[i];
|
||||
const curIsBody = !HEADING_RE.test(cur.content) && !cur.content.startsWith("|");
|
||||
const nextIsBody = !HEADING_RE.test(next.content) && !next.content.startsWith("|");
|
||||
const gap = cur.lastTopY - next.topY;
|
||||
const isWrap =
|
||||
curIsBody &&
|
||||
nextIsBody &&
|
||||
!cur.isTabular &&
|
||||
!next.isTabular &&
|
||||
gap > 0 &&
|
||||
gap <= maxGap &&
|
||||
cur.content.length > MIN_WRAP_LENGTH &&
|
||||
!SENTENCE_END_RE.test(cur.content);
|
||||
if (isWrap) {
|
||||
cur = {
|
||||
topY: cur.topY,
|
||||
lastTopY: next.topY,
|
||||
content: `${cur.content.trimEnd()} ${next.content.trimStart()}`,
|
||||
isTabular: false,
|
||||
};
|
||||
} else {
|
||||
merged.push({ topY: cur.topY, content: cur.content });
|
||||
cur = { ...next, lastTopY: next.topY };
|
||||
}
|
||||
}
|
||||
merged.push({ topY: cur.topY, content: cur.content });
|
||||
return merged;
|
||||
}
|
||||
|
||||
/** Remove page number blocks near the bottom of the page. */
|
||||
function removePageNumbers(blocks: ContentBlock[]): ContentBlock[] {
|
||||
const PAGE_NUM_RE = /^(?:#{1,6}\s*)?\d+\s*$/;
|
||||
const BOTTOM_Y = 120;
|
||||
return blocks.filter((block, idx) => {
|
||||
const isBottom = idx >= blocks.length - 3;
|
||||
const isLowY = block.topY <= BOTTOM_Y;
|
||||
const isPageNum = PAGE_NUM_RE.test(block.content.trim());
|
||||
return !(isBottom && isLowY && isPageNum);
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Detached first-column table reconstruction
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Fix tables where the first column was emitted as free text blocks
|
||||
* around a markdown table containing only the right-side columns.
|
||||
*
|
||||
* Detects: a plain-text header line with (N+1) tokens above an N-column
|
||||
* markdown table, plus short label lines whose count matches the table's
|
||||
* logical row count. Reconstructs into a proper (N+1)-column table.
|
||||
*/
|
||||
function normalizeDetachedFirstColumnTables(blocks: ContentBlock[]): ContentBlock[] {
|
||||
const HEADING_RE = /^#{1,6}\s/;
|
||||
const isTableBlock = (text: string) => text.trimStart().startsWith("|");
|
||||
const isPlainBlock = (text: string) => !HEADING_RE.test(text) && !isTableBlock(text);
|
||||
const isShortLabel = (text: string) => {
|
||||
const t = text.trim();
|
||||
return t.length > 0 && t.length <= 40;
|
||||
};
|
||||
const splitTokens = (text: string) =>
|
||||
text
|
||||
.trim()
|
||||
.split(/[ \t]+/)
|
||||
.filter(Boolean);
|
||||
const replacements = new Map<number, string>();
|
||||
const remove = new Set<number>();
|
||||
for (let tableIdx = 0; tableIdx < blocks.length; tableIdx++) {
|
||||
if (remove.has(tableIdx)) continue;
|
||||
const tableBlock = blocks[tableIdx];
|
||||
if (!isTableBlock(tableBlock.content)) continue;
|
||||
const tableLines = tableBlock.content
|
||||
.split("\n")
|
||||
.map(line => line.trim())
|
||||
.filter(line => line.startsWith("|"));
|
||||
const dataRows = tableLines
|
||||
.filter(line => !/^\|\s*[-: ]+\|/.test(line))
|
||||
.map(parsePipeRow)
|
||||
.filter(row => row.length > 0);
|
||||
if (dataRows.length === 0) continue;
|
||||
const cols = dataRows[0].length;
|
||||
if (cols < 2 || dataRows.some(row => row.length !== cols)) continue;
|
||||
// Expand by <br> count to get logical row count
|
||||
const logicalRows: string[][] = [];
|
||||
for (const row of dataRows) {
|
||||
const splitCells = row.map(cell => cell.split("<br>").map(p => p.trim()));
|
||||
const rowSpan = Math.max(...splitCells.map(parts => parts.length));
|
||||
for (let k = 0; k < rowSpan; k++) {
|
||||
logicalRows.push(splitCells.map(parts => parts[k] ?? ""));
|
||||
}
|
||||
}
|
||||
if (logicalRows.length < 2) continue;
|
||||
// Find header with (cols + 1) non-numeric tokens
|
||||
let headerIdx = -1;
|
||||
let headerTokens: string[] = [];
|
||||
for (let i = Math.max(0, tableIdx - 4); i <= tableIdx - 1; i++) {
|
||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
||||
if (!isPlainBlock(text)) continue;
|
||||
const tokens = splitTokens(text);
|
||||
if (tokens.length === cols + 1 && tokens.every(tok => !/[0-9]/.test(tok))) {
|
||||
headerIdx = i;
|
||||
headerTokens = tokens;
|
||||
}
|
||||
}
|
||||
if (headerIdx < 0) continue;
|
||||
// Collect short label lines above/below table
|
||||
const aboveLabels: Array<{ idx: number; text: string }> = [];
|
||||
for (let i = tableIdx - 1; i > headerIdx; i--) {
|
||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
||||
if (!isPlainBlock(text) || !isShortLabel(text)) break;
|
||||
aboveLabels.push({ idx: i, text });
|
||||
}
|
||||
aboveLabels.reverse();
|
||||
const belowLabels: Array<{ idx: number; text: string }> = [];
|
||||
for (let i = tableIdx + 1; i < blocks.length; i++) {
|
||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
||||
if (!isPlainBlock(text) || !isShortLabel(text)) break;
|
||||
belowLabels.push({ idx: i, text });
|
||||
}
|
||||
const labels = [...aboveLabels, ...belowLabels];
|
||||
if (labels.length !== logicalRows.length) continue;
|
||||
// Reconstruct the full table
|
||||
const normalizedLines: string[] = [];
|
||||
normalizedLines.push(`| ${headerTokens.join(" | ")} |`);
|
||||
normalizedLines.push(`| ${Array.from({ length: cols + 1 }, () => "---").join(" | ")} |`);
|
||||
for (let r = 0; r < logicalRows.length; r++) {
|
||||
normalizedLines.push(`| ${labels[r].text} | ${logicalRows[r].join(" | ")} |`);
|
||||
}
|
||||
replacements.set(tableIdx, normalizedLines.join("\n"));
|
||||
remove.add(headerIdx);
|
||||
for (const label of labels) remove.add(label.idx);
|
||||
}
|
||||
if (replacements.size === 0 && remove.size === 0) return blocks;
|
||||
const out: ContentBlock[] = [];
|
||||
for (let i = 0; i < blocks.length; i++) {
|
||||
if (remove.has(i)) continue;
|
||||
const replaced = replacements.get(i);
|
||||
if (replaced) {
|
||||
out.push({ topY: blocks[i].topY, content: replaced });
|
||||
} else {
|
||||
out.push(blocks[i]);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Render one page's content: free text and tables interleaved top-to-bottom.
|
||||
*/
|
||||
export function renderPageContent(
|
||||
freeTextBoxes: TextBox[],
|
||||
tables: TableGrid[],
|
||||
imageBlocks: Array<{ topY: number; markdown: string }> = [],
|
||||
allTextBoxes?: TextBox[],
|
||||
): string {
|
||||
const blocks: ContentBlock[] = [];
|
||||
// Use ALL text boxes (before table/diagram filtering) for modal font size,
|
||||
// so that diagram labels released as free text don't skew the body size.
|
||||
const bodyFS = modalFontSize(allTextBoxes ?? freeTextBoxes);
|
||||
// Free text lines
|
||||
for (const line of groupFreeTextIntoLines(freeTextBoxes)) {
|
||||
const prefix = headingPrefix(line.fontSize, bodyFS, line.isBold);
|
||||
blocks.push({
|
||||
topY: line.topY,
|
||||
content: prefix + line.text,
|
||||
isTabular: prefix === "" && line.isTabular,
|
||||
});
|
||||
}
|
||||
// Tables
|
||||
for (const table of tables) {
|
||||
const md = renderTableToMarkdown(table);
|
||||
if (md.length > 0) {
|
||||
blocks.push({ topY: table.topY, content: md });
|
||||
}
|
||||
}
|
||||
// Images
|
||||
for (const img of imageBlocks) {
|
||||
blocks.push({ topY: img.topY, content: img.markdown });
|
||||
}
|
||||
// Sort top-to-bottom (higher Y = higher on page = comes first)
|
||||
blocks.sort((a, b) => b.topY - a.topY);
|
||||
const cleaned = removePageNumbers(blocks);
|
||||
const headingsMerged = mergeConsecutiveHeadings(cleaned, bodyFS);
|
||||
const merged = mergeParagraphWraps(headingsMerged, bodyFS);
|
||||
const normalized = normalizeDetachedFirstColumnTables(merged);
|
||||
return normalized
|
||||
.map(b => b.content)
|
||||
.join("\n\n")
|
||||
.trim();
|
||||
}
|
||||
@@ -1,84 +0,0 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/** Bounding box in PDF coordinate space (origin = bottom-left). */
|
||||
export type Bounds = {
|
||||
left: number;
|
||||
right: number;
|
||||
/** Higher value = higher on the page. */
|
||||
top: number;
|
||||
bottom: number;
|
||||
};
|
||||
|
||||
/** A text fragment with position and font metadata. */
|
||||
export type TextBox = {
|
||||
id: string;
|
||||
text: string;
|
||||
bounds: Bounds;
|
||||
pageNumber: number;
|
||||
/** Dominant font size in points. */
|
||||
fontSize: number;
|
||||
/** True if rendered bold (font name or rendering mode). */
|
||||
isBold: boolean;
|
||||
};
|
||||
|
||||
/** A horizontal or vertical line segment extracted from vector graphics. */
|
||||
export type Segment = {
|
||||
id: string;
|
||||
x1: number;
|
||||
y1: number;
|
||||
x2: number;
|
||||
y2: number;
|
||||
};
|
||||
|
||||
/** A single cell in a resolved table grid. */
|
||||
export type TableCell = {
|
||||
row: number;
|
||||
col: number;
|
||||
text: string;
|
||||
rowSpan: number;
|
||||
colSpan: number;
|
||||
};
|
||||
|
||||
/** A resolved table grid ready for markdown rendering. */
|
||||
export type TableGrid = {
|
||||
pageNumber: number;
|
||||
rows: number;
|
||||
cols: number;
|
||||
cells: TableCell[];
|
||||
warnings: string[];
|
||||
/** Top Y coordinate (PDF space: larger = higher on page). */
|
||||
topY: number;
|
||||
/** True for tables detected without vector borders. */
|
||||
isBorderless: boolean;
|
||||
};
|
||||
|
||||
/** An image/diagram region detected on a page. */
|
||||
export type ImageRegion = {
|
||||
id: string;
|
||||
pageNumber: number;
|
||||
/** Bounding box in mupdf coordinates (top-left origin). */
|
||||
bbox: {
|
||||
x: number;
|
||||
y: number;
|
||||
w: number;
|
||||
h: number;
|
||||
};
|
||||
/** Y position in PDF coordinates (bottom-left) for ordering. */
|
||||
topY: number;
|
||||
};
|
||||
|
||||
/** Result of extracting content from a single PDF page. */
|
||||
export type PageContent = {
|
||||
pageNumber: number;
|
||||
textBoxes: TextBox[];
|
||||
segments: Segment[];
|
||||
images: ImageRegion[];
|
||||
};
|
||||
|
||||
/** A block of rendered content (text paragraph or table). */
|
||||
export type ContentBlock = {
|
||||
topY: number;
|
||||
content: string;
|
||||
/** True if this line has wide gaps between text boxes (column headers). */
|
||||
isTabular?: boolean;
|
||||
};
|
||||
@@ -1,250 +0,0 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||
import { isEexist, isEnotempty, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import type { ToolSession } from "../sdk";
|
||||
import { loadImageInput, MAX_IMAGE_INPUT_BYTES, webpExclusionForModel } from "../utils/image-loading";
|
||||
import { convertFileWithMarkit } from "../utils/markit";
|
||||
import type { ReadToolDetails } from "./read";
|
||||
import { prependSuffixResolutionNotice } from "./read-format";
|
||||
import { isNotFoundError } from "./read-path-resolution";
|
||||
import { formatBytes } from "./render-utils";
|
||||
import { ToolError } from "./tool-errors";
|
||||
import { toolResult } from "./tool-result";
|
||||
|
||||
const MAX_IMAGE_SIZE = MAX_IMAGE_INPUT_BYTES;
|
||||
|
||||
const PDF_IMAGE_PLACEHOLDER_RE = /<!--\s*image:\s*([^\s<>]+)(.*?)-->/g;
|
||||
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
|
||||
const PDF_IMAGE_MEMBER_EXTENSION_RE = /\.png$/i;
|
||||
const PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH = 96;
|
||||
|
||||
interface PdfImageSnapshot {
|
||||
directory: string;
|
||||
filePath: string;
|
||||
digest: string;
|
||||
}
|
||||
|
||||
interface PdfImageExtraction {
|
||||
controller: AbortController;
|
||||
promise: Promise<string>;
|
||||
settled: boolean;
|
||||
waiters: number;
|
||||
}
|
||||
|
||||
const pdfImageExtractions = new Map<string, PdfImageExtraction>();
|
||||
|
||||
function pdfImageMemberPath(pdfPath: string, imageId: string): string {
|
||||
const member = PDF_IMAGE_MEMBER_EXTENSION_RE.test(imageId) ? imageId : `${imageId}.png`;
|
||||
return `${pdfPath}:${member}`;
|
||||
}
|
||||
|
||||
export function rewritePdfImagePlaceholders(markdown: string, pdfPath: string): string {
|
||||
return markdown.replace(PDF_IMAGE_PLACEHOLDER_RE, (_match: string, imageId: string, metadataText: string) => {
|
||||
const metadata = metadataText.trim();
|
||||
const suffix = metadata.length > 0 ? ` (${metadata})` : "";
|
||||
return `Image ${imageId}${suffix}: read \`${pdfImageMemberPath(pdfPath, imageId)}\``;
|
||||
});
|
||||
}
|
||||
|
||||
export function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; member: string } | null {
|
||||
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
|
||||
if (!match) return null;
|
||||
const pdfPath = match[1];
|
||||
const member = match[2];
|
||||
if (pdfPath === undefined || member === undefined) return null;
|
||||
if (member.length !== 0 && !PDF_IMAGE_MEMBER_EXTENSION_RE.test(member)) return null;
|
||||
return { pdfPath, member };
|
||||
}
|
||||
function pdfImageCacheDir(session: ToolSession, absolutePdfPath: string, contentDigest: string): string {
|
||||
const artifactsDir = session.getArtifactsDir?.();
|
||||
let root = artifactsDir ?? undefined;
|
||||
if (root === undefined) {
|
||||
const sessionFile = session.getSessionFile();
|
||||
root = sessionFile?.endsWith(".jsonl") ? sessionFile.slice(0, -6) : path.join(os.tmpdir(), "omp-read-pdf-images");
|
||||
}
|
||||
const basename = path
|
||||
.basename(absolutePdfPath)
|
||||
.replace(/[^A-Za-z0-9._-]/g, "_")
|
||||
.slice(0, PDF_IMAGE_CACHE_BASENAME_MAX_LENGTH);
|
||||
const pathDigest = Bun.hash(absolutePdfPath).toString(36);
|
||||
return path.join(root, "read-pdf-images", `${basename}-${pathDigest}-${contentDigest}`);
|
||||
}
|
||||
|
||||
async function snapshotPdfSource(absolutePdfPath: string, signal?: AbortSignal): Promise<PdfImageSnapshot> {
|
||||
const directory = await fs.mkdtemp(path.join(os.tmpdir(), "omp-read-pdf-"));
|
||||
try {
|
||||
const bytes = await untilAborted(signal, () => Bun.file(absolutePdfPath).bytes());
|
||||
signal?.throwIfAborted();
|
||||
const digest = new Bun.CryptoHasher("sha256").update(bytes).digest("hex");
|
||||
const filePath = path.join(directory, "source.pdf");
|
||||
await Bun.write(filePath, bytes);
|
||||
signal?.throwIfAborted();
|
||||
return { directory, filePath, digest };
|
||||
} catch (error) {
|
||||
await fs.rm(directory, { recursive: true, force: true });
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
async function listPdfImageMembers(imageDir: string): Promise<string[]> {
|
||||
try {
|
||||
const entries = await fs.readdir(imageDir, { withFileTypes: true });
|
||||
const members: string[] = [];
|
||||
for (const entry of entries) {
|
||||
if (entry.isFile() && PDF_IMAGE_MEMBER_EXTENSION_RE.test(entry.name)) members.push(entry.name);
|
||||
}
|
||||
return members.sort();
|
||||
} catch (error) {
|
||||
if (isNotFoundError(error)) return [];
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
async function extractPdfImages(snapshot: PdfImageSnapshot, imageDir: string, signal: AbortSignal): Promise<string> {
|
||||
const markerPath = path.join(imageDir, ".extracted");
|
||||
try {
|
||||
await fs.stat(markerPath);
|
||||
return imageDir;
|
||||
} catch (error) {
|
||||
if (!isNotFoundError(error)) throw error;
|
||||
}
|
||||
|
||||
await fs.mkdir(path.dirname(imageDir), { recursive: true });
|
||||
const stagingDir = await fs.mkdtemp(`${imageDir}.tmp-`);
|
||||
let published = false;
|
||||
try {
|
||||
const result = await convertFileWithMarkit(snapshot.filePath, signal, { imageDir: stagingDir });
|
||||
if (!result.ok) {
|
||||
throw new ToolError(`Cannot extract images from PDF: ${result.error ?? "conversion failed"}`);
|
||||
}
|
||||
await Bun.write(path.join(stagingDir, ".extracted"), "ok");
|
||||
try {
|
||||
await fs.rename(stagingDir, imageDir);
|
||||
published = true;
|
||||
} catch (error) {
|
||||
if (!isEexist(error) && !isEnotempty(error)) throw error;
|
||||
try {
|
||||
await fs.stat(markerPath);
|
||||
} catch (markerError) {
|
||||
if (isNotFoundError(markerError)) throw error;
|
||||
throw markerError;
|
||||
}
|
||||
}
|
||||
return imageDir;
|
||||
} finally {
|
||||
if (!published) await fs.rm(stagingDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
function createPdfImageExtraction(snapshot: PdfImageSnapshot, imageDir: string): PdfImageExtraction {
|
||||
const controller = new AbortController();
|
||||
const promise = extractPdfImages(snapshot, imageDir, controller.signal).finally(() =>
|
||||
fs.rm(snapshot.directory, { recursive: true, force: true }),
|
||||
);
|
||||
const extraction: PdfImageExtraction = { controller, promise, settled: false, waiters: 0 };
|
||||
const settle = () => {
|
||||
extraction.settled = true;
|
||||
if (pdfImageExtractions.get(imageDir) === extraction) pdfImageExtractions.delete(imageDir);
|
||||
};
|
||||
void promise.then(settle, settle);
|
||||
return extraction;
|
||||
}
|
||||
|
||||
async function waitForPdfImageExtraction(
|
||||
extraction: PdfImageExtraction,
|
||||
signal: AbortSignal | undefined,
|
||||
): Promise<string> {
|
||||
extraction.waiters++;
|
||||
try {
|
||||
return await untilAborted(signal, extraction.promise);
|
||||
} finally {
|
||||
extraction.waiters--;
|
||||
if (extraction.waiters === 0 && !extraction.settled) {
|
||||
extraction.controller.abort();
|
||||
try {
|
||||
await extraction.promise;
|
||||
} catch {}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function ensurePdfImageCache(
|
||||
session: ToolSession,
|
||||
absolutePdfPath: string,
|
||||
signal?: AbortSignal,
|
||||
): Promise<string> {
|
||||
const snapshot = await snapshotPdfSource(absolutePdfPath, signal);
|
||||
const imageDir = pdfImageCacheDir(session, absolutePdfPath, snapshot.digest);
|
||||
const existing = pdfImageExtractions.get(imageDir);
|
||||
if (existing && !existing.settled && !existing.controller.signal.aborted) {
|
||||
await fs.rm(snapshot.directory, { recursive: true, force: true });
|
||||
return waitForPdfImageExtraction(existing, signal);
|
||||
}
|
||||
|
||||
const extraction = createPdfImageExtraction(snapshot, imageDir);
|
||||
pdfImageExtractions.set(imageDir, extraction);
|
||||
return waitForPdfImageExtraction(extraction, signal);
|
||||
}
|
||||
|
||||
export async function readPdfImageMember(
|
||||
session: ToolSession,
|
||||
autoResizeImages: boolean,
|
||||
absolutePdfPath: string,
|
||||
pdfDisplayPath: string,
|
||||
member: string,
|
||||
suffixResolution: { from: string; to: string } | undefined,
|
||||
signal?: AbortSignal,
|
||||
): Promise<AgentToolResult<ReadToolDetails>> {
|
||||
const imageDir = await ensurePdfImageCache(session, absolutePdfPath, signal);
|
||||
const members = await listPdfImageMembers(imageDir);
|
||||
if (member.length === 0) {
|
||||
const text =
|
||||
members.length === 0
|
||||
? "No extractable PDF image members found."
|
||||
: `Extractable PDF image members:\n${members
|
||||
.map(imageMember => `- read \`${pdfDisplayPath}:${imageMember}\``)
|
||||
.join("\n")}`;
|
||||
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
|
||||
.text(prependSuffixResolutionNotice(text, suffixResolution))
|
||||
.sourcePath(absolutePdfPath)
|
||||
.done();
|
||||
}
|
||||
|
||||
if (!members.includes(member)) {
|
||||
const available = members.length === 0 ? "(none)" : members.join(", ");
|
||||
throw new ToolError(`PDF image member '${member}' not found. Available members: ${available}`);
|
||||
}
|
||||
|
||||
const imagePath = path.join(imageDir, member);
|
||||
const imageStat = await Bun.file(imagePath).stat();
|
||||
if (imageStat.size > MAX_IMAGE_SIZE) {
|
||||
const sizeStr = formatBytes(imageStat.size);
|
||||
const maxStr = formatBytes(MAX_IMAGE_SIZE);
|
||||
throw new ToolError(`Image file too large: ${sizeStr} exceeds ${maxStr} limit.`);
|
||||
}
|
||||
const metadata = await readImageMetadata(imagePath);
|
||||
const mimeType = metadata?.mimeType;
|
||||
if (!mimeType) throw new ToolError(`PDF image member '${member}' is not a supported image.`);
|
||||
const imageInput = await loadImageInput({
|
||||
path: `${pdfDisplayPath}:${member}`,
|
||||
cwd: session.cwd,
|
||||
autoResize: autoResizeImages,
|
||||
maxBytes: MAX_IMAGE_SIZE,
|
||||
resolvedPath: imagePath,
|
||||
detectedMimeType: mimeType,
|
||||
excludeWebP: webpExclusionForModel(session.getActiveModel?.()),
|
||||
});
|
||||
if (!imageInput) {
|
||||
throw new ToolError(`Read image file [${mimeType}] failed: unsupported image format.`);
|
||||
}
|
||||
const textNote = prependSuffixResolutionNotice(imageInput.textNote, suffixResolution);
|
||||
return toolResult<ReadToolDetails>({ resolvedPath: absolutePdfPath, suffixResolution })
|
||||
.content([
|
||||
{ type: "text", text: textNote },
|
||||
{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType },
|
||||
])
|
||||
.sourcePath(imageInput.resolvedPath)
|
||||
.done();
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
const PDF_IMAGE_MEMBER_RE = /^(.*\.pdf):(.*)$/i;
|
||||
|
||||
/** Parse a former PDF image-member read without claiming normal selectors. */
|
||||
export function splitUnsupportedPdfImageReadPath(readPath: string): { pdfPath: string } | null {
|
||||
const match = PDF_IMAGE_MEMBER_RE.exec(readPath);
|
||||
const pdfPath = match?.[1];
|
||||
return pdfPath ? { pdfPath } : null;
|
||||
}
|
||||
|
||||
/** Explain how to render a PDF now that the text backend has no rasterizer. */
|
||||
export function pdfImageRenderingUnsupportedMessage(pdfPath: string): string {
|
||||
return `pdf-inspector cannot render PDF images. Use the Puppeteer browser tool to render '${pdfPath}', or read '${pdfPath}' for extracted text.`;
|
||||
}
|
||||
@@ -100,7 +100,7 @@ import {
|
||||
isRemoteMountPath,
|
||||
type SuffixMatchCache,
|
||||
} from "./read-path-resolution";
|
||||
import { readPdfImageMember, rewritePdfImagePlaceholders, splitPdfImageMemberReadPath } from "./read-pdf-images";
|
||||
import { pdfImageRenderingUnsupportedMessage, splitUnsupportedPdfImageReadPath } from "./read-pdf";
|
||||
import { isMultiRange, isRawSelector, type ParsedSelector, parseSel, selToOffsetLimit } from "./read-selector";
|
||||
import { readSqlite, resolveSqliteReadPath } from "./read-sqlite";
|
||||
import { isProseSummaryPath, renderSummary, routeReadThroughBridge, trySummarize } from "./read-summary";
|
||||
@@ -892,7 +892,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
|
||||
// Prefer a literal filesystem match over selector interpretation so real
|
||||
// POSIX filenames containing selector-looking suffixes win over structured
|
||||
// archive / sqlite / pdf-image dispatch. A selector promoted from local://
|
||||
// archive / sqlite / unsupported PDF-image dispatch. A selector promoted from local://
|
||||
// remains separate so it cannot be mistaken for part of the resolved path.
|
||||
const literalSplit =
|
||||
promotedSelector === undefined
|
||||
@@ -925,35 +925,10 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
return readSqlite(sqlitePath, signal);
|
||||
}
|
||||
|
||||
const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath);
|
||||
if (pdfImageMemberPath) {
|
||||
let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd);
|
||||
let suffixResolution: { from: string; to: string } | undefined;
|
||||
try {
|
||||
const stat = await Bun.file(absolutePdfPath).stat();
|
||||
if (stat.isDirectory())
|
||||
throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`);
|
||||
} catch (error) {
|
||||
if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error;
|
||||
const suffixMatch = await findSuffixMatchCached(
|
||||
this.session,
|
||||
suffixCache,
|
||||
pdfImageMemberPath.pdfPath,
|
||||
signal,
|
||||
);
|
||||
if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`);
|
||||
absolutePdfPath = suffixMatch.absolutePath;
|
||||
suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath };
|
||||
}
|
||||
return readPdfImageMember(
|
||||
this.session,
|
||||
this.#autoResizeImages,
|
||||
absolutePdfPath,
|
||||
pdfImageMemberPath.pdfPath,
|
||||
pdfImageMemberPath.member,
|
||||
suffixResolution,
|
||||
signal,
|
||||
);
|
||||
const unsupportedPdfImageRead =
|
||||
literalSplit.sel === undefined ? splitUnsupportedPdfImageReadPath(readPath) : null;
|
||||
if (unsupportedPdfImageRead && (await probeLiteralPathExists(readPath, this.session.cwd)) === "missing") {
|
||||
throw new ToolError(pdfImageRenderingUnsupportedMessage(unsupportedPdfImageRead.pdfPath));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1102,8 +1077,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
// Convert document via markit.
|
||||
const result = await convertFileWithMarkit(absolutePath, signal);
|
||||
if (result.ok) {
|
||||
const renderedContent =
|
||||
ext === ".pdf" ? rewritePdfImagePlaceholders(result.content, resolvedDisplayPath) : result.content;
|
||||
const renderedContent = result.content;
|
||||
// Route the converted markdown through the in-memory text builder
|
||||
// so line-range selectors (`file.pdf:50-100`, `:5-16,40-80`) and
|
||||
// raw mode apply against the converted output. Without this,
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import * as path from "node:path";
|
||||
import { logger, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import { untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import type { ConversionResult, Markit, StreamInfo } from "../markit";
|
||||
import { ToolAbortError } from "../tools/tool-errors";
|
||||
import {
|
||||
@@ -8,7 +8,6 @@ import {
|
||||
readMarkitConversionCache,
|
||||
writeMarkitConversionCache,
|
||||
} from "./markit-cache";
|
||||
import { loadEmbeddedMupdfWasm } from "./mupdf-wasm-embed";
|
||||
|
||||
/**
|
||||
* File extensions markit can actually convert to markdown — one per registered
|
||||
@@ -31,53 +30,16 @@ export interface MarkitConversionResult {
|
||||
|
||||
export interface MarkitFileConversionOptions {
|
||||
/**
|
||||
* Directory the PDF converter writes extracted images/diagrams into. When
|
||||
* set, each embedded image is rendered to `<id>.png` and referenced by path
|
||||
* in the markdown; when unset, markit emits an `<!-- image: <id> ... -->`
|
||||
* placeholder comment instead.
|
||||
* Directory converters may use for extracted image or diagram files. Since
|
||||
* those files are conversion side effects, conversions using this option
|
||||
* bypass the markdown cache.
|
||||
*/
|
||||
imageDir?: string;
|
||||
}
|
||||
|
||||
interface MuPdfWasmModuleConfig {
|
||||
print?: (...values: unknown[]) => void;
|
||||
printErr?: (...values: unknown[]) => void;
|
||||
wasmBinary?: Uint8Array;
|
||||
}
|
||||
|
||||
function logMuPdfWasmOutput(stream: "stdout" | "stderr", values: unknown[]): void {
|
||||
const message = values.length === 1 && typeof values[0] === "string" ? values[0] : values.map(String).join(" ");
|
||||
logger.debug("mupdf wasm output", { stream, message });
|
||||
}
|
||||
|
||||
// `$libmupdf_wasm_Module` is declared globally (as `any`) by the mupdf package.
|
||||
// Install print hooks before the WASM module initializes so its stdout/stderr
|
||||
// route to the file logger instead of corrupting the TUI.
|
||||
function installMuPdfWasmLogger(): void {
|
||||
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
|
||||
moduleConfig.print = (...values: unknown[]) => logMuPdfWasmOutput("stdout", values);
|
||||
moduleConfig.printErr = (...values: unknown[]) => logMuPdfWasmOutput("stderr", values);
|
||||
globalThis.$libmupdf_wasm_Module = moduleConfig;
|
||||
}
|
||||
|
||||
// Hand the WASM module its bytes directly when the compiled binary embedded them
|
||||
// (scripts/embed-mupdf-wasm.ts); a single-file binary has no node_modules for
|
||||
// mupdf to read `mupdf-wasm.wasm` from. Source/npm builds get undefined here and
|
||||
// mupdf loads its own wasm. Must run before the mupdf module evaluates.
|
||||
function installEmbeddedMupdfWasm(): void {
|
||||
const wasmBinary = loadEmbeddedMupdfWasm();
|
||||
if (!wasmBinary) return;
|
||||
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
|
||||
moduleConfig.wasmBinary = wasmBinary;
|
||||
globalThis.$libmupdf_wasm_Module = moduleConfig;
|
||||
}
|
||||
|
||||
installMuPdfWasmLogger();
|
||||
|
||||
let markit: () => Markit | Promise<Markit> = async () => {
|
||||
// Lazy: keep the document engine (mammoth/mupdf) off the startup
|
||||
// import graph — it loads only when a document is first converted.
|
||||
installEmbeddedMupdfWasm();
|
||||
// Lazy: keep the document engine off the startup import graph — it loads
|
||||
// only when a document is first converted.
|
||||
const promise = import("../markit").then(({ Markit }) => {
|
||||
const instance = new Markit();
|
||||
markit = () => instance;
|
||||
|
||||
@@ -1,12 +0,0 @@
|
||||
// AUTOGENERATED -- managed by scripts/embed-mupdf-wasm.ts. Do not edit by hand.
|
||||
//
|
||||
// Compiled single-file binaries cannot let mupdf resolve its `mupdf-wasm.wasm`
|
||||
// sibling from the read-only bunfs, so the binary build (scripts/build-binary.ts
|
||||
// and scripts/ci-release-build-binaries.ts) regenerates this module to embed the
|
||||
// wasm bytes via `with { type: "file" }` and copies the wasm next to it. Source
|
||||
// checkouts, `bun test`, and the npm `dist/cli.js` bundle keep mupdf external and
|
||||
// load the wasm from node_modules, so this placeholder returns undefined and the
|
||||
// build resets back to it afterward.
|
||||
export function loadEmbeddedMupdfWasm(): Uint8Array | undefined {
|
||||
return undefined;
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as piNatives from "@oh-my-pi/pi-natives";
|
||||
import { PdfConverter } from "../src/markit/converters/pdf";
|
||||
|
||||
describe("PdfConverter", () => {
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("keeps accepting PDF extensions and MIME types", () => {
|
||||
const converter = new PdfConverter();
|
||||
|
||||
expect(converter.accepts({ extension: ".pdf" })).toBe(true);
|
||||
expect(converter.accepts({ mimetype: "application/pdf" })).toBe(true);
|
||||
expect(converter.accepts({ mimetype: "application/pdf; charset=binary" })).toBe(true);
|
||||
expect(converter.accepts({ mimetype: "application/x-pdf" })).toBe(true);
|
||||
expect(converter.accepts({ extension: ".txt", mimetype: "text/plain" })).toBe(false);
|
||||
});
|
||||
|
||||
it("returns a browser and OCR notice for an image-only PDF", async () => {
|
||||
vi.spyOn(piNatives, "pdfToMarkdown").mockResolvedValue({
|
||||
markdown: "",
|
||||
pageCount: 3,
|
||||
pagesNeedingOcr: [1, 3],
|
||||
hasEncodingIssues: false,
|
||||
});
|
||||
|
||||
const result = await new PdfConverter().convert(Buffer.from("image-only pdf"), { extension: ".pdf" });
|
||||
|
||||
expect(result.markdown).toBe(
|
||||
"Text extraction is incomplete for PDF pages 1, 3. Use the browser tool to render those pages or OCR them.",
|
||||
);
|
||||
expect(result.markdown.length).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
@@ -1,447 +0,0 @@
|
||||
/**
|
||||
* PDF image extraction: markit emits inert `<!-- image: <id> ... -->`
|
||||
* placeholders for embedded PDF images. The read tool rewrites those into
|
||||
* browsable `read <pdf>:<id>.png` handles, and serves the actual PNG when that
|
||||
* handle is read — extracting via markit's `imageDir` into a session-artifact
|
||||
* cache. These lock the rewrite, the member extraction, member validation, and
|
||||
* the caching contract.
|
||||
*/
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
|
||||
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
|
||||
import * as piUtils from "@oh-my-pi/pi-utils";
|
||||
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
|
||||
|
||||
// 1x1 transparent PNG — small enough to pass through image loading untouched.
|
||||
const TINY_PNG = Buffer.from(
|
||||
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg==",
|
||||
"base64",
|
||||
);
|
||||
|
||||
function makeSession(testDir: string): ToolSession {
|
||||
const sessionFile = path.join(testDir, "session.jsonl");
|
||||
const artifactsDir = sessionFile.slice(0, -6);
|
||||
return {
|
||||
cwd: testDir,
|
||||
hasUI: false,
|
||||
getSessionFile: () => sessionFile,
|
||||
getArtifactsDir: () => artifactsDir,
|
||||
getSessionSpawns: () => null,
|
||||
settings: Settings.isolated({ "images.autoResize": false }),
|
||||
} as unknown as ToolSession;
|
||||
}
|
||||
|
||||
/** Spy on markit so PDF "extraction" writes the given members into imageDir. */
|
||||
function mockExtraction(members: Record<string, Buffer> = { "p11-img0.png": TINY_PNG }) {
|
||||
return vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_filePath: string, _signal, options) => {
|
||||
if (options?.imageDir) {
|
||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
||||
for (const name in members) {
|
||||
fs.writeFileSync(path.join(options.imageDir, name), members[name]!);
|
||||
}
|
||||
}
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
}
|
||||
|
||||
function imageBytes(result: AgentToolResult<ReadToolDetails>): Buffer {
|
||||
const image = result.content.find(content => content.type === "image");
|
||||
if (image?.type !== "image") throw new Error("Expected an image result");
|
||||
return Buffer.from(image.data, "base64");
|
||||
}
|
||||
|
||||
function mockBlockedExtraction() {
|
||||
const entered = Promise.withResolvers<void>();
|
||||
const release = Promise.withResolvers<void>();
|
||||
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (_sourcePath, signal, options) => {
|
||||
entered.resolve();
|
||||
await release.promise;
|
||||
signal?.throwIfAborted();
|
||||
if (options?.imageDir) {
|
||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
|
||||
}
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
return { entered, release, spy };
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolves once `count` callers are attached as waiters on the shared PDF
|
||||
* extraction. Waiter attachment is the only `untilAborted` call that receives
|
||||
* a promise (source snapshots pass thunks), so counting promise arguments
|
||||
* observes it. The abort tests need this barrier: aborting a caller while it
|
||||
* is the sole waiter tears the extraction down and deadlocks against the
|
||||
* blocked conversion mock.
|
||||
*/
|
||||
function extractionWaitersAttached(count: number): Promise<void> {
|
||||
const attached = Promise.withResolvers<void>();
|
||||
const original = piUtils.untilAborted;
|
||||
let seen = 0;
|
||||
vi.spyOn(piUtils, "untilAborted").mockImplementation((signal, pr) => {
|
||||
if (typeof pr !== "function" && ++seen === count) attached.resolve();
|
||||
return original(signal, pr);
|
||||
});
|
||||
return attached.promise;
|
||||
}
|
||||
|
||||
describe("read PDF image extraction", () => {
|
||||
let testDir: string;
|
||||
let pdfPath: string;
|
||||
beforeEach(() => {
|
||||
testDir = path.join(os.tmpdir(), `read-pdf-img-${Snowflake.next()}`);
|
||||
fs.mkdirSync(testDir, { recursive: true });
|
||||
pdfPath = path.join(testDir, "doc.pdf");
|
||||
fs.writeFileSync(pdfPath, "%PDF-stub");
|
||||
});
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
removeSyncWithRetries(testDir);
|
||||
});
|
||||
|
||||
it("rewrites image placeholders into browse handles on a full read", async () => {
|
||||
const converted = [
|
||||
"Heading",
|
||||
"",
|
||||
"<!-- image: p11-img0 (page 11, 199x124pt) -->",
|
||||
"",
|
||||
"<!-- image: p11-img1 (page 11, 199x54pt) -->",
|
||||
"",
|
||||
"Footer",
|
||||
].join("\n");
|
||||
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: converted });
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("call", { path: pdfPath });
|
||||
const text = result.content
|
||||
.filter(c => c.type === "text")
|
||||
.map(c => c.text)
|
||||
.join("\n");
|
||||
|
||||
expect(text).not.toContain("<!-- image:");
|
||||
expect(text).toContain("read `doc.pdf:p11-img0.png`");
|
||||
expect(text).toContain("read `doc.pdf:p11-img1.png`");
|
||||
// Page/size metadata is preserved in the handle text.
|
||||
expect(text).toContain("page 11, 199x124pt");
|
||||
});
|
||||
|
||||
it("rewrites placeholders inside a line-range view", async () => {
|
||||
const lines = Array.from({ length: 20 }, (_, i) => `pdf line ${i + 1}`);
|
||||
lines[9] = "<!-- image: p3-img0 (page 3, 100x50pt) -->"; // line 10
|
||||
vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({ ok: true, content: lines.join("\n") });
|
||||
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("call", { path: `${pdfPath}:8-12` });
|
||||
const text = result.content
|
||||
.filter(c => c.type === "text")
|
||||
.map(c => c.text)
|
||||
.join("\n");
|
||||
|
||||
expect(text).not.toContain("<!-- image:");
|
||||
expect(text).toContain("read `doc.pdf:p3-img0.png`");
|
||||
});
|
||||
|
||||
it("extracts a PDF image member as an inline image block", async () => {
|
||||
const spy = mockExtraction();
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
|
||||
const image = result.content.find(c => c.type === "image");
|
||||
expect(image).toBeDefined();
|
||||
expect(image && "mimeType" in image ? image.mimeType : undefined).toBe("image/png");
|
||||
const text = result.content
|
||||
.filter(c => c.type === "text")
|
||||
.map(c => c.text)
|
||||
.join("\n");
|
||||
expect(text).toContain("Read image file");
|
||||
// Extraction was driven through markit with an imageDir target.
|
||||
expect(spy).toHaveBeenCalledTimes(1);
|
||||
expect(spy.mock.calls[0]?.[2]?.imageDir).toBeTruthy();
|
||||
});
|
||||
|
||||
it("reuses the extraction cache across member reads", async () => {
|
||||
const spy = mockExtraction();
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
// Second read is served from the `.extracted` cache, not re-converted.
|
||||
expect(spy).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("re-extracts image members after same-path PDF replacement", async () => {
|
||||
const sourceA = Buffer.from("%PDF-source-a");
|
||||
const sourceB = Buffer.from("%PDF-source-b");
|
||||
fs.writeFileSync(pdfPath, sourceA);
|
||||
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
|
||||
if (options?.imageDir) {
|
||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
||||
fs.writeFileSync(
|
||||
path.join(options.imageDir, "p11-img0.png"),
|
||||
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
|
||||
);
|
||||
}
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
|
||||
const originalStat = fs.statSync(pdfPath);
|
||||
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
fs.writeFileSync(pdfPath, sourceB);
|
||||
fs.utimesSync(pdfPath, originalStat.atime, originalStat.mtime);
|
||||
const second = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
|
||||
expect(imageBytes(first).subarray(TINY_PNG.length)).toEqual(sourceA);
|
||||
expect(imageBytes(second).subarray(TINY_PNG.length)).toEqual(sourceB);
|
||||
expect(spy).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it("converts an immutable snapshot when the source changes during extraction", async () => {
|
||||
const sourceA = Buffer.from("%PDF-source-a");
|
||||
const sourceB = Buffer.from("%PDF-source-b");
|
||||
fs.writeFileSync(pdfPath, sourceA);
|
||||
const entered = Promise.withResolvers<void>();
|
||||
const release = Promise.withResolvers<void>();
|
||||
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
|
||||
entered.resolve();
|
||||
await release.promise;
|
||||
if (options?.imageDir) {
|
||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
||||
fs.writeFileSync(
|
||||
path.join(options.imageDir, "p11-img0.png"),
|
||||
Buffer.concat([TINY_PNG, fs.readFileSync(sourcePath)]),
|
||||
);
|
||||
}
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
const pending = new ReadTool(makeSession(testDir)).execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
|
||||
await entered.promise;
|
||||
fs.writeFileSync(pdfPath, sourceB);
|
||||
release.resolve();
|
||||
const result = await pending;
|
||||
|
||||
expect(imageBytes(result).subarray(TINY_PNG.length)).toEqual(sourceA);
|
||||
});
|
||||
|
||||
it("coalesces concurrent cold image extraction", async () => {
|
||||
const { entered, release, spy } = mockBlockedExtraction();
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
const second = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
|
||||
await entered.promise;
|
||||
const conversionCount = spy.mock.calls.length;
|
||||
release.resolve();
|
||||
const [firstResult, secondResult] = await Promise.all([first, second]);
|
||||
|
||||
expect(conversionCount).toBe(1);
|
||||
expect(imageBytes(firstResult)).toEqual(imageBytes(secondResult));
|
||||
});
|
||||
|
||||
it("keeps shared extraction running when its owner aborts", async () => {
|
||||
const { entered, release, spy } = mockBlockedExtraction();
|
||||
const bothAttached = extractionWaitersAttached(2);
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const ownerController = new AbortController();
|
||||
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, ownerController.signal);
|
||||
await entered.promise;
|
||||
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
await bothAttached;
|
||||
|
||||
ownerController.abort();
|
||||
await expect(owner).rejects.toThrow(/Aborted|Cancelled/);
|
||||
release.resolve();
|
||||
const result = await joiner;
|
||||
|
||||
expect(result.content.some(content => content.type === "image")).toBe(true);
|
||||
expect(spy).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("keeps shared extraction running when a joiner aborts", async () => {
|
||||
const { entered, release, spy } = mockBlockedExtraction();
|
||||
const bothAttached = extractionWaitersAttached(2);
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const joinerController = new AbortController();
|
||||
const owner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
await entered.promise;
|
||||
const joiner = tool.execute("call", { path: `${pdfPath}:p11-img0.png` }, joinerController.signal);
|
||||
await bothAttached;
|
||||
|
||||
joinerController.abort();
|
||||
await expect(joiner).rejects.toThrow(/Aborted|Cancelled/);
|
||||
release.resolve();
|
||||
const result = await owner;
|
||||
|
||||
expect(result.content.some(content => content.type === "image")).toBe(true);
|
||||
expect(spy).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("cleans temporary extraction state when the only caller aborts", async () => {
|
||||
const entered = Promise.withResolvers<void>();
|
||||
let snapshotPath: string | undefined;
|
||||
let stagingDir: string | undefined;
|
||||
vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, signal, options) => {
|
||||
snapshotPath = sourcePath;
|
||||
stagingDir = options?.imageDir;
|
||||
entered.resolve();
|
||||
const aborted = Promise.withResolvers<void>();
|
||||
const onAbort = () => aborted.resolve();
|
||||
if (signal?.aborted) onAbort();
|
||||
else signal?.addEventListener("abort", onAbort, { once: true });
|
||||
await aborted.promise;
|
||||
signal?.removeEventListener("abort", onAbort);
|
||||
signal?.throwIfAborted();
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
const controller = new AbortController();
|
||||
const pending = new ReadTool(makeSession(testDir)).execute(
|
||||
"call",
|
||||
{ path: `${pdfPath}:p11-img0.png` },
|
||||
controller.signal,
|
||||
);
|
||||
|
||||
await entered.promise;
|
||||
controller.abort();
|
||||
await expect(pending).rejects.toThrow(/Aborted|Cancelled/);
|
||||
if (!snapshotPath || !stagingDir) throw new Error("Expected extraction paths");
|
||||
|
||||
expect(fs.existsSync(path.dirname(snapshotPath))).toBe(false);
|
||||
expect(fs.existsSync(stagingDir)).toBe(false);
|
||||
});
|
||||
|
||||
it("does not let a failed generation delete a replacement generation", async () => {
|
||||
const sourceA = Buffer.from("%PDF-source-a");
|
||||
const sourceB = Buffer.from("%PDF-source-b");
|
||||
fs.writeFileSync(pdfPath, sourceA);
|
||||
const firstEntered = Promise.withResolvers<void>();
|
||||
const failFirst = Promise.withResolvers<void>();
|
||||
const spy = vi.spyOn(markit, "convertFileWithMarkit").mockImplementation(async (sourcePath, _signal, options) => {
|
||||
const source = fs.readFileSync(sourcePath);
|
||||
if (source.equals(sourceA)) {
|
||||
firstEntered.resolve();
|
||||
await failFirst.promise;
|
||||
return { ok: false, content: "", error: "generation A failed" };
|
||||
}
|
||||
if (options?.imageDir) {
|
||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), Buffer.concat([TINY_PNG, source]));
|
||||
}
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const first = tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
await firstEntered.promise;
|
||||
fs.writeFileSync(pdfPath, sourceB);
|
||||
|
||||
const replacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
failFirst.resolve();
|
||||
await expect(first).rejects.toThrow(/Cannot extract images/);
|
||||
const cachedReplacement = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
|
||||
expect(imageBytes(replacement).subarray(TINY_PNG.length)).toEqual(sourceB);
|
||||
expect(imageBytes(cachedReplacement)).toEqual(imageBytes(replacement));
|
||||
expect(spy).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it("isolates equal-content PDFs with the same basename in different directories", async () => {
|
||||
const otherDir = path.join(testDir, "other");
|
||||
const otherPdfPath = path.join(otherDir, path.basename(pdfPath));
|
||||
fs.mkdirSync(otherDir, { recursive: true });
|
||||
fs.writeFileSync(otherPdfPath, fs.readFileSync(pdfPath));
|
||||
let conversion = 0;
|
||||
const spy = vi
|
||||
.spyOn(markit, "convertFileWithMarkit")
|
||||
.mockImplementation(async (_sourcePath, _signal, options) => {
|
||||
conversion++;
|
||||
if (options?.imageDir) {
|
||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
||||
fs.writeFileSync(
|
||||
path.join(options.imageDir, "p11-img0.png"),
|
||||
Buffer.concat([TINY_PNG, Buffer.from(String(conversion))]),
|
||||
);
|
||||
}
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
|
||||
const first = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
const second = await tool.execute("call", { path: `${otherPdfPath}:p11-img0.png` });
|
||||
|
||||
expect(imageBytes(first).subarray(TINY_PNG.length).toString()).toBe("1");
|
||||
expect(imageBytes(second).subarray(TINY_PNG.length).toString()).toBe("2");
|
||||
expect(spy).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it("supports PDF basenames at the filesystem component limit", async () => {
|
||||
const longPdfPath = path.join(testDir, `${"a".repeat(250)}.pdf`);
|
||||
fs.writeFileSync(longPdfPath, "%PDF-stub");
|
||||
mockExtraction();
|
||||
|
||||
const result = await new ReadTool(makeSession(testDir)).execute("call", {
|
||||
path: `${longPdfPath}:p11-img0.png`,
|
||||
});
|
||||
|
||||
expect(result.content.some(content => content.type === "image")).toBe(true);
|
||||
});
|
||||
|
||||
it("errors with the available members for an unknown member", async () => {
|
||||
mockExtraction();
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
await expect(tool.execute("call", { path: `${pdfPath}:does-not-exist.png` })).rejects.toThrow(
|
||||
/not found.*p11-img0\.png/s,
|
||||
);
|
||||
});
|
||||
|
||||
it("rejects member traversal attempts", async () => {
|
||||
mockExtraction();
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
// `../../escape.png` matches the image-member shape but is not a known
|
||||
// basename, so it must be refused rather than joined into the cache path.
|
||||
await expect(tool.execute("call", { path: `${pdfPath}:../../escape.png` })).rejects.toThrow(/not found/);
|
||||
});
|
||||
|
||||
it("lists extractable members for a trailing-colon read", async () => {
|
||||
mockExtraction({ "p1-img0.png": TINY_PNG, "p2-img0.png": TINY_PNG });
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
const result = await tool.execute("call", { path: `${pdfPath}:` });
|
||||
const text = result.content
|
||||
.filter(c => c.type === "text")
|
||||
.map(c => c.text)
|
||||
.join("\n");
|
||||
expect(text).toContain(`read \`${pdfPath}:p1-img0.png\``);
|
||||
expect(text).toContain(`read \`${pdfPath}:p2-img0.png\``);
|
||||
});
|
||||
|
||||
it("does not cache a failed conversion", async () => {
|
||||
let failedSnapshotPath: string | undefined;
|
||||
let failedImageDir: string | undefined;
|
||||
const spy = vi.spyOn(markit, "convertFileWithMarkit");
|
||||
spy.mockImplementationOnce(async (sourcePath, _signal, options) => {
|
||||
failedSnapshotPath = sourcePath;
|
||||
failedImageDir = options?.imageDir;
|
||||
return { ok: false, content: "", error: "boom" };
|
||||
});
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
await expect(tool.execute("call", { path: `${pdfPath}:p11-img0.png` })).rejects.toThrow(/Cannot extract images/);
|
||||
if (!failedSnapshotPath || !failedImageDir) throw new Error("Expected failed extraction paths");
|
||||
expect(fs.existsSync(path.dirname(failedSnapshotPath))).toBe(false);
|
||||
expect(fs.existsSync(path.join(failedImageDir, ".extracted"))).toBe(false);
|
||||
|
||||
spy.mockImplementationOnce(async (_filePath: string, _signal, options) => {
|
||||
if (options?.imageDir) {
|
||||
fs.mkdirSync(options.imageDir, { recursive: true });
|
||||
fs.writeFileSync(path.join(options.imageDir, "p11-img0.png"), TINY_PNG);
|
||||
}
|
||||
return { ok: true, content: "" };
|
||||
});
|
||||
const result = await tool.execute("call", { path: `${pdfPath}:p11-img0.png` });
|
||||
expect(result.content.some(c => c.type === "image")).toBe(true);
|
||||
expect(spy).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,79 @@
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import type { AgentToolResult } from "@oh-my-pi/pi-agent-core";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { ReadTool, type ReadToolDetails } from "@oh-my-pi/pi-coding-agent/tools/read";
|
||||
import * as markit from "@oh-my-pi/pi-coding-agent/utils/markit";
|
||||
import { removeWithRetries } from "@oh-my-pi/pi-utils";
|
||||
|
||||
function makeSession(cwd: string): ToolSession {
|
||||
return {
|
||||
cwd,
|
||||
hasUI: false,
|
||||
getSessionFile: () => null,
|
||||
getSessionSpawns: () => "*",
|
||||
settings: Settings.isolated({ "images.autoResize": false }),
|
||||
} as ToolSession;
|
||||
}
|
||||
|
||||
function textOf(result: AgentToolResult<ReadToolDetails>): string {
|
||||
return result.content
|
||||
.filter(entry => entry.type === "text")
|
||||
.map(entry => entry.text)
|
||||
.join("\n");
|
||||
}
|
||||
|
||||
describe("read unsupported PDF image members", () => {
|
||||
let testDir: string;
|
||||
let pdfPath: string;
|
||||
|
||||
beforeEach(async () => {
|
||||
testDir = await fs.mkdtemp(path.join(os.tmpdir(), "read-pdf-image-unsupported-"));
|
||||
pdfPath = path.join(testDir, "doc.pdf");
|
||||
await fs.writeFile(pdfPath, `%PDF-stub-${testDir}`);
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
vi.restoreAllMocks();
|
||||
await removeWithRetries(testDir);
|
||||
});
|
||||
|
||||
it("directs former image listing and PNG member reads to browser rendering", async () => {
|
||||
const tool = new ReadTool(makeSession(testDir));
|
||||
|
||||
for (const readPath of [`${pdfPath}:`, `${pdfPath}:p1-img0.png`]) {
|
||||
try {
|
||||
await tool.execute("read-pdf-image", { path: readPath });
|
||||
throw new Error("Expected the PDF image read to fail");
|
||||
} catch (error) {
|
||||
expect(error).toBeInstanceOf(Error);
|
||||
const message = (error as Error).message;
|
||||
expect(message).toContain("pdf-inspector cannot render PDF images");
|
||||
expect(message).toContain("Puppeteer browser tool");
|
||||
expect(message).toContain(`read '${pdfPath}' for extracted text`);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it("preserves a literal filename that looks like a PDF image listing", async () => {
|
||||
const literalPath = `${pdfPath}:`;
|
||||
await fs.writeFile(literalPath, "literal colon path wins\n");
|
||||
|
||||
const result = await new ReadTool(makeSession(testDir)).execute("read-literal", { path: literalPath });
|
||||
expect(textOf(result)).toContain("literal colon path wins");
|
||||
});
|
||||
|
||||
it("routes PDF line selectors through normal document conversion", async () => {
|
||||
const convert = vi.spyOn(markit, "convertFileWithMarkit").mockResolvedValue({
|
||||
ok: true,
|
||||
content: "first line\nselected line\nthird line\n",
|
||||
});
|
||||
|
||||
const result = await new ReadTool(makeSession(testDir)).execute("read-pdf-lines", { path: `${pdfPath}:2-2` });
|
||||
expect(convert).toHaveBeenCalledTimes(1);
|
||||
expect(textOf(result)).toContain("selected line");
|
||||
});
|
||||
});
|
||||
@@ -105,7 +105,7 @@ describe("document conversion cache", () => {
|
||||
|
||||
it("skips cache for imageDir conversions", async () => {
|
||||
const convert = vi.spyOn(Markit.prototype, "convert").mockResolvedValue({ markdown: "image body" });
|
||||
const docPath = path.join(testDir, "image-doc.pdf");
|
||||
const docPath = path.join(testDir, "image-doc.docx");
|
||||
await fs.writeFile(docPath, new TextEncoder().encode("image bytes"));
|
||||
const imageDir = path.join(testDir, "images");
|
||||
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added the async `pdfToMarkdown` native API backed by `pdf-inspector`, with page numbering, page-count, OCR-needed-page, and encoding-issue metadata.
|
||||
|
||||
### Changed
|
||||
|
||||
- Docker images (`Dockerfile`, `scripts/install-tests/*.dockerfile`) build the native addon through the cargo/napi-rs backend (`OMP_NATIVE_BUILD_BACKEND=cargo`) instead of Bazel: a single fixed host target gains nothing from hermetic cross toolchains, and none of those images shipped bazelisk. `OMP_NATIVE_CARGO_PROFILE` picks the profile for that path (images use `ci`, local default stays `local`).
|
||||
|
||||
@@ -10,6 +10,7 @@ Native Rust functionality via N-API.
|
||||
- **Audio**: Cross-platform low-latency microphone capture and gapless speaker playback
|
||||
- **WebRTC**: Native Opus media, SDP offer/answer negotiation, and data-channel events for live sessions
|
||||
- **File locking**: Process-owned cross-process locks with in-memory kernel names on Linux/Windows and `flock(2)` sidecars on other Unix platforms
|
||||
- **PDF**: In-memory PDF-to-Markdown extraction with OCR-page classification via `pdf-inspector`
|
||||
|
||||
General-purpose image processing (decode/resize/encode for files and buffers)
|
||||
lives in [`Bun.Image`](https://bun.com/docs/runtime/image) on the JS side; this
|
||||
@@ -19,7 +20,7 @@ that terminal protocol.
|
||||
## Usage
|
||||
|
||||
```typescript
|
||||
import { grep, find, encodeSixel } from "@oh-my-pi/pi-natives";
|
||||
import { encodeSixel, grep, pdfToMarkdown } from "@oh-my-pi/pi-natives";
|
||||
|
||||
// Grep for a pattern
|
||||
const results = await grep({
|
||||
@@ -38,6 +39,10 @@ const files = await find({
|
||||
|
||||
// SIXEL encode for a terminal cell box (px)
|
||||
const sequence = encodeSixel(pngBytes, widthPx, heightPx);
|
||||
|
||||
// Extract PDF text and identify pages that still need OCR
|
||||
const pdf = await pdfToMarkdown(pdfBytes);
|
||||
console.log(pdf.markdown, pdf.pagesNeedingOcr);
|
||||
```
|
||||
|
||||
## Building
|
||||
|
||||
Vendored
+26
@@ -1578,6 +1578,32 @@ export interface PatchHunk {
|
||||
lines: Array<string>
|
||||
}
|
||||
|
||||
/** Markdown and inspection metadata produced from a PDF document. */
|
||||
export interface PdfMarkdownResult {
|
||||
/** Extracted document content in Markdown format. */
|
||||
markdown: string
|
||||
/** Document title from PDF metadata, when present. */
|
||||
title?: string
|
||||
/** Total number of pages in the document. */
|
||||
pageCount: number
|
||||
/** One-indexed page numbers whose content requires OCR. */
|
||||
pagesNeedingOcr: Array<number>
|
||||
/** Whether the document contains text encoding problems. */
|
||||
hasEncodingIssues: boolean
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert an in-memory PDF to Markdown and return its inspection metadata.
|
||||
*
|
||||
* Conversion copies the typed array before dispatch so JavaScript mutation
|
||||
* cannot race the native worker.
|
||||
*
|
||||
* # Errors
|
||||
* Returns an error prefixed with `PDF conversion failed:` when the PDF cannot
|
||||
* be parsed or converted.
|
||||
*/
|
||||
export declare function pdfToMarkdown(input: Uint8Array): Promise<PdfMarkdownResult>
|
||||
|
||||
export interface PointerOptions {
|
||||
button?: string
|
||||
count?: number
|
||||
|
||||
@@ -69,6 +69,7 @@ export const matchesLegacySequence = nativeBindings.matchesLegacySequence;
|
||||
export const mmrRerankIndices = nativeBindings.mmrRerankIndices;
|
||||
export const parseKey = nativeBindings.parseKey;
|
||||
export const parseKittySequence = nativeBindings.parseKittySequence;
|
||||
export const pdfToMarkdown = nativeBindings.pdfToMarkdown;
|
||||
export const readImageFromClipboard = nativeBindings.readImageFromClipboard;
|
||||
export const renderSnapcompactPng = nativeBindings.renderSnapcompactPng;
|
||||
export const search = nativeBindings.search;
|
||||
|
||||
+1
@@ -26,6 +26,7 @@ export interface DetectCompiledBinaryInput {
|
||||
|
||||
export function detectCompiledBinary(input: DetectCompiledBinaryInput): boolean;
|
||||
|
||||
|
||||
export interface GetAddonFilenamesInput {
|
||||
tag: string;
|
||||
arch: string;
|
||||
|
||||
@@ -87,7 +87,6 @@ export function detectCompiledBinary({ embeddedAddon, env, importMetaUrl }) {
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {{ tag: string; arch: string; variant: "modern" | "baseline" | null | undefined }} input
|
||||
* @returns {string[]}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "@oh-my-pi/pi-natives",
|
||||
"version": "17.3.3",
|
||||
"description": "Native Rust bindings for audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API",
|
||||
"description": "Native Rust bindings for PDF conversion, audio, WebRTC, grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API",
|
||||
"type": "module",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -23,6 +23,7 @@ import {
|
||||
matchesKey,
|
||||
PtySession,
|
||||
parseKey,
|
||||
pdfToMarkdown,
|
||||
summarizeCode,
|
||||
supportsLanguage,
|
||||
truncateToWidth,
|
||||
@@ -87,6 +88,30 @@ async function createFifo(fifoPath: string) {
|
||||
throw new Error(await new Response(process.stderr).text());
|
||||
}
|
||||
|
||||
function textPdf(text: string): Uint8Array {
|
||||
const stream = `BT /F1 12 Tf 72 720 Td (${text}) Tj ET`;
|
||||
const objects = [
|
||||
"<< /Type /Catalog /Pages 2 0 R >>",
|
||||
"<< /Type /Pages /Kids [3 0 R] /Count 1 >>",
|
||||
"<< /Type /Page /Parent 2 0 R /MediaBox [0 0 612 792] /Resources << /Font << /F1 5 0 R >> >> /Contents 4 0 R >>",
|
||||
`<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`,
|
||||
"<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica /Encoding /WinAnsiEncoding >>",
|
||||
];
|
||||
let document = "%PDF-1.4\n";
|
||||
const offsets: number[] = [];
|
||||
for (const [index, object] of objects.entries()) {
|
||||
offsets.push(document.length);
|
||||
document += `${index + 1} 0 obj\n${object}\nendobj\n`;
|
||||
}
|
||||
const xrefOffset = document.length;
|
||||
document += `xref\n0 ${objects.length + 1}\n0000000000 65535 f \n`;
|
||||
for (const offset of offsets) {
|
||||
document += `${offset.toString().padStart(10, "0")} 00000 n \n`;
|
||||
}
|
||||
document += `trailer\n<< /Size ${objects.length + 1} /Root 1 0 R >>\nstartxref\n${xrefOffset}\n%%EOF\n`;
|
||||
return Buffer.from(document);
|
||||
}
|
||||
|
||||
describe("pi-natives", () => {
|
||||
beforeAll(async () => {
|
||||
await setupFixtures();
|
||||
@@ -763,6 +788,19 @@ describe("pi-natives", () => {
|
||||
expect(await Bun.file(markerPath).exists()).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("pdfToMarkdown", () => {
|
||||
it("isolates blocking conversion from later JavaScript buffer mutation", async () => {
|
||||
const input = textPdf("Copied PDF bytes");
|
||||
const conversion = pdfToMarkdown(input);
|
||||
input.fill(0);
|
||||
|
||||
const result = await conversion;
|
||||
|
||||
expect(result.pageCount).toBe(1);
|
||||
expect(result.markdown).toContain("Copied PDF bytes");
|
||||
});
|
||||
});
|
||||
describe("htmlToMarkdown", () => {
|
||||
it("should convert basic HTML to markdown", async () => {
|
||||
const html = "<h1>Hello World</h1><p>This is a paragraph.</p>";
|
||||
|
||||
@@ -158,24 +158,20 @@ async function generateBundle(): Promise<void> {
|
||||
if (isDryRun) {
|
||||
console.log("DRY RUN bun run gen:stats");
|
||||
console.log("DRY RUN bun --cwd=packages/collab-web run gen:tool-views");
|
||||
console.log("DRY RUN bun run gen:mupdf");
|
||||
return;
|
||||
}
|
||||
await runCommand(["bun", "run", "gen:stats"], repoRoot);
|
||||
await runCommand(["bun", "--cwd=packages/collab-web", "run", "gen:tool-views"], repoRoot);
|
||||
await runCommand(["bun", "run", "gen:mupdf"], repoRoot);
|
||||
}
|
||||
|
||||
async function resetArtifacts(): Promise<void> {
|
||||
if (isDryRun) {
|
||||
console.log("DRY RUN bun run gen:native:reset");
|
||||
console.log("DRY RUN bun run gen:stats:reset");
|
||||
console.log("DRY RUN bun run gen:mupdf:reset");
|
||||
return;
|
||||
}
|
||||
await runCommand(["bun", "run", "gen:native:reset"], repoRoot);
|
||||
await runCommand(["bun", "run", "gen:stats:reset"], repoRoot);
|
||||
await runCommand(["bun", "run", "gen:mupdf:reset"], repoRoot);
|
||||
}
|
||||
|
||||
async function main(): Promise<void> {
|
||||
|
||||
Reference in New Issue
Block a user