feat(coding-agent): vendor relevant parts of markit

This commit is contained in:
can1357
2026-06-18 18:59:35 +02:00
parent 81563e0155
commit 38d1d3a2f5
33 changed files with 3450 additions and 274 deletions
-1
View File
@@ -55,7 +55,6 @@ out.html
pi-*.html
# Generated files
packages/coding-agent/src/internal-urls/docs-index.generated.ts
packages/coding-agent/src/export/html/tool-views.generated.js
packages/natives/npm/
/runs/
-1
View File
@@ -60,7 +60,6 @@
"!**/vendor/**/*",
"!**/node_modules/**/*",
"!**/test-sessions.ts",
"!**/docs-index.generated.ts",
"!**/agent_pb.ts",
"!.worktrees/**/*",
"!.wt/**/*"
+6 -47
View File
@@ -98,11 +98,13 @@
"arktype": "catalog:",
"chalk": "catalog:",
"diff": "catalog:",
"fast-xml-parser": "catalog:",
"fflate": "catalog:",
"handlebars": "catalog:",
"linkedom": "catalog:",
"lru-cache": "catalog:",
"markit-ai": "catalog:",
"mammoth": "catalog:",
"mupdf": "catalog:",
"puppeteer-core": "catalog:",
"turndown": "catalog:",
"turndown-plugin-gfm": "catalog:",
@@ -277,7 +279,6 @@
"version": "16.0.7",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"beautiful-mermaid": "catalog:",
"handlebars": "catalog:",
"winston": "catalog:",
"winston-daily-rotate-file": "catalog:",
@@ -311,7 +312,6 @@
},
"patchedDependencies": {
"@ark/schema@0.56.0": "patches/@ark%2Fschema@0.56.0.patch",
"beautiful-mermaid@1.1.3": "patches/beautiful-mermaid@1.1.3.patch",
},
"catalog": {
"@agentclientprotocol/sdk": "0.25.0",
@@ -355,11 +355,11 @@
"@typescript/native-preview": "7.0.0-dev.20260609.1",
"@xterm/headless": "^6.0.0",
"arktype": "^2.2.0",
"beautiful-mermaid": "^1.1.3",
"chalk": "^5.6.2",
"chart.js": "^4.5.1",
"date-fns": "^4.4.0",
"diff": "^9.0.0",
"fast-xml-parser": "^5.9.0",
"fastembed": "2.1.0",
"fflate": "0.8.3",
"ghostty-web": "^0.4.0",
@@ -368,8 +368,9 @@
"lint-staged": "^17.0.7",
"lru-cache": "11.5.1",
"lucide-react": "^1.17.0",
"mammoth": "^1.12.0",
"marked": "^18.0.5",
"markit-ai": "0.5.3",
"mupdf": "^1.27.0",
"onnxruntime-node": "1.26.0",
"partial-json": "^0.1.7",
"postcss": "^8.5.15",
@@ -460,8 +461,6 @@
"@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.5.0", "", { "os": "win32", "cpu": "x64" }, "sha512-VT/lF+GId+67j8aDfLkxdxNoVApsPSTbyAtB3jJq0IWTrY77WXfbPfpngxq0bA6JCEv/7k8C9qWjDRKRznDlyw=="],
"@borewit/text-codec": ["@borewit/text-codec@0.2.2", "", {}, "sha512-DDaRehssg1aNrH4+2hnj1B7vnUGEjU6OIlyRdkMd0aUdIUvKXrJfXsy8LVtXAy7DRvYVluWbMspsRhz2lcW0mQ=="],
"@bufbuild/protobuf": ["@bufbuild/protobuf@2.12.0", "", {}, "sha512-B/XlCaFIP8LOwzo+bz5uFzATYokcwCKQcghqnlfwSmM5eX/qTkvDBnDPs+gXtX/RyjxJ4DRikECcPJbyALA8FA=="],
"@colors/colors": ["@colors/colors@1.6.0", "", {}, "sha512-Ir+AOibqzrIsL6ajt3Rz3LskB7OiMVHqltZmspbW/TJuTVuyOMirVqAkjfY6JISiLHgyNqicAC8AyHHGzNd/dA=="],
@@ -860,10 +859,6 @@
"@tailwindcss/vite": ["@tailwindcss/vite@4.3.1", "", { "dependencies": { "@tailwindcss/node": "4.3.1", "@tailwindcss/oxide": "4.3.1", "tailwindcss": "4.3.1" }, "peerDependencies": { "vite": "^5.2.0 || ^6 || ^7 || ^8" } }, "sha512-hItDHuIIlEV61R+faXu66s1K36aTurO/Qw0e45Vskz57gXl9pWOT6eg3zmcEui6CZXddbN7zd41bwmvag4JGwQ=="],
"@tokenizer/inflate": ["@tokenizer/inflate@0.4.1", "", { "dependencies": { "debug": "^4.4.3", "token-types": "^6.1.1" } }, "sha512-2mAv+8pkG6GIZiF1kNg1jAjh27IDxEPKwdGul3snfztFerfPGI1LjDezZp3i7BElXompqEtPmoPx6c2wgtWsOA=="],
"@tokenizer/token": ["@tokenizer/token@0.3.0", "", {}, "sha512-OvjF+z51L3ov0OyAU0duzsYuvO01PH7x4t6DJx+guahgTnBHkhJdG7soQeTSFLWN3efnHyibZ4Z8l2EuWwJN3A=="],
"@ts-morph/common": ["@ts-morph/common@0.29.0", "", { "dependencies": { "minimatch": "^10.0.1", "path-browserify": "^1.0.1", "tinyglobby": "^0.2.14" } }, "sha512-35oUmphHbJvQ/+UTwFNme/t2p3FoKiGJ5auTjjpNTop2dyREspirjMy82PLSC1pnDJ8ah1GU98hwpVt64YXQsg=="],
"@tybys/wasm-util": ["@tybys/wasm-util@0.10.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg=="],
@@ -936,8 +931,6 @@
"baseline-browser-mapping": ["baseline-browser-mapping@2.10.37", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-girxaJ7WZssDOFhzCGZTDKoTa1gk6A1TbflaYTpykLJ4UU9Fz9kx1aREM8JCuoVHbL8X8T/mJg7w2oYSq72Oig=="],
"beautiful-mermaid": ["beautiful-mermaid@1.1.3", "", { "dependencies": { "elkjs": "^0.11.0", "entities": "^7.0.1" } }, "sha512-TItrtrAyHp1vwFfFVYauWGrquouk/6SS21Aq3RsxindSYZODcN4xYrPZD6BiZRU+o5mKJzDPz9MUSMvELdylyg=="],
"before-after-hook": ["before-after-hook@4.0.0", "", {}, "sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ=="],
"bluebird": ["bluebird@3.4.7", "", {}, "sha512-iD3898SR7sWVRHbiQv+sHUtHnMvC1o3nW5rAcqnq3uOn07DSAppZYUkIGslDz6gXC7HfunPe7YVBgoEJASPcHA=="],
@@ -988,8 +981,6 @@
"colorette": ["colorette@2.0.20", "", {}, "sha512-IfEDxwoWIjkeXL1eXcDiow4UbKjhLdq6/EuSVR9GMN7KVH3r9gQ83e73hsz1Nd1T3ijd5xv1wcWRYO+D6kCI2w=="],
"commander": ["commander@14.0.3", "", {}, "sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw=="],
"content-type": ["content-type@2.0.0", "", {}, "sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ=="],
"convert-source-map": ["convert-source-map@2.0.0", "", {}, "sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg=="],
@@ -1034,8 +1025,6 @@
"electron-to-chromium": ["electron-to-chromium@1.5.372", "", {}, "sha512-M3yhbAlilnwqC8D21t28UCDGHyitShTmmLRU/H+b74P6Ski16Nb9HONYEaVpMj/pwC7BEo5B95FpjODLCWbtfA=="],
"elkjs": ["elkjs@0.11.1", "", {}, "sha512-zxxR9k+rx5ktMwT/FwyLdPCrq7xN6e4VGGHH8hA01vVYKjTFik7nHOxBnAYtrgYUB1RpAiLvA1/U2YraWxyKKg=="],
"emnapi": ["emnapi@1.11.1", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-kSRjhIcxjMFsBqk7ORvoc9aA5SBKDmecrtF5RMcmOTao0kD/zamaxsuTxMI8C1//wGUuvE7a+19pCE7AEhGVnA=="],
"emoji-regex": ["emoji-regex@8.0.0", "", {}, "sha512-MSjYzcWNOA0ewAHpz0MxpYFvwg6yjy1NG3xteoqz644VCo/RPgnr1/GGt+ic3iJTzQ8Eu3TdM14SawnVUmGE6A=="],
@@ -1062,8 +1051,6 @@
"eventemitter3": ["eventemitter3@5.0.4", "", {}, "sha512-mlsTRyGaPBjPedk6Bvw+aqbsXDtoAyAzm5MO7JgU+yVRyMQ5O8bD4Kcci7BS85f93veegeCPkL8R4GLClnjLFw=="],
"exifr": ["exifr@7.1.3", "", {}, "sha512-g/aje2noHivrRSLbAUtBPWFbxKdKhgj/xr1vATDdUXPOFYJlQ62Ft0oy+72V6XLIpDJfHs6gXLbBLAolqOXYRw=="],
"fast-string-truncated-width": ["fast-string-truncated-width@3.0.3", "", {}, "sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g=="],
"fast-string-width": ["fast-string-width@3.0.2", "", { "dependencies": { "fast-string-truncated-width": "^3.0.2" } }, "sha512-gX8LrtNEI5hq8DVUfRQMbr5lpaS4nMIWV+7XEbXk2b8kiQIizgnlr12B4dA3ZEx3308ze0O4Q1R+cHts8kyUJg=="],
@@ -1084,8 +1071,6 @@
"file-stream-rotator": ["file-stream-rotator@0.6.1", "", { "dependencies": { "moment": "^2.29.1" } }, "sha512-u+dBid4PvZw17PmDeRcNOtCP9CCK/9lRN2w+r1xIS7yOL9JFrIBKTvrYsxT4P0pGtThYTn++QS5ChHaUov3+zQ=="],
"file-type": ["file-type@21.3.4", "", { "dependencies": { "@tokenizer/inflate": "^0.4.1", "strtok3": "^10.3.4", "token-types": "^6.1.1", "uint8array-extras": "^1.4.0" } }, "sha512-Ievi/yy8DS3ygGvT47PjSfdFoX+2isQueoYP1cntFW1JLYAuS4GD7NUPGg4zv2iZfV52uDyk5w5Z0TdpRS6Q1g=="],
"flatbuffers": ["flatbuffers@25.9.23", "", {}, "sha512-MI1qs7Lo4Syw0EOzUl0xjs2lsoeqFku44KpngfIduHBYvzm8h2+7K8YMQh1JtVVVrUvhLpNwqVi4DERegUJhPQ=="],
"fn.name": ["fn.name@1.1.0", "", {}, "sha512-GRnmB5gPyJpAhTQdSZTSp9uaPSvl09KoYcMQtsB9rQoOmzs9dH6ffeccH+Z+cv6P68Hu5bC6JjRh4Ah/mHSNRw=="],
@@ -1126,8 +1111,6 @@
"iconv-lite": ["iconv-lite@0.7.2", "", { "dependencies": { "safer-buffer": ">= 2.1.2 < 3.0.0" } }, "sha512-im9DjEDQ55s9fL4EYzOAv0yMqmMBSZp6G0VvFyTMPKWxiSBHUj9NW/qqLmXUwXrrM7AvqSlTCfvqRb0cM8yYqw=="],
"ieee754": ["ieee754@1.2.1", "", {}, "sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA=="],
"immediate": ["immediate@3.0.6", "", {}, "sha512-XXOFtyqDjNDAQxVfYxuF7g9Il/IbWmmlQg2MYKOH8ExIT1qg6xc4zyS3HaEEATgs1btfzxq15ciUiY7gjSXRGQ=="],
"inherits": ["inherits@2.0.4", "", {}, "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ=="],
@@ -1210,12 +1193,8 @@
"marked": ["marked@18.0.5", "", { "bin": { "marked": "bin/marked.js" } }, "sha512-S6GcvALHg6K4ohtu4E7x0a1AqhAjp6cV8KhLSyN9qVapnzJkusVBxZRcIU9AeYsbe6P1hKDusSbEOzGyyuce6w=="],
"markit-ai": ["markit-ai@0.5.3", "", { "dependencies": { "chalk": "^5.6.2", "commander": "^14.0.3", "exifr": "^7.1.3", "fast-xml-parser": "^5.5.9", "jszip": "^3.10.1", "mammoth": "^1.9.0", "mupdf": "^1.27.0", "music-metadata": "^11.12.3", "rss-parser": "^3.13.0", "turndown": "^7.2.0", "turndown-plugin-gfm": "^1.0.2" }, "bin": { "markit": "dist/main.js" } }, "sha512-h4nhn6a/SNXEdc3kLVtL37TspxjUNCNL0OM7LRWxd389ZByI/B7bjNNgxFdVAT0O+H7ZekSwLdVe/lws1l2AZQ=="],
"matcher": ["matcher@4.0.0", "", { "dependencies": { "escape-string-regexp": "^4.0.0" } }, "sha512-S6x5wmcDmsDRRU/c2dkccDwQPXoFczc5+HpQ2lON8pnvHlnvHAHj5WlLVvw6n6vNyHuVugYrFohYxbS+pvFpKQ=="],
"media-typer": ["media-typer@2.0.0", "", {}, "sha512-kOy3OxT2HH39N70UnKgu4NWDZjLOz8W/mfyvniHjRH/DrL3f2pOfvWQ4p60offbbtDAnXWp0v9LfMIqMec269Q=="],
"merge-anything": ["merge-anything@5.1.7", "", { "dependencies": { "is-what": "^4.1.8" } }, "sha512-eRtbOb1N5iyH0tkQDAoQ4Ipsp/5qSR79Dzrz8hEPxRX10RWWR/iQXdoKmBSRCThY1Fh5EhISDtpSc93fpxUniQ=="],
"mimic-function": ["mimic-function@5.0.1", "", {}, "sha512-VP79XUPxV2CigYP3jWwAUFSku2aKqBH7uTAapFWCBqutsbmDo96KY5o8uh6U+/YSIn5OxJnXp73beVkpqMIGhA=="],
@@ -1240,8 +1219,6 @@
"mupdf": ["mupdf@1.27.0", "", {}, "sha512-vEPUYwZeu5NgiFLz4e20R7Vp2pNY7szirGEvTxHyQQpQs6ab4DeGdonwT6sH1JZG5EhyHSrojZrZn2/0ee6qZQ=="],
"music-metadata": ["music-metadata@11.13.0", "", { "dependencies": { "@borewit/text-codec": "^0.2.2", "@tokenizer/token": "^0.3.0", "content-type": "^2.0.0", "debug": "^4.4.3", "file-type": "^21.3.4", "media-typer": "^2.0.0", "strtok3": "^10.3.5", "token-types": "^6.1.2", "uint8array-extras": "^1.5.0", "win-guid": "^0.2.1" } }, "sha512-uXRaov9dfjSpQufXIU7sMxVZnh+FilCQv2mXn+K5EJ/decP3dTWrgvPYa5r6MtRbieNSCE708Da4J0u1UGfQIw=="],
"mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="],
"nanoid": ["nanoid@3.3.12", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-ZB9RH/39qpq5Vu6Y+NmUaFhQR6pp+M2Xt76XBnEwDaGcVAqhlvxrl3B2bKS5D3NH3QR76v3aSrKaF/Kiy7lEtQ=="],
@@ -1322,16 +1299,12 @@
"rolldown": ["rolldown@1.0.3", "", { "dependencies": { "@oxc-project/types": "=0.133.0", "@rolldown/pluginutils": "^1.0.0" }, "optionalDependencies": { "@rolldown/binding-android-arm64": "1.0.3", "@rolldown/binding-darwin-arm64": "1.0.3", "@rolldown/binding-darwin-x64": "1.0.3", "@rolldown/binding-freebsd-x64": "1.0.3", "@rolldown/binding-linux-arm-gnueabihf": "1.0.3", "@rolldown/binding-linux-arm64-gnu": "1.0.3", "@rolldown/binding-linux-arm64-musl": "1.0.3", "@rolldown/binding-linux-ppc64-gnu": "1.0.3", "@rolldown/binding-linux-s390x-gnu": "1.0.3", "@rolldown/binding-linux-x64-gnu": "1.0.3", "@rolldown/binding-linux-x64-musl": "1.0.3", "@rolldown/binding-openharmony-arm64": "1.0.3", "@rolldown/binding-wasm32-wasi": "1.0.3", "@rolldown/binding-win32-arm64-msvc": "1.0.3", "@rolldown/binding-win32-x64-msvc": "1.0.3" }, "bin": { "rolldown": "./bin/cli.mjs" } }, "sha512-i00lAJ2ks1BYr7rjNjKC7BcqAS7nVfiT3QX1SI5aY+AFHblCmaUf9OE9dbdzDvW6dJxbi2ZCZiy9v3CcwOiX3g=="],
"rss-parser": ["rss-parser@3.13.0", "", { "dependencies": { "entities": "^2.0.3", "xml2js": "^0.5.0" } }, "sha512-7jWUBV5yGN3rqMMj7CZufl/291QAhvrrGpDNE4k/02ZchL0npisiYYqULF71jCEKoIiHvK/Q2e6IkDwPziT7+w=="],
"safe-buffer": ["safe-buffer@5.1.2", "", {}, "sha512-Gd2UZBJDkXlY7GbJxfsE8/nvKkUEU1G38c1siN6QP6a9PT9MmHB8GnpscSmMJSoF8LOIrt8ud/wPtojys4G6+g=="],
"safe-stable-stringify": ["safe-stable-stringify@2.5.0", "", {}, "sha512-b3rppTKm9T+PsVCBEOUR46GWI7fdOs00VKZ1+9c1EWDaDMvjQc6tUwuFyIprgGgTcWoVHSKrU8H31ZHA2e0RHA=="],
"safer-buffer": ["safer-buffer@2.1.2", "", {}, "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg=="],
"sax": ["sax@1.6.0", "", {}, "sha512-6R3J5M4AcbtLUdZmRv2SygeVaM7IhrLXu9BmnOGmmACak8fiUtOsYNWUS4uK7upbmHIBbLBeFeI//477BKLBzA=="],
"scheduler": ["scheduler@0.27.0", "", {}, "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q=="],
"semver": ["semver@7.8.4", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-rUCObTnP32Q08R2uuIrt7r9PlEonuTmtuXYcW6s5kjdlj3xbnwe+21yXptAUYcMAABLkYYTtnmzb3w3EDZfueA=="],
@@ -1390,8 +1363,6 @@
"strnum": ["strnum@2.4.0", "", { "dependencies": { "anynum": "^1.0.0" } }, "sha512-sHrVyWWdq28RbhjuJdZsA1SnGRJV6NiXbk6AXBxDOsgAcA+lmpUZCYjOdLBxkXMwis6RRe7dlZt4VlIWFVzkmg=="],
"strtok3": ["strtok3@10.3.5", "", { "dependencies": { "@tokenizer/token": "^0.3.0" } }, "sha512-ki4hZQfh5rX0QDLLkOCj+h+CVNkqmp/CMf8v8kZpkNVK6jGQooMytqzLZYUVYIZcFZ6yDB70EfD8POcFXiF5oA=="],
"tailwindcss": ["tailwindcss@4.3.1", "", {}, "sha512-hk+TB1m+K8CYNrP6rjQaq/Y+4Zylwpa87mLYBKCunwnnQ9p+fHb7kmSfGqyEJoxF/O6CDyABWVFEafNSYKll+Q=="],
"tapable": ["tapable@2.3.3", "", {}, "sha512-uxc/zpqFg6x7C8vOE7lh6Lbda8eEL9zmVm/PLeTPBRhh1xCgdWaQ+J1CUieGpIfm2HdtsUpRv+HshiasBMcc6A=="],
@@ -1404,8 +1375,6 @@
"tinyglobby": ["tinyglobby@0.2.17", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g=="],
"token-types": ["token-types@6.1.2", "", { "dependencies": { "@borewit/text-codec": "^0.2.1", "@tokenizer/token": "^0.3.0", "ieee754": "^1.2.1" } }, "sha512-dRXchy+C0IgK8WPC6xvCHFRIWYUbqqdEIKPaKo/AcTUNzwLTK6AH7RjdLWsEZcAN/TBdtfUw3PYEgPr5VPr6ww=="],
"triple-beam": ["triple-beam@1.4.1", "", {}, "sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg=="],
"ts-morph": ["ts-morph@28.0.0", "", { "dependencies": { "@ts-morph/common": "~0.29.0", "code-block-writer": "^13.0.3" } }, "sha512-Wp3tnZ2bzwxyTZMtgWVzXDfm7lB1Drz+y9DmmYH/L702PQhPyVrp3pkou3yIz4qjS14GY9kcpmLiOOMvl8oG1g=="],
@@ -1428,8 +1397,6 @@
"uhyphen": ["uhyphen@0.2.0", "", {}, "sha512-qz3o9CHXmJJPGBdqzab7qAYuW8kQGKNEuoHFYrBwV6hWIMcpAmxDLXojcHfFr9US1Pe6zUswEIJIbLI610fuqA=="],
"uint8array-extras": ["uint8array-extras@1.5.0", "", {}, "sha512-rvKSBiC5zqCCiDZ9kAOszZcDvdAHwwIKJG33Ykj43OKcWsnmcBRL09YTU4nOeHZ8Y2a7l1MgTd08SBe9A8Qj6A=="],
"underscore": ["underscore@1.13.8", "", {}, "sha512-DXtD3ZtEQzc7M8m4cXotyHR+FAS18C64asBYY5vqZexfYryNNnDc02W4hKg3rdQuqOYas1jkseX0+nZXjTXnvQ=="],
"undici-types": ["undici-types@7.24.6", "", {}, "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg=="],
@@ -1448,8 +1415,6 @@
"webdriver-bidi-protocol": ["webdriver-bidi-protocol@0.4.2", "", {}, "sha512-VSV+fzfChirL3e7jay2yUC7B4HQCGtEWEg/MSSQbK+qWbqeGlRLlXTzPpYr3XGUvbpDHumWZBJxgesg4N7dbtA=="],
"win-guid": ["win-guid@0.2.1", "", {}, "sha512-gEIQU4mkgl2OPeoNrWflcJFJ3Ae2BPd4eCsHHA/XikslkIVms/nHhvnvzIZV7VLmBvtFlDOzLt9rrZT+n6D67A=="],
"winston": ["winston@3.19.0", "", { "dependencies": { "@colors/colors": "^1.6.0", "@dabh/diagnostics": "^2.0.8", "async": "^3.2.3", "is-stream": "^2.0.0", "logform": "^2.7.0", "one-time": "^1.0.0", "readable-stream": "^3.4.0", "safe-stable-stringify": "^2.3.1", "stack-trace": "0.0.x", "triple-beam": "^1.3.0", "winston-transport": "^4.9.0" } }, "sha512-LZNJgPzfKR+/J3cHkxcpHKpKKvGfDZVPS4hfJCc4cCG0CgYzvlD6yE/S3CIL/Yt91ak327YCpiF/0MyeZHEHKA=="],
"winston-daily-rotate-file": ["winston-daily-rotate-file@5.0.0", "", { "dependencies": { "file-stream-rotator": "^0.6.1", "object-hash": "^3.0.0", "triple-beam": "^1.4.1", "winston-transport": "^4.7.0" }, "peerDependencies": { "winston": "^3" } }, "sha512-JDjiXXkM5qvwY06733vf09I2wnMXpZEhxEVOSPenZMii+g7pcDcTBt2MRugnoi8BwVSuCT2jfRXBUy+n1Zz/Yw=="],
@@ -1464,8 +1429,6 @@
"xml-naming": ["xml-naming@0.1.0", "", {}, "sha512-k8KO9hrMyNk6tUWqUfkTEZbezRRpONVOzUTnc97VnCvyj6Tf9lyUR9EDAIeiVLv56jsMcoXEwjW8Kv5yPY52lw=="],
"xml2js": ["xml2js@0.5.0", "", { "dependencies": { "sax": ">=0.6.0", "xmlbuilder": "~11.0.0" } }, "sha512-drPFnkQJik/O+uPKpqSgr22mpuFHqKdbS835iAQrUC73L2F5WkboIRd63ai/2Yg6I1jzifPFKH2NTK+cfglkIA=="],
"xmlbuilder": ["xmlbuilder@10.1.1", "", {}, "sha512-OyzrcFLL/nb6fMGHbiRDuPup9ljBycsdCypwuyg5AAHvyWzGfChJpCXMG88AGTIMFhGZ9RccFN1e6lhg3hkwKg=="],
"y18n": ["y18n@5.0.8", "", {}, "sha512-0pfFzegeDWJHJIAmTLRP2DwHjdF5s7jo9tuztdQxAhINCdvS+3nGINqPd00AphqJR/0LhANUS6/+7SCb98YOfA=="],
@@ -1560,8 +1523,6 @@
"roarr/sprintf-js": ["sprintf-js@1.1.3", "", {}, "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA=="],
"rss-parser/entities": ["entities@2.2.0", "", {}, "sha512-p92if5Nz619I0w+akJrLZH0MX0Pb5DX39XOwQTtXSdQQOaYH03S1uIQp4mhOZtAXrxq4ViO67YTiLBo2638o9A=="],
"slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="],
"string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
@@ -1570,8 +1531,6 @@
"wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="],
"xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="],
"@babel/helper-compilation-targets/lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="],
"@huggingface/transformers/onnxruntime-node/global-agent": ["global-agent@3.0.0", "", { "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", "matcher": "^3.0.0", "roarr": "^2.15.3", "semver": "^7.3.2", "serialize-error": "^7.0.1" } }, "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q=="],
+4 -2
View File
@@ -60,14 +60,16 @@
"diff": "^9.0.0",
"fflate": "0.8.3",
"fastembed": "2.1.0",
"fast-xml-parser": "^5.9.0",
"ghostty-web": "^0.4.0",
"handlebars": "^4.7.9",
"linkedom": "^0.18.12",
"lint-staged": "^17.0.7",
"lru-cache": "11.5.1",
"lucide-react": "^1.17.0",
"mammoth": "^1.12.0",
"marked": "^18.0.5",
"markit-ai": "0.5.3",
"mupdf": "^1.27.0",
"onnxruntime-node": "1.26.0",
"partial-json": "^0.1.7",
"postcss": "^8.5.15",
@@ -169,7 +171,7 @@
"lint:py": "ruff check python && ruff format --check python",
"fix:py": "ruff check --fix python && ruff format python",
"prepublishOnly": "bun run check",
"prepare": "bun run generate-docs-index && bun run build-tool-views",
"prepare": "bun run build-tool-views",
"publish": "bun run prepublishOnly && npm publish -ws --access public",
"publish:dry": "bun run prepublishOnly && npm publish -ws --access public --dry-run",
"release": "bun scripts/release.ts",
+9
View File
@@ -1,14 +1,23 @@
# Changelog
## [Unreleased]
### Changed
- Updated internal image processing to no longer include metadata text for fetched images
- Optimized `omp://` documentation indexing by compressing doc bodies into a lazily-inflated blob
- Changed Mermaid fenced-block ASCII rendering to use the first-party vendored renderer in `@oh-my-pi/pi-utils` (`src/vendor/mermaid-ascii`), dropping the `beautiful-mermaid` npm package, its transitive `elkjs` (~3.13MB), and the `beautiful-mermaid` `bun patch`; CJK/emoji width handling and the layout-direction override are preserved.
- Changed `omp://` documentation embedding to a gzipped base64 index (`docs-index.generated.txt`, populated at build time and reset afterward) inflated on first read, instead of a ~1.6MB raw TypeScript map. The compiled binary / npm bundle drops ~0.9MB; the dev tree and source checkouts read `docs/` from disk.
- Replaced the `markit-ai` package with a vendored in-house document engine (`src/markit`) for the document formats it converts: PDF (via `mupdf`), DOCX (via `mammoth`), and PPTX/XLSX/EPUB (via `fast-xml-parser`). Conversion output is preserved, including the PDF column/table-detection pipeline and HTML-table normalization; `markit-ai`'s unused converters and their dependency tail are gone. Legacy `.doc`/`.ppt`/`.xls`/`.rtf` remain unsupported (a conversion error), as before.
- Centralized ZIP handling behind a single `src/utils/zip.ts` (`fflate`): the new document converters, the `write` tool's in-place archive editing, and the `read` tool's ranged archive reader now share one ZIP implementation instead of mixing `jszip` and `fflate`.
### Removed
- Removed the `markit-ai`, `exifr`, `music-metadata`, and direct `jszip` dependencies. As a side effect, fetching an image or audio URL no longer appends EXIF/audio metadata text; image inlining and resizing are unchanged, and document conversion (PDF/DOCX/PPTX/XLSX/EPUB) is unaffected.
### Fixed
- Fixed auto-retry after transient model-stream socket closes to replay text/thinking-only partial assistant output, including turns where incomplete tool-call arguments were dropped; completed tool calls still block retry to avoid duplicating tool execution.
- Fixed `omp update` reporting `EPERM: operation not permitted, unlink '<binary>.bak'` on Windows when self-replacing a standalone binary, even though the new binary had already been installed. The backed-up old executable is still the running process image and cannot be unlinked until the process exits, so the post-verify backup cleanup is now best-effort, backups use a unique per-attempt name, and stale backups are swept on the next update ([#845](https://github.com/can1357/oh-my-pi/issues/845)).
- Fixed plan-mode `Refine plan` so the internal approval abort is hidden and the editor is ready for a follow-up prompt instead of showing `Operation aborted` ([#2971](https://github.com/can1357/oh-my-pi/issues/2971)).
- Fixed TUI prompts beginning with shell-style variables such as `$HOME` being misrouted to Python eval; Python shortcuts now require `$ <code>` or `$$ <code>`. ([#2944](https://github.com/can1357/oh-my-pi/issues/2944))
+4 -2
View File
@@ -35,7 +35,7 @@
"check": "biome check . && bun run check:types",
"check:types": "tsgo -p tsconfig.json --noEmit",
"lint": "biome lint .",
"test": "bun test --parallel=4",
"test": "bun test --parallel=4 test src",
"fix": "biome check --write --unsafe . && bun run format-prompts",
"fmt": "biome format --write . && bun run format-prompts",
"format-prompts": "bun scripts/format-prompts.ts",
@@ -71,11 +71,13 @@
"arktype": "catalog:",
"chalk": "catalog:",
"diff": "catalog:",
"fast-xml-parser": "catalog:",
"fflate": "catalog:",
"handlebars": "catalog:",
"linkedom": "catalog:",
"lru-cache": "catalog:",
"markit-ai": "catalog:",
"mammoth": "catalog:",
"mupdf": "catalog:",
"puppeteer-core": "catalog:",
"turndown": "catalog:",
"turndown-plugin-gfm": "catalog:",
+32
View File
@@ -0,0 +1,32 @@
This directory contains an in-house document-to-markdown engine adapted from
markit-ai (https://github.com/Michaelliv/markit), used under the MIT License.
Copyright (c) 2026 Michael Liv
Only the converters for the document formats omp supports are ported (pdf,
docx, pptx, xlsx, epub); the CLI, plugin/provider, and unused converters
(html, image, audio, plain-text, rss, github, wikipedia, csv, json, yaml,
ipynb, iwork, zip, xml) were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and
`.rtf` are routed by the read/fetch tools but have no converter — they surface
a conversion error, exactly as upstream markit did. Logic is ported faithfully
so conversion output matches the upstream package.
MIT License
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
@@ -0,0 +1,56 @@
// Adapted from markit-ai (MIT). See ../NOTICE.
import * as path from "node:path";
import mammoth from "mammoth";
import { createTurndown, normalizeTablesHtml } from "../../utils/turndown";
import type { ConversionResult, Converter, StreamInfo } from "../types";
const EXTENSIONS = [".docx"];
const MIMETYPES = ["application/vnd.openxmlformats-officedocument.wordprocessingml.document"];
export class DocxConverter implements Converter {
name = "docx";
accepts(streamInfo: StreamInfo): boolean {
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) {
return true;
}
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) {
return true;
}
return false;
}
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
const imageDir = streamInfo.imageDir;
let imageCount = 0;
const convertImage = imageDir
? mammoth.images.imgElement(image => {
imageCount++;
const ext = (image.contentType?.split("/")[1] || "png").replace("jpeg", "jpg");
const filename = `image_${imageCount}.${ext}`;
const filepath = path.join(imageDir, filename);
return image.read("base64").then(async base64 => {
await Bun.write(filepath, Buffer.from(base64, "base64"));
return { src: filepath, alt: `image_${imageCount}` };
});
})
: mammoth.images.imgElement(image => {
imageCount++;
const contentType = image.contentType || "image/png";
return image.read("base64").then(base64 => {
return {
src: `data:${contentType};base64,${base64.slice(0, 0)}`,
alt: `image_${imageCount}`,
};
});
});
const { value: html } = await mammoth.convertToHtml({ buffer: input }, { convertImage });
const turndown = createTurndown();
let markdown = turndown.turndown(normalizeTablesHtml(html));
// Replace data URI images with comment placeholders when no imageDir
if (!imageDir) {
markdown = markdown.replace(/!\[([^\]]*)\]\(data:[^)]*\)/g, "<!-- image: $1 -->");
}
return { markdown: markdown.trim() };
}
}
@@ -0,0 +1,136 @@
// Adapted from markit-ai (MIT). See ../NOTICE.
import { XMLParser } from "fast-xml-parser";
import { createTurndown, normalizeTablesHtml } from "../../utils/turndown";
import { unzip, unzipText } from "../../utils/zip";
import type { ConversionResult, Converter, StreamInfo } from "../types";
const EXTENSIONS = [".epub"];
const MIMETYPES = ["application/epub", "application/epub+zip", "application/x-epub+zip"];
/** A metadata value: a bare string, or a node carrying `#text` and/or array children. */
type MetaValue = string | MetaNode;
interface MetaNode {
"#text"?: string;
[index: number]: MetaValue;
}
interface Metadata {
"dc:title"?: MetaValue;
"dc:creator"?: MetaValue;
"dc:language"?: MetaValue;
"dc:publisher"?: MetaValue;
"dc:date"?: MetaValue;
"dc:description"?: MetaValue;
}
interface ManifestItem {
"@_id": string;
"@_href": string;
}
interface SpineItem {
"@_idref": string;
}
interface OpfDoc {
package?: {
metadata?: Metadata;
manifest?: { item?: ManifestItem | ManifestItem[] };
spine?: { itemref?: SpineItem | SpineItem[] };
};
}
interface Rootfile {
"@_full-path": string;
}
interface ContainerDoc {
container?: { rootfiles?: { rootfile?: Rootfile | Rootfile[] } };
}
export class EpubConverter implements Converter {
name = "epub";
accepts(streamInfo: StreamInfo): boolean {
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true;
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true;
return false;
}
async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
const entries = unzip(input);
const parser = new XMLParser({
ignoreAttributes: false,
attributeNamePrefix: "@_",
textNodeName: "#text",
processEntities: { maxTotalExpansions: 1_000_000 },
});
// Find content.opf path from container.xml
const containerXml = unzipText(entries, "META-INF/container.xml");
if (!containerXml) throw new Error("Invalid EPUB: missing container.xml");
const container = parser.parse(containerXml) as ContainerDoc;
const rootfile = container.container?.rootfiles?.rootfile;
const opfPath = Array.isArray(rootfile) ? rootfile[0]["@_full-path"] : rootfile?.["@_full-path"];
if (!opfPath) throw new Error("Invalid EPUB: missing rootfile path");
// Parse content.opf
const opfXml = unzipText(entries, opfPath);
if (!opfXml) throw new Error("Invalid EPUB: missing content.opf");
const opf = parser.parse(opfXml) as OpfDoc;
// Extract metadata
const meta: Metadata = opf.package?.metadata ?? {};
const metadata: Record<string, string | undefined> = {
title: this.getText(meta["dc:title"]),
authors: this.getTextArray(meta["dc:creator"]).join(", ") || undefined,
language: this.getText(meta["dc:language"]),
publisher: this.getText(meta["dc:publisher"]),
date: this.getText(meta["dc:date"]),
description: this.getText(meta["dc:description"]),
};
// Build manifest map (id → href)
const manifestItems = opf.package?.manifest?.item;
const itemList = Array.isArray(manifestItems) ? manifestItems : manifestItems ? [manifestItems] : [];
const manifest = new Map<string, string>();
for (const item of itemList) {
manifest.set(item["@_id"], item["@_href"]);
}
// Get spine order
const spineItems = opf.package?.spine?.itemref;
const spineList = Array.isArray(spineItems) ? spineItems : spineItems ? [spineItems] : [];
const spineOrder = spineList.map(s => s["@_idref"]);
// Resolve file paths
const basePath = opfPath.includes("/") ? opfPath.substring(0, opfPath.lastIndexOf("/")) : "";
const turndown = createTurndown();
const sections: string[] = [];
// Add metadata header
const metaLines: string[] = [];
for (const key in metadata) {
const value = metadata[key];
if (value) metaLines.push(`**${key.charAt(0).toUpperCase() + key.slice(1)}:** ${value}`);
}
if (metaLines.length > 0) sections.push(metaLines.join("\n"));
// Convert spine files
for (const idref of spineOrder) {
const href = manifest.get(idref);
if (!href) continue;
const filePath = basePath ? `${basePath}/${href}` : href;
const html = unzipText(entries, filePath);
if (!html) continue;
// Strip script/style, convert to markdown
const cleaned = html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "");
const md = turndown.turndown(normalizeTablesHtml(cleaned)).trim();
if (md) sections.push(md);
}
return {
markdown: sections.join("\n\n").trim(),
title: metadata.title,
};
}
getText(node: MetaValue | undefined): string | undefined {
if (!node) return undefined;
if (typeof node === "string") return node;
if (node["#text"]) return String(node["#text"]);
if (Array.isArray(node)) return this.getText(node[0]);
return undefined;
}
getTextArray(node: MetaValue | undefined): (string | undefined)[] {
if (!node) return [];
const list = Array.isArray(node) ? node : [node];
return list.map(n => this.getText(n)).filter(Boolean);
}
}
@@ -0,0 +1,24 @@
// Minimal ambient types for `mammoth` (ships no types). Declares only what
// DocxConverter uses. See ../NOTICE.
declare module "mammoth" {
interface MammothImage {
contentType?: string;
read(encoding: "base64"): Promise<string>;
}
interface ImgAttributes {
src: string;
alt?: string;
}
type ConvertImageHandler = (image: MammothImage) => Promise<ImgAttributes>;
interface ConvertOptions {
convertImage?: ConvertImageHandler;
}
interface ConvertResult {
value: string;
messages: unknown[];
}
export const images: { imgElement(fn: ConvertImageHandler): ConvertImageHandler };
export function convertToHtml(input: { buffer: Buffer }, options?: ConvertOptions): Promise<ConvertResult>;
const _default: { convertToHtml: typeof convertToHtml; images: typeof images };
export default _default;
}
@@ -0,0 +1,103 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Multi-column layout detection and text box reordering.
*
* Many PDFs (legal documents, datasheets, academic papers) use two-column
* layouts. Without column detection, text boxes are ordered by Y position
* only, interleaving left and right column content.
*
* Algorithm:
* 1. Collect left edges of all text boxes on the page
* 2. Find the largest horizontal gap between consecutive left edges
* 3. If gap > MIN_GAP_RATIO of the text width and both sides have
* enough boxes → multi-column detected
* 4. Assign each text box to a column based on its center X
* 5. Return columns in reading order (left-to-right, top-to-bottom)
*
* This only detects the column structure. The caller is responsible for
* processing each column's text boxes independently (table detection,
* rendering, etc.).
*/
import type { TextBox } from "./types";
export interface ColumnLayout {
/** Number of columns detected (1 = single column, 2+ = multi-column). */
columnCount: number;
/** Text boxes grouped by column, in reading order (left to right). */
columns: TextBox[][];
/** X positions of column boundaries (between columns). */
boundaries: number[];
}
/**
* Minimum gap as a fraction of the total text width to consider a column
* boundary. A two-column layout typically has ~50% gap; we use a lower
* threshold to catch asymmetric columns.
*/
const MIN_GAP_RATIO = 0.15;
/** Minimum number of text boxes on each side of the gap. */
const MIN_BOXES_PER_COLUMN = 4;
/** Minimum gap in absolute points to avoid splitting on small whitespace. */
const MIN_GAP_PTS = 40;
/**
* Detect column layout and return text boxes grouped by column.
*
* For single-column pages, returns all boxes in one group.
* For multi-column pages, returns boxes split by column in reading order.
*/
export function detectColumns(textBoxes: TextBox[]): ColumnLayout {
if (textBoxes.length < MIN_BOXES_PER_COLUMN * 2) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Collect unique left edges (rounded to avoid float noise)
const lefts = [...new Set(textBoxes.map(tb => Math.round(tb.bounds.left)))].sort((a, b) => a - b);
if (lefts.length < 2) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
const textXMin = lefts[0];
const textXMax = Math.max(...textBoxes.map(tb => Math.round(tb.bounds.right)));
const textWidth = textXMax - textXMin;
if (textWidth <= 0) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Find the largest gap between consecutive left-edge positions
let maxGap = 0;
let gapLeft = 0;
let gapRight = 0;
for (let i = 1; i < lefts.length; i++) {
const gap = lefts[i] - lefts[i - 1];
if (gap > maxGap) {
maxGap = gap;
gapLeft = lefts[i - 1];
gapRight = lefts[i];
}
}
const gapRatio = maxGap / textWidth;
if (gapRatio < MIN_GAP_RATIO || maxGap < MIN_GAP_PTS) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
// Split point is the midpoint of the gap
const splitX = (gapLeft + gapRight) / 2;
// Assign boxes to columns based on center X
const leftCol: TextBox[] = [];
const rightCol: TextBox[] = [];
for (const tb of textBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
if (cx < splitX) {
leftCol.push(tb);
} else {
rightCol.push(tb);
}
}
// Validate both columns have enough content
if (leftCol.length < MIN_BOXES_PER_COLUMN || rightCol.length < MIN_BOXES_PER_COLUMN) {
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
}
return {
columnCount: 2,
columns: [leftCol, rightCol],
boundaries: [splitX],
};
}
@@ -0,0 +1,557 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* PDF content extraction using mupdf.
*
* Extracts text boxes (with position, font size, bold) and vector line
* segments (table borders) from each page. Uses mupdf's native WASM
* engine for fast parsing, and reads raw content streams for vector graphics.
*
* Coordinate system: PDF native (origin = bottom-left, Y increases upward).
*/
import * as mupdf from "mupdf";
import type { ImageRegion, PageContent, Segment, TextBox } from "./types";
/** mupdf structured-text JSON bounding box (top-left origin). */
interface StextBBox {
x: number;
y: number;
w: number;
h: number;
}
/** Font metadata attached to a structured-text line. */
interface StextFont {
size?: number;
weight?: string;
name?: string;
}
/** A line within a text block in mupdf structured-text JSON. */
interface StextLine {
text?: string;
font?: StextFont;
bbox: StextBBox;
}
/** A block (text or image) in mupdf structured-text JSON. */
interface StextBlock {
type: string;
bbox: StextBBox;
lines: StextLine[];
}
/** Parsed mupdf structured-text JSON for a page. */
interface StructuredTextJSON {
blocks: StextBlock[];
}
/** A raw text fragment before merging into word/phrase boxes. */
interface RawTextItem {
text: string;
x: number;
y: number;
width: number;
height: number;
fontSize: number;
isBold: boolean;
}
// ---------------------------------------------------------------------------
// Text extraction
// ---------------------------------------------------------------------------
/** Y tolerance for merging text fragments on the same visual line. */
const SAME_LINE_Y_TOLERANCE = 2;
/** Max horizontal gap (pts) to merge adjacent fragments into one text box. */
const MAX_MERGE_GAP = 14;
/**
* Merge horizontally adjacent raw text items on the same visual line into
* word/phrase-level text boxes.
*/
function mergeIntoWords(raws: RawTextItem[]): RawTextItem[] {
if (raws.length === 0) return [];
// Sort by Y descending (top-first in bottom-left coords), then X ascending
const sorted = [...raws].sort((a, b) => {
const dy = b.y - a.y;
return Math.abs(dy) > SAME_LINE_Y_TOLERANCE ? dy : a.x - b.x;
});
const merged: RawTextItem[] = [];
let cur = { ...sorted[0] };
for (let i = 1; i < sorted.length; i++) {
const next = sorted[i];
const sameY = Math.abs(next.y - cur.y) <= SAME_LINE_Y_TOLERANCE;
const close = next.x <= cur.x + cur.width + MAX_MERGE_GAP;
if (sameY && close) {
const gap = next.x - (cur.x + cur.width);
const sep = gap > 1 ? " " : "";
cur.text += sep + next.text;
cur.width = next.x + next.width - cur.x;
cur.height = Math.max(cur.height, next.height);
cur.fontSize = Math.max(cur.fontSize, next.fontSize);
cur.isBold = cur.isBold || next.isBold;
} else {
merged.push(cur);
cur = { ...next };
}
}
merged.push(cur);
return merged;
}
/**
* Extract text boxes from a mupdf page using structured text output.
*
* mupdf's structured text JSON uses top-left origin; we convert to
* bottom-left (standard PDF coordinates) using the page height.
*/
function extractTextBoxes(
page: mupdf.Page,
pageNumber: number,
pageHeight: number,
stext?: StructuredTextJSON,
): TextBox[] {
if (!stext) {
stext = JSON.parse(page.toStructuredText("preserve-whitespace").asJSON()) as StructuredTextJSON;
}
const raws: RawTextItem[] = [];
for (const block of stext.blocks) {
if (block.type !== "text") continue;
for (const line of block.lines) {
const text = line.text?.trim();
if (!text) continue;
const fontSize = line.font?.size ?? 0;
const weight = line.font?.weight ?? "normal";
const fontName = line.font?.name ?? "";
const isBold = weight === "bold" || /bold/i.test(fontName) || /Black|Heavy/i.test(fontName);
// mupdf bbox: {x, y, w, h} in top-left coords
// Convert to bottom-left: pdfY = pageHeight - (bbox.y + bbox.h)
const bboxY = line.bbox.y;
const bboxH = line.bbox.h;
const pdfY = pageHeight - (bboxY + bboxH);
raws.push({
text,
x: line.bbox.x,
y: pdfY,
width: line.bbox.w,
height: bboxH,
fontSize,
isBold,
});
}
}
const words = mergeIntoWords(raws);
return words
.map((w, i) => ({
id: `p${pageNumber}-t${i}`,
text: w.text.trim(),
pageNumber,
fontSize: w.fontSize,
isBold: w.isBold,
bounds: {
left: w.x,
right: w.x + w.width,
bottom: w.y,
top: w.y + w.height,
},
}))
.filter(b => b.text.length > 0);
}
// ---------------------------------------------------------------------------
// Vector segment extraction from raw content stream
// ---------------------------------------------------------------------------
/** Minimum aspect ratio for a filled rect to be considered a line. */
const LINE_ASPECT_THRESHOLD = 6;
/** Minimum length (pts) for a segment to count. */
const MIN_LENGTH = 2;
/** Maximum thickness (pts) for a border line (filters out filled areas). */
const MAX_THICKNESS = 3;
/**
* Convert a thin filled rectangle to a horizontal or vertical segment.
* Returns null if the rect doesn't look like a border line.
*/
function thinRectToSegment(id: string, x: number, y: number, w: number, h: number): Segment | null {
const aw = Math.abs(w);
const ah = Math.abs(h);
if (aw > ah * LINE_ASPECT_THRESHOLD && aw >= MIN_LENGTH && ah <= MAX_THICKNESS) {
// Horizontal line
const cy = y + ah / 2;
return { id, x1: x, y1: cy, x2: x + aw, y2: cy };
}
if (ah > aw * LINE_ASPECT_THRESHOLD && ah >= MIN_LENGTH && aw <= MAX_THICKNESS) {
// Vertical line
const cx = x + aw / 2;
return { id, x1: cx, y1: y, x2: cx, y2: y + ah };
}
return null;
}
/**
* Emit 4 edge segments from a stroked rectangle.
*/
function pushStrokedRectEdges(segments: Segment[], id: string, x: number, y: number, w: number, h: number): void {
const aw = Math.abs(w);
const ah = Math.abs(h);
const base = id;
if (aw >= MIN_LENGTH) {
segments.push({ id: `${base}-b`, x1: x, y1: y, x2: x + aw, y2: y });
segments.push({
id: `${base}-t`,
x1: x,
y1: y + ah,
x2: x + aw,
y2: y + ah,
});
}
if (ah >= MIN_LENGTH) {
segments.push({ id: `${base}-l`, x1: x, y1: y, x2: x, y2: y + ah });
segments.push({
id: `${base}-r`,
x1: x + aw,
y1: y,
x2: x + aw,
y2: y + ah,
});
}
}
const CTM_IDENTITY = [1, 0, 0, 1, 0, 0];
/** Concatenate two affine matrices: result = parent × child. */
function ctmConcat(p: number[], c: number[]): number[] {
return [
p[0] * c[0] + p[2] * c[1],
p[1] * c[0] + p[3] * c[1],
p[0] * c[2] + p[2] * c[3],
p[1] * c[2] + p[3] * c[3],
p[0] * c[4] + p[2] * c[5] + p[4],
p[1] * c[4] + p[3] * c[5] + p[5],
];
}
function ctmApply(m: number[], x: number, y: number): [number, number] {
return [m[0] * x + m[2] * y + m[4], m[1] * x + m[3] * y + m[5]];
}
// ---------------------------------------------------------------------------
// Content stream parsing
// ---------------------------------------------------------------------------
/**
* Parse a PDF content stream and extract line segments from thin filled
* rectangles (re+f), stroked rectangles (re+S), and explicit lines (m/l+S).
* Tracks the CTM via q/Q/cm operators so coordinates are in page space.
*/
function extractSegmentsFromContentStream(raw: string, pageNumber: number): Segment[] {
const segments: Segment[] = [];
const tokens = tokenizeContentStream(raw);
let idx = 0;
let strokeWidth = 1.0;
// Graphics state stack (q/Q): saves CTM + strokeWidth
let ctm = [...CTM_IDENTITY];
const stateStack: Array<{ ctm: number[]; strokeWidth: number }> = [];
// State for path building (in user coordinates, pre-CTM)
let curX = 0;
let curY = 0;
let pathStartX = 0;
let pathStartY = 0;
const pendingRects: Array<{ x: number; y: number; w: number; h: number }> = [];
const pendingLines: Array<{ x1: number; y1: number; x2: number; y2: number }> = [];
function flushPath(mode: "fill" | "stroke"): void {
const sid = () => `p${pageNumber}-s${segments.length}`;
if (mode === "fill") {
for (const r of pendingRects) {
// Transform the rect corners through CTM, then check if it's a thin line
const [x0, y0] = ctmApply(ctm, r.x, r.y);
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
const seg = thinRectToSegment(
sid(),
Math.min(x0, x1),
Math.min(y0, y1),
Math.abs(x1 - x0),
Math.abs(y1 - y0),
);
if (seg) segments.push(seg);
}
} else if (mode === "stroke" && strokeWidth <= MAX_THICKNESS) {
for (const r of pendingRects) {
const [x0, y0] = ctmApply(ctm, r.x, r.y);
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
pushStrokedRectEdges(
segments,
sid(),
Math.min(x0, x1),
Math.min(y0, y1),
Math.abs(x1 - x0),
Math.abs(y1 - y0),
);
}
for (const l of pendingLines) {
const [lx1, ly1] = ctmApply(ctm, l.x1, l.y1);
const [lx2, ly2] = ctmApply(ctm, l.x2, l.y2);
const dx = Math.abs(lx2 - lx1);
const dy = Math.abs(ly2 - ly1);
// Only keep H/V lines
if ((dx >= MIN_LENGTH && dy < 1) || (dy >= MIN_LENGTH && dx < 1)) {
segments.push({ id: sid(), x1: lx1, y1: ly1, x2: lx2, y2: ly2 });
}
}
}
pendingRects.length = 0;
pendingLines.length = 0;
}
while (idx < tokens.length) {
const t = tokens[idx];
if (t === "q") {
stateStack.push({ ctm: [...ctm], strokeWidth });
} else if (t === "Q") {
const saved = stateStack.pop();
if (saved) {
ctm = saved.ctm;
strokeWidth = saved.strokeWidth;
}
} else if (t === "cm" && idx >= 6) {
const a = Number(tokens[idx - 6]);
const b = Number(tokens[idx - 5]);
const c = Number(tokens[idx - 4]);
const d = Number(tokens[idx - 3]);
const e = Number(tokens[idx - 2]);
const f = Number(tokens[idx - 1]);
ctm = ctmConcat(ctm, [a, b, c, d, e, f]);
} else if (t === "w" && idx >= 1) {
strokeWidth = Number(tokens[idx - 1]) || strokeWidth;
} else if (t === "re" && idx >= 4) {
const x = Number(tokens[idx - 4]);
const y = Number(tokens[idx - 3]);
const w = Number(tokens[idx - 2]);
const h = Number(tokens[idx - 1]);
if (Number.isFinite(x + y + w + h)) {
pendingRects.push({ x, y, w, h });
}
} else if (t === "m" && idx >= 2) {
curX = Number(tokens[idx - 2]);
curY = Number(tokens[idx - 1]);
pathStartX = curX;
pathStartY = curY;
} else if (t === "l" && idx >= 2) {
const x2 = Number(tokens[idx - 2]);
const y2 = Number(tokens[idx - 1]);
pendingLines.push({ x1: curX, y1: curY, x2, y2 });
curX = x2;
curY = y2;
} else if (t === "h") {
// closePath: line back to start
if (curX !== pathStartX || curY !== pathStartY) {
pendingLines.push({
x1: curX,
y1: curY,
x2: pathStartX,
y2: pathStartY,
});
}
curX = pathStartX;
curY = pathStartY;
} else if (t === "f" || t === "F" || t === "f*") {
flushPath("fill");
} else if (t === "S" || t === "s") {
if (t === "s") {
// closeStroke: implicit closePath
if (curX !== pathStartX || curY !== pathStartY) {
pendingLines.push({
x1: curX,
y1: curY,
x2: pathStartX,
y2: pathStartY,
});
}
}
flushPath("stroke");
} else if (t === "B" || t === "B*" || t === "b" || t === "b*") {
// fill + stroke combined
flushPath("fill");
flushPath("stroke");
} else if (t === "n") {
// end path without painting — discard
pendingRects.length = 0;
pendingLines.length = 0;
}
idx++;
}
return segments;
}
/**
* Fast tokenizer for PDF content streams.
* Splits on whitespace, skipping comments and string literals.
*/
function tokenizeContentStream(raw: string): string[] {
const tokens: string[] = [];
const len = raw.length;
let i = 0;
while (i < len) {
const ch = raw.charCodeAt(i);
// Skip whitespace
if (ch <= 32) {
i++;
continue;
}
// Skip comments
if (ch === 37 /* % */) {
while (i < len && raw.charCodeAt(i) !== 10) i++;
continue;
}
// Skip string literals (...)
if (ch === 40 /* ( */) {
let depth = 1;
i++;
while (i < len && depth > 0) {
const c = raw.charCodeAt(i);
if (c === 92 /* \ */) {
i++;
} else if (c === 40) {
depth++;
} else if (c === 41) {
depth--;
}
i++;
}
continue;
}
// Skip hex strings <...>
if (ch === 60 /* < */ && i + 1 < len && raw.charCodeAt(i + 1) !== 60) {
i++;
while (i < len && raw.charCodeAt(i) !== 62) i++;
i++; // skip >
continue;
}
// Skip dict delimiters << >>
if (ch === 60 && i + 1 < len && raw.charCodeAt(i + 1) === 60) {
i += 2;
continue;
}
if (ch === 62 && i + 1 < len && raw.charCodeAt(i + 1) === 62) {
i += 2;
continue;
}
// Regular token: read until whitespace or delimiter
const start = i;
while (i < len) {
const c = raw.charCodeAt(i);
if (c <= 32 || c === 40 || c === 41 || c === 60 || c === 62 || c === 37) break;
i++;
}
if (i > start) {
tokens.push(raw.substring(start, i));
}
}
return tokens;
}
// ---------------------------------------------------------------------------
// Image region detection
// ---------------------------------------------------------------------------
/** Minimum area (pts²) for an image to be considered a diagram, not an icon. */
const MIN_IMAGE_AREA = 5000;
function extractImageRegions(stext: StructuredTextJSON, pageNumber: number, pageHeight: number): ImageRegion[] {
const regions: ImageRegion[] = [];
for (const block of stext.blocks) {
if (block.type !== "image") continue;
const { x, y, w, h } = block.bbox;
if (w * h < MIN_IMAGE_AREA) continue; // skip tiny icons
// Convert Y from mupdf (top-left) to PDF (bottom-left) for ordering
const pdfTopY = pageHeight - y;
regions.push({
id: `p${pageNumber}-img${regions.length}`,
pageNumber,
bbox: { x, y, w, h },
topY: pdfTopY,
});
}
return regions;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/**
* Render an image region from a PDF page as a PNG buffer.
* Uses mupdf's DrawDevice to render just the cropped area at 2x resolution.
*/
export function renderImageRegion(input: Uint8Array, region: ImageRegion): Uint8Array {
const doc = mupdf.Document.openDocument(input, "application/pdf");
const page = doc.loadPage(region.pageNumber - 1);
const pad = 10;
const bx = region.bbox.x - pad;
const by = region.bbox.y - pad;
const bw = region.bbox.w + 2 * pad;
const bh = region.bbox.h + 2 * pad;
const scale = 2;
const pw = Math.round(bw * scale);
const ph = Math.round(bh * scale);
const pix = new mupdf.Pixmap(mupdf.ColorSpace.DeviceRGB, [0, 0, pw, ph], false);
pix.clear(255);
const matrix: mupdf.Matrix = [scale, 0, 0, scale, -bx * scale, -by * scale];
const dl = page.toDisplayList();
const dev = new mupdf.DrawDevice(matrix, pix);
dl.run(dev, mupdf.Matrix.identity);
dev.close();
return pix.asPNG();
}
/**
* Extract text boxes and vector segments from all pages of a PDF buffer.
*/
export async function extractPages(input: Uint8Array): Promise<PageContent[]> {
const doc = mupdf.Document.openDocument(input, "application/pdf");
const pages: PageContent[] = [];
for (let i = 0; i < doc.countPages(); i++) {
const pageNumber = i + 1;
const page = doc.loadPage(i);
const bounds = page.getBounds();
const pageHeight = bounds[3] - bounds[1];
// Single structured text pass with both flags
const stext = JSON.parse(
page.toStructuredText("preserve-whitespace,preserve-images").asJSON(),
) as StructuredTextJSON;
// Extract text boxes and image regions from the same parse
const textBoxes = extractTextBoxes(page, pageNumber, pageHeight, stext);
const images = extractImageRegions(stext, pageNumber, pageHeight);
// Extract vector segments from raw content stream
let segments: Segment[] = [];
try {
const pageObj = (page as mupdf.PDFPage).getObject();
const contents = pageObj.get("Contents");
if (contents) {
let rawBytes: Uint8Array;
if (contents.isArray()) {
// Multiple content streams — concatenate
const parts: Uint8Array[] = [];
const len = contents.length ?? 0;
for (let j = 0; j < len; j++) {
const stream = contents.get(j);
if (stream?.readStream) {
parts.push(stream.readStream().asUint8Array());
}
}
const totalLen = parts.reduce((s, p) => s + p.length, 0);
rawBytes = new Uint8Array(totalLen);
let offset = 0;
for (const part of parts) {
rawBytes.set(part, offset);
offset += part.length;
}
} else {
rawBytes = contents.readStream().asUint8Array();
}
const raw = new TextDecoder().decode(rawBytes);
segments = extractSegmentsFromContentStream(raw, pageNumber);
}
} catch {
// Content stream extraction failed — proceed with text only
}
pages.push({ pageNumber, textBoxes, segments, images });
}
return pages;
}
@@ -0,0 +1,780 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Table grid detection from vector segments and text boxes.
*
* Ported from @oharato/pdf2md-ts with TypeScript types and without
* CJK-specific borderless table heuristics. The core algorithm:
*
* 1. Classify segments as horizontal or vertical lines
* 2. Group horizontal Y-lines into table groups (split by vertical gaps)
* 3. For each group:
* a. Full grid (H+V lines): build cells from grid intersections,
* place text via raycasting
* b. H-line only (no V lines): infer columns from text X positions
* 4. Prune empty rows/cols
*
* Coordinate system: PDF native (bottom-left origin, Y increases upward).
*/
import type { Segment, TableCell, TableGrid, TextBox } from "./types";
export interface GridResult {
grids: TableGrid[];
consumedIds: string[];
}
type RayDirection = "up" | "down" | "left" | "right";
interface Ray {
direction: RayDirection;
segmentId: string | null;
distance: number;
}
interface Interval {
min: number;
max: number;
}
function castRaysForTextBox(textBox: TextBox, segments: Segment[]): Ray[] {
const cx = (textBox.bounds.left + textBox.bounds.right) / 2;
const cy = (textBox.bounds.top + textBox.bounds.bottom) / 2;
let up: Ray = { direction: "up", segmentId: null, distance: Infinity };
let down: Ray = { direction: "down", segmentId: null, distance: Infinity };
let left: Ray = { direction: "left", segmentId: null, distance: Infinity };
let right: Ray = {
direction: "right",
segmentId: null,
distance: Infinity,
};
for (const seg of segments) {
const isH = Math.abs(seg.y1 - seg.y2) < 0.5;
const isV = Math.abs(seg.x1 - seg.x2) < 0.5;
if (isH) {
const minX = Math.min(seg.x1, seg.x2);
const maxX = Math.max(seg.x1, seg.x2);
if (cx >= minX && cx <= maxX) {
const d = seg.y1 - cy;
if (d >= 0 && d < up.distance) up = { direction: "up", segmentId: seg.id, distance: d };
const dd = cy - seg.y1;
if (dd >= 0 && dd < down.distance) down = { direction: "down", segmentId: seg.id, distance: dd };
}
}
if (isV) {
const minY = Math.min(seg.y1, seg.y2);
const maxY = Math.max(seg.y1, seg.y2);
if (cy >= minY && cy <= maxY) {
const d = cx - seg.x1;
if (d >= 0 && d < left.distance) left = { direction: "left", segmentId: seg.id, distance: d };
const rd = seg.x1 - cx;
if (rd >= 0 && rd < right.distance) right = { direction: "right", segmentId: seg.id, distance: rd };
}
}
}
return [up, down, left, right];
}
// ---------------------------------------------------------------------------
// Utility
// ---------------------------------------------------------------------------
const AXIS_EPSILON = 0.8;
const PAGE_MARGIN = 20;
function uniqueSorted(values: number[]): number[] {
const sorted = [...values].sort((a, b) => a - b);
const result: number[] = [];
for (const v of sorted) {
if (result.length === 0 || Math.abs(result[result.length - 1] - v) > 1) result.push(v);
}
return result;
}
// ---------------------------------------------------------------------------
// Y-line group splitting
// ---------------------------------------------------------------------------
function chainCoversRange(intervals: Interval[], lowerY: number, upperY: number, eps: number): boolean {
const sorted = [...intervals].sort((a, b) => a.min - b.min);
let covered = lowerY;
for (const iv of sorted) {
if (iv.min > covered + eps) break;
if (iv.max > covered) covered = iv.max;
if (covered >= upperY - eps) return true;
}
return false;
}
function countBridgingVLineCols(upperY: number, lowerY: number, verticals: Segment[]): number {
const eps = 1.5;
const byX = new Map<number, Interval[]>();
for (const seg of verticals) {
const rx = Math.round(seg.x1);
if (!byX.has(rx)) byX.set(rx, []);
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
}
let count = 0;
for (const intervals of byX.values()) {
if (chainCoversRange(intervals, lowerY, upperY, eps)) count++;
}
return count;
}
function bridgingXSet(upperY: number, lowerY: number, verticals: Segment[]): Set<number> {
const eps = 1.5;
const xs = new Set<number>();
const byX = new Map<number, Interval[]>();
for (const seg of verticals) {
const rx = Math.round(seg.x1);
if (!byX.has(rx)) byX.set(rx, []);
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
}
for (const [rx, intervals] of byX) {
if (chainCoversRange(intervals, lowerY, upperY, eps)) xs.add(rx);
}
return xs;
}
const MIN_RICH_BRIDGING_COLS = 3;
function splitYLinesIntoGroups(yLines: number[], verticals: Segment[]): number[][] {
if (yLines.length === 0) return [];
const eps = 1.5;
const allX = verticals.map(s => Math.round(s.x1));
const globalXMin = allX.length > 0 ? Math.min(...allX) : 0;
const globalXMax = allX.length > 0 ? Math.max(...allX) : 0;
const groups: number[][] = [];
let currentGroup = [yLines[0]];
let prevBridgingCols = -1;
for (let i = 1; i < yLines.length; i++) {
const upperY = yLines[i - 1];
const lowerY = yLines[i];
const cols = countBridgingVLineCols(upperY, lowerY, verticals);
if (cols === 0) {
groups.push(currentGroup);
currentGroup = [yLines[i]];
prevBridgingCols = -1;
continue;
}
if (prevBridgingCols >= MIN_RICH_BRIDGING_COLS && cols < MIN_RICH_BRIDGING_COLS) {
const bxs = bridgingXSet(upperY, lowerY, verticals);
const isOuterFrameOnly = [...bxs].every(
x => Math.abs(x - globalXMin) <= eps || Math.abs(x - globalXMax) <= eps,
);
if (!isOuterFrameOnly) {
groups.push(currentGroup);
currentGroup = [yLines[i - 1], yLines[i]];
prevBridgingCols = cols;
continue;
}
}
currentGroup.push(yLines[i]);
prevBridgingCols = cols;
}
groups.push(currentGroup);
return groups;
}
// ---------------------------------------------------------------------------
// Sub-row Y-cluster expansion
// ---------------------------------------------------------------------------
const Y_CLUSTER_GAP = 10;
const MIN_COLS_IN_TOP_CLUSTER = 2;
function assignToYCluster(y: number, clusters: number[]): number {
let closest = 0;
let closestDist = Math.abs(y - clusters[0]);
for (let k = 1; k < clusters.length; k++) {
const d = Math.abs(y - clusters[k]);
if (d < closestDist) {
closestDist = d;
closest = k;
}
}
return closest;
}
function expandSubRowsByYClusters(
originalRows: number,
cols: number,
cells: TableCell[],
cellBoxes: Map<TableCell, TextBox[]>,
): number {
let addedRows = 0;
for (let origRow = 0; origRow < originalRows; origRow++) {
const currentRow = origRow + addedRows;
const rowCellInfos: Array<{ cell: TableCell; col: number; boxes: TextBox[] }> = [];
for (let col = 0; col < cols; col++) {
const cell = cells.find(c => c.row === currentRow && c.col === col);
if (!cell) continue;
const boxes = cellBoxes.get(cell);
if (boxes && boxes.length > 0) rowCellInfos.push({ cell, col, boxes });
}
if (rowCellInfos.length === 0) continue;
const allMidYs = rowCellInfos.flatMap(({ boxes }) => boxes.map(b => (b.bounds.top + b.bounds.bottom) / 2));
const sortedY = [...new Set(allMidYs.map(y => Math.round(y * 10) / 10))].sort((a, b) => b - a);
const clusters = [sortedY[0]];
for (let i = 1; i < sortedY.length; i++) {
if (clusters[clusters.length - 1] - sortedY[i] > Y_CLUSTER_GAP) {
clusters.push(sortedY[i]);
}
}
if (clusters.length < 2) continue;
const colsInTopCluster = new Set<number>();
const totalNonEmptyCols = new Set<number>();
for (const { col, boxes } of rowCellInfos) {
totalNonEmptyCols.add(col);
if (boxes.some(b => assignToYCluster((b.bounds.top + b.bounds.bottom) / 2, clusters) === 0)) {
colsInTopCluster.add(col);
}
}
if (colsInTopCluster.size < MIN_COLS_IN_TOP_CLUSTER) continue;
if (colsInTopCluster.size >= totalNonEmptyCols.size) continue;
const sparseColsHaveMultipleBoxes = rowCellInfos.some(
({ col, boxes }) => !colsInTopCluster.has(col) && boxes.length > 1,
);
if (!sparseColsHaveMultipleBoxes) continue;
const numSubRows = clusters.length;
const numNewRows = numSubRows - 1;
for (const cell of cells) {
if (cell.row > currentRow) cell.row += numNewRows;
}
for (let subRow = 1; subRow < numSubRows; subRow++) {
for (let col = 0; col < cols; col++) {
cells.push({
row: currentRow + subRow,
col,
text: "",
rowSpan: 1,
colSpan: 1,
});
}
}
for (const { cell: origCell, col, boxes } of rowCellInfos) {
const subRowBoxGroups: TextBox[][] = Array.from({ length: numSubRows }, () => []);
for (const box of boxes) {
const cy = (box.bounds.top + box.bounds.bottom) / 2;
subRowBoxGroups[assignToYCluster(cy, clusters)].push(box);
}
cellBoxes.set(origCell, subRowBoxGroups[0]);
if (subRowBoxGroups[0].length === 0) cellBoxes.delete(origCell);
for (let subRow = 1; subRow < numSubRows; subRow++) {
if (subRowBoxGroups[subRow].length > 0) {
const newCell = cells.find(c => c.row === currentRow + subRow && c.col === col);
if (newCell) cellBoxes.set(newCell, subRowBoxGroups[subRow]);
}
}
}
addedRows += numNewRows;
}
return originalRows + addedRows;
}
// ---------------------------------------------------------------------------
// Cross-column text box splitting
// ---------------------------------------------------------------------------
/**
* Find which column a horizontal position falls into.
* Returns -1 if outside the grid.
*/
function findCol(x: number, xLines: number[]): number {
for (let i = 0; i < xLines.length - 1; i++) {
if (x >= xLines[i] && x <= xLines[i + 1]) return i;
}
return -1;
}
/**
* When a text box spans across one or more vertical column boundaries,
* split it into multiple virtual text boxes — one per column — with the
* text divided proportionally by width.
*
* We split at word boundaries closest to the proportional split point
* so we don't chop words in half.
*/
function splitCrossColumnBoxes(textBoxes: TextBox[], xLines: number[]): TextBox[] {
const result: TextBox[] = [];
const MARGIN = 5; // allow small overlap before considering it cross-column
for (const tb of textBoxes) {
const leftCol = findCol(tb.bounds.left + MARGIN, xLines);
const rightCol = findCol(tb.bounds.right - MARGIN, xLines);
// Not spanning columns, or outside grid — keep as-is
if (leftCol < 0 || rightCol < 0 || leftCol === rightCol) {
result.push(tb);
continue;
}
// Text box spans from leftCol to rightCol — split it
const totalWidth = tb.bounds.right - tb.bounds.left;
if (totalWidth <= 0) {
result.push(tb);
continue;
}
const words = tb.text.split(/\s+/);
if (words.length <= 1) {
// Single word spanning columns — just assign to whichever col has more overlap
result.push(tb);
continue;
}
// For each column boundary crossing, find the best word-boundary split
let remainingWords = [...words];
let currentLeft = tb.bounds.left;
for (let col = leftCol; col <= rightCol && remainingWords.length > 0; col++) {
const colRight = col < xLines.length - 1 ? xLines[col + 1] : tb.bounds.right;
const segmentRight = Math.min(colRight, tb.bounds.right);
if (col === rightCol) {
// Last column — take all remaining words
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: remainingWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: tb.bounds.right,
},
});
remainingWords = [];
} else {
// Find how many words fit in this column segment proportionally
const segmentWidth = segmentRight - currentLeft;
const fractionOfTotal = segmentWidth / totalWidth;
const approxChars = Math.round(fractionOfTotal * tb.text.length);
// Walk words to find the split closest to the proportional point
let charCount = 0;
let splitIdx = 0;
for (let w = 0; w < remainingWords.length; w++) {
const nextCount = charCount + remainingWords[w].length + (w > 0 ? 1 : 0);
if (nextCount > approxChars && splitIdx > 0) break;
charCount = nextCount;
splitIdx = w + 1;
}
if (splitIdx === 0) splitIdx = 1; // take at least one word
if (splitIdx >= remainingWords.length) {
// All remaining words fit here
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: remainingWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: segmentRight,
},
});
remainingWords = [];
} else {
const partWords = remainingWords.slice(0, splitIdx);
result.push({
...tb,
id: `${tb.id}-split${col}`,
text: partWords.join(" "),
bounds: {
...tb.bounds,
left: currentLeft,
right: segmentRight,
},
});
remainingWords = remainingWords.slice(splitIdx);
currentLeft = segmentRight;
}
}
}
}
return result;
}
// ---------------------------------------------------------------------------
// Full grid table (H + V lines)
// ---------------------------------------------------------------------------
function buildCells(rows: number, cols: number): TableCell[] {
const cells: TableCell[] = [];
for (let row = 0; row < rows; row++) {
for (let col = 0; col < cols; col++) {
cells.push({ row, col, text: "", rowSpan: 1, colSpan: 1 });
}
}
return cells;
}
function buildTableGrid(
pageNumber: number,
yLines: number[],
xLines: number[],
filteredSegments: Segment[],
textBoxes: TextBox[],
): { grid: TableGrid; consumedIds: string[] } {
let rows = yLines.length - 1;
const cols = xLines.length - 1;
const cells = buildCells(rows, cols);
const consumedIds: string[] = [];
const yMin = yLines[yLines.length - 1];
const yMax = yLines[0];
const xMin = xLines[0];
const xMax = xLines[xLines.length - 1];
// Split text boxes that span multiple columns before placement
const splitBoxes = splitCrossColumnBoxes(textBoxes, xLines);
// Track which split piece IDs get placed in cells, so we can consume
// the original (unsplit) text box IDs too.
const placedSplitIds = new Set<string>();
// Look for header text boxes just above the grid.
// Use the ORIGINAL (unsplit) text boxes for header detection so that
// wide paragraph text isn't falsely split into column-sized header chunks.
// Reject boxes wider than 1.5 columns — those are paragraph text, not headers.
const avgColWidth = (xMax - xMin) / cols;
const maxHeaderBoxWidth = avgColWidth * 1.5;
const headerBoxes = textBoxes.filter(tb => {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const boxWidth = tb.bounds.right - tb.bounds.left;
return cy > yMax && cy <= yMax + 20 && cx >= xMin && cx <= xMax && boxWidth <= maxHeaderBoxWidth;
});
if (headerBoxes.length > 0) {
rows += 1;
for (const cell of cells) cell.row += 1;
for (let col = 0; col < cols; col++) {
cells.push({ row: 0, col, text: "", rowSpan: 1, colSpan: 1 });
}
for (const tb of headerBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col >= 0 && col < cols) {
const cell = cells.find(c => c.row === 0 && c.col === col);
if (cell) {
cell.text = cell.text.length === 0 ? tb.text : `${cell.text} ${tb.text}`;
consumedIds.push(tb.id);
}
}
}
}
const cellBoxes = new Map<TableCell, TextBox[]>();
for (const tb of splitBoxes) {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
if (cy < yMin || cy > yMax || cx < xMin || cx > xMax) continue;
const rays = castRaysForTextBox(tb, filteredSegments);
const rayConfidence = rays.filter(r => r.segmentId !== null).length;
let row = yLines.findIndex((lineY, idx) => {
const next = yLines[idx + 1];
return next !== undefined && cy <= lineY && cy >= next;
});
if (row < 0 || row >= (headerBoxes.length > 0 ? rows - 1 : rows)) continue;
if (headerBoxes.length > 0) row += 1;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col < 0 || col >= cols) continue;
if (rayConfidence === 0) continue;
const cell = cells.find(c => c.row === row && c.col === col);
if (!cell) continue;
if (!cellBoxes.has(cell)) cellBoxes.set(cell, []);
cellBoxes.get(cell)?.push(tb);
consumedIds.push(tb.id);
if (tb.id.includes("-split")) placedSplitIds.add(tb.id);
}
rows = expandSubRowsByYClusters(rows, cols, cells, cellBoxes);
// Merge text boxes within each cell into cell text
for (const [cell, boxes] of cellBoxes.entries()) {
boxes.sort((a, b) => b.bounds.top - a.bounds.top);
const lines: string[] = [];
let currentLine: string[] = [];
let currentY = boxes[0].bounds.top;
for (const box of boxes) {
if (Math.abs(box.bounds.top - currentY) > 5) {
lines.push(currentLine.join(" "));
currentLine = [box.text];
currentY = box.bounds.top;
} else {
currentLine.push(box.text);
}
}
if (currentLine.length > 0) lines.push(currentLine.join(" "));
cell.text = lines.join("<br>");
}
const grid = pruneEmptyRowsAndCols({
pageNumber,
rows,
cols,
cells,
warnings: [],
topY: yLines[0],
isBorderless: false,
});
// Also consume the original (unsplit) text box IDs when any of their
// split pieces were placed in a cell.
for (const splitId of placedSplitIds) {
const origId = splitId.replace(/-split\d+$/, "");
if (!consumedIds.includes(origId)) {
consumedIds.push(origId);
}
}
return { grid, consumedIds };
}
// ---------------------------------------------------------------------------
// H-line-only table (inferred columns)
// ---------------------------------------------------------------------------
const COL_GAP_THRESHOLD = 20;
const HONLY_ROW_GAP = 30;
const HONLY_ROW_TOLERANCE = 8;
const MIN_TABLE_HEIGHT = 24;
const MIN_LEFT_SPREAD = 50;
function inferXLinesFromBoxes(textBoxes: TextBox[], xMin: number, xMax: number): number[] {
const centers = textBoxes.map(tb => (tb.bounds.left + tb.bounds.right) / 2).sort((a, b) => a - b);
if (centers.length === 0) return [xMin, xMax];
const boundaries = [xMin];
for (let i = 1; i < centers.length; i++) {
if (centers[i] - centers[i - 1] >= COL_GAP_THRESHOLD) {
boundaries.push((centers[i - 1] + centers[i]) / 2);
}
}
boundaries.push(xMax);
return boundaries;
}
function buildHLineOnlyTable(
pageNumber: number,
yLines: number[],
xMin: number,
xMax: number,
textBoxes: TextBox[],
alreadyConsumed: Set<string>,
): { grid: TableGrid; consumedIds: string[] } | null {
const yMax = yLines[0];
const yMin = yLines[yLines.length - 1];
const candidates = textBoxes.filter(tb => !alreadyConsumed.has(tb.id));
const BOX_LEFT_TOLERANCE = 30;
const inRange = candidates.filter(tb => {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
return (
tb.bounds.left >= xMin - BOX_LEFT_TOLERANCE &&
tb.bounds.right <= xMax + BOX_LEFT_TOLERANCE &&
cy >= yMin &&
cy <= yMax
);
});
// Extend downward below yMin
const belowYMin = candidates
.filter(tb => {
const cx = (tb.bounds.left + tb.bounds.right) / 2;
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
return cx >= xMin && cx <= xMax && cy < yMin;
})
.sort((a, b) => (b.bounds.top + b.bounds.bottom) / 2 - (a.bounds.top + a.bounds.bottom) / 2);
const extensionBoxes: TextBox[] = [];
let lastY = yMin;
for (const tb of belowYMin) {
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
if (lastY - cy > HONLY_ROW_GAP) break;
extensionBoxes.push(tb);
lastY = cy;
}
const allBoxes = [...inRange, ...extensionBoxes];
if (allBoxes.length === 0) return null;
const leftEdges = allBoxes.map(tb => tb.bounds.left);
if (Math.max(...leftEdges) - Math.min(...leftEdges) < MIN_LEFT_SPREAD) return null;
const xLines = inferXLinesFromBoxes(allBoxes, xMin, xMax);
if (xLines.length < 2) return null;
const cols = xLines.length - 1;
// Build visual rows
const visualRows: Array<{ midY: number; boxes: TextBox[] }> = [];
const sortedBoxes = [...allBoxes].sort((a, b) => {
const ya = (a.bounds.top + a.bounds.bottom) / 2;
const yb = (b.bounds.top + b.bounds.bottom) / 2;
if (Math.abs(ya - yb) > 0.5) return yb - ya;
return a.bounds.left - b.bounds.left;
});
for (const box of sortedBoxes) {
const cy = (box.bounds.top + box.bounds.bottom) / 2;
const last = visualRows[visualRows.length - 1];
if (last && Math.abs(last.midY - cy) <= HONLY_ROW_TOLERANCE) {
last.boxes.push(box);
} else {
visualRows.push({ midY: cy, boxes: [box] });
}
}
if (visualRows.length === 0) return null;
const cells: TableCell[] = [];
const consumedIds: string[] = [];
for (let rowIdx = 0; rowIdx < visualRows.length; rowIdx++) {
const vrow = visualRows[rowIdx];
const colBoxes = new Map<number, TextBox[]>();
for (const box of vrow.boxes) {
const cx = (box.bounds.left + box.bounds.right) / 2;
const col = xLines.findIndex((lineX, idx) => {
const next = xLines[idx + 1];
return next !== undefined && cx >= lineX && cx <= next;
});
if (col >= 0 && col < cols) {
if (!colBoxes.has(col)) colBoxes.set(col, []);
colBoxes.get(col)?.push(box);
}
}
for (let c = 0; c < cols; c++) {
const cbs = (colBoxes.get(c) ?? []).sort((a, b) => a.bounds.left - b.bounds.left);
cells.push({
row: rowIdx,
col: c,
text: cbs.map(b => b.text).join(" "),
rowSpan: 1,
colSpan: 1,
});
consumedIds.push(...cbs.map(b => b.id));
}
}
const contentTopY = visualRows.length > 0 ? visualRows[0].midY : yMax;
const grid = pruneEmptyRowsAndCols({
pageNumber,
rows: visualRows.length,
cols,
cells,
warnings: [],
topY: contentTopY,
isBorderless: false,
});
return { grid, consumedIds };
}
// ---------------------------------------------------------------------------
// Pruning
// ---------------------------------------------------------------------------
function pruneEmptyRowsAndCols(table: TableGrid): TableGrid {
const occupiedRows = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.row));
const occupiedCols = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.col));
if (occupiedRows.size === 0) return table;
const rowMap = new Map<number, number>();
let newRow = 0;
for (let r = 0; r < table.rows; r++) {
if (occupiedRows.has(r)) rowMap.set(r, newRow++);
}
const colMap = new Map<number, number>();
let newCol = 0;
for (let c = 0; c < table.cols; c++) {
if (occupiedCols.has(c)) colMap.set(c, newCol++);
}
const prunedCells = table.cells
.filter(c => occupiedRows.has(c.row) && occupiedCols.has(c.col))
.map(c => ({
...c,
row: rowMap.get(c.row) ?? c.row,
col: colMap.get(c.col) ?? c.col,
}));
return { ...table, rows: newRow, cols: newCol, cells: prunedCells };
}
// ---------------------------------------------------------------------------
// Diagram vs table discrimination
// ---------------------------------------------------------------------------
/** Maximum column count for a plausible data table. */
const MAX_TABLE_COLS = 25;
/**
* Returns true if a grid looks like a vector diagram rather than a data table.
*
* Heuristics (any match → diagram):
* 1. Column count > 25 (diagrams create many X-lines from box edges)
* 2. Fill ratio < 25% (most cells empty — scattered boxes)
* 3. Fill < 50% AND duplicate text ratio > 30% (repeating labels in a
* diagram layout, e.g. "Hash", "Transaction" appearing in each column)
* 4. Fill < 50% AND cols >= 6 (moderate sparseness with wide grid)
*/
function isDiagram(grid: TableGrid): boolean {
const totalCells = grid.rows * grid.cols;
if (totalCells === 0) return true;
const filled = grid.cells.filter(c => c.text.trim().length > 0);
const fillRatio = filled.length / totalCells;
// Very high column count
if (grid.cols > MAX_TABLE_COLS) return true;
// Very sparse
if (fillRatio < 0.25) return true;
// Compute duplicate text ratio among non-trivial cells.
// Exclude short values (≤3 chars) like "—", "V", "YES", "NO" which
// naturally repeat in real data tables.
const substantive = filled.filter(c => c.text.trim().length > 3);
const uniqueTexts = new Set(substantive.map(c => c.text.trim())).size;
const dupRatio = substantive.length > 2 ? 1 - uniqueTexts / substantive.length : 0;
// Sparse + highly duplicated substantive text → repeating diagram
if (fillRatio < 0.5 && dupRatio > 0.3) return true;
// High duplication + wide grid → repeating diagram even at moderate fill
if (dupRatio > 0.4 && grid.cols >= 6) return true;
// Sparse + wide grid with no substantive text to judge
if (fillRatio < 0.4 && grid.cols >= 6) return true;
return false;
}
/**
* Detect all table grids on a single page from its text boxes and segments.
*/
export function resolveTableGrids(pageNumber: number, textBoxes: TextBox[], segments: Segment[]): GridResult {
const vertical = segments.filter(s => Math.abs(s.x1 - s.x2) <= AXIS_EPSILON);
const horizontal = segments.filter(s => Math.abs(s.y1 - s.y2) <= AXIS_EPSILON);
// Filter segments to the text's visible area
const textYValues = textBoxes.flatMap(t => [t.bounds.bottom, t.bounds.top]);
const textYMin = textYValues.length > 0 ? Math.min(...textYValues) - PAGE_MARGIN : -Infinity;
const textYMax = textYValues.length > 0 ? Math.max(...textYValues) + PAGE_MARGIN : Infinity;
const textXValues = textBoxes.flatMap(t => [t.bounds.left, t.bounds.right]);
const textXMin = textXValues.length > 0 ? Math.min(...textXValues) - 100 : -Infinity;
const textXMax = textXValues.length > 0 ? Math.max(...textXValues) + 100 : Infinity;
const filteredH = horizontal.filter(
s => s.y1 >= textYMin && s.y1 <= textYMax && s.x1 <= textXMax && s.x2 >= textXMin,
);
const hMaxX2 = filteredH.length > 0 ? Math.max(...filteredH.map(s => s.x2)) : textXMax;
const vSegXMax = Math.max(textXMax, hMaxX2 + PAGE_MARGIN);
const filteredV = vertical.filter(s => {
const segMin = Math.min(s.y1, s.y2);
const segMax = Math.max(s.y1, s.y2);
return segMax >= textYMin && segMin <= textYMax && s.x1 >= textXMin && s.x1 <= vSegXMax;
});
const allYLines = uniqueSorted(filteredH.flatMap(s => [s.y1, s.y2])).sort((a, b) => b - a);
if (allYLines.length < 2) {
return { grids: [], consumedIds: [] };
}
const filteredSegments = [...filteredH, ...filteredV];
const yGroups = splitYLinesIntoGroups(allYLines, filteredV);
const grids: TableGrid[] = [];
const gridConsumedIds: string[][] = [];
// Flat set for the alreadyConsumed check in H-line-only tables
const allConsumedIds: string[] = [];
for (const yLines of yGroups) {
if (yLines.length < 2) continue;
const yMin = yLines[yLines.length - 1];
const yMax = yLines[0];
const groupVerticals = filteredV.filter(s => {
const segMin = Math.min(s.y1, s.y2);
const segMax = Math.max(s.y1, s.y2);
return segMin < yMax - 1.5 && segMax > yMin + 1.5;
});
const groupXLines = uniqueSorted(groupVerticals.flatMap(s => [s.x1, s.x2]));
if (groupXLines.length < 2) {
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
const groupHoriz = filteredH.filter(s => s.y1 >= yMin - 1.5 && s.y1 <= yMax + 1.5);
if (groupHoriz.length === 0) continue;
const hxMin = Math.min(...groupHoriz.map(s => s.x1));
const hxMax = Math.max(...groupHoriz.map(s => s.x2));
const result = buildHLineOnlyTable(pageNumber, yLines, hxMin, hxMax, textBoxes, new Set(allConsumedIds));
if (result) {
grids.push(result.grid);
gridConsumedIds.push(result.consumedIds);
allConsumedIds.push(...result.consumedIds);
}
continue;
}
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
const result = buildTableGrid(pageNumber, yLines, groupXLines, filteredSegments, textBoxes);
grids.push(result.grid);
gridConsumedIds.push(result.consumedIds);
allConsumedIds.push(...result.consumedIds);
}
// Filter out grids that look like vector diagrams, not data tables.
// Their consumed text box IDs are released so the text becomes free text.
const filteredGrids: TableGrid[] = [];
const filteredConsumedIds: string[] = [];
for (let i = 0; i < grids.length; i++) {
if (isDiagram(grids[i])) continue;
filteredGrids.push(grids[i]);
filteredConsumedIds.push(...gridConsumedIds[i]);
}
return { grids: filteredGrids, consumedIds: filteredConsumedIds };
}
@@ -0,0 +1,106 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Running header/footer detection and removal.
*
* Many PDFs have repeated text at the top or bottom of every page:
* document titles, chapter names, page numbers, copyright notices.
* These pollute the markdown output as false headings or noise.
*
* Algorithm:
* 1. For each page, bucket text boxes by Y position (top/bottom zones)
* 2. Collect the text content at each zone across all pages
* 3. Text appearing on >20% of pages OR 8+ consecutive pages is a
* running header/footer
* 4. Remove matching text boxes before further processing
*/
import type { PageContent } from "./types";
/** Minimum number of pages to enable header/footer detection. */
const MIN_PAGES = 5;
/** Minimum Y position for top zone (from bottom of page in PDF coords). */
const TOP_ZONE_MIN_Y = 700;
/** Maximum Y position for bottom zone. */
const BOTTOM_ZONE_MAX_Y = 80;
/**
* Minimum consecutive pages a text must appear on to be considered a
* running header/footer. Catches both document-wide headers (appearing
* on every page) and chapter-specific headers (appearing on 4+ consecutive
* pages within a chapter).
*/
const MIN_CONSECUTIVE_PAGES = 8;
/**
* Detect and remove running headers and footers from all pages.
* Mutates the pages array in place, removing header/footer text boxes.
*
* Uses two strategies:
* 1. Global frequency: text appearing on > 20% of all pages
* 2. Consecutive runs: text appearing on 8+ consecutive pages
*/
export function stripHeadersFooters(pages: PageContent[]): void {
if (pages.length < MIN_PAGES) return;
// Step 1: Build per-page zone text sets
const pageZoneTexts: Set<string>[] = [];
for (const page of pages) {
const zoneTexts = new Set<string>();
for (const tb of page.textBoxes) {
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
if (midY >= TOP_ZONE_MIN_Y || midY <= BOTTOM_ZONE_MAX_Y) {
const key = tb.text.trim().replace(/\s+/g, " ");
if (key.length > 0) zoneTexts.add(key);
}
}
pageZoneTexts.push(zoneTexts);
}
// Step 2: Count global frequency AND longest consecutive run for each text
const globalCount = new Map<string, number>();
const maxConsecutive = new Map<string, number>();
// Collect all unique zone texts
const allTexts = new Set<string>();
for (const zts of pageZoneTexts) {
for (const t of zts) allTexts.add(t);
}
for (const text of allTexts) {
let total = 0;
let consecutive = 0;
let maxRun = 0;
for (const zts of pageZoneTexts) {
if (zts.has(text)) {
total++;
consecutive++;
if (consecutive > maxRun) maxRun = consecutive;
} else {
consecutive = 0;
}
}
globalCount.set(text, total);
maxConsecutive.set(text, maxRun);
}
// Step 3: Identify running headers/footers
const globalThreshold = Math.max(3, Math.floor(pages.length * 0.2));
const repeatedTexts = new Set<string>();
for (const text of allTexts) {
const gc = globalCount.get(text) ?? 0;
const mc = maxConsecutive.get(text) ?? 0;
// Global: appears on 20%+ of pages
if (gc >= globalThreshold) {
repeatedTexts.add(text);
continue;
}
// Consecutive: appears on 8+ consecutive pages (chapter-level headers)
if (mc >= MIN_CONSECUTIVE_PAGES) {
repeatedTexts.add(text);
}
}
if (repeatedTexts.size === 0) return;
// Step 4: Remove matching text boxes from each page
for (const page of pages) {
page.textBoxes = page.textBoxes.filter(tb => {
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
if (midY < TOP_ZONE_MIN_Y && midY > BOTTOM_ZONE_MAX_Y) return true;
const normalized = tb.text.trim().replace(/\s+/g, " ");
return !repeatedTexts.has(normalized);
});
}
}
@@ -0,0 +1,146 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* PDF to Markdown converter.
*
* Uses mupdf (native WASM) for fast PDF parsing and a custom pipeline for
* table detection via vector line extraction + raycasting.
*
* Pipeline:
* 1. Extract text boxes + vector segments + image regions per page (mupdf)
* 2. Detect column layout (single vs multi-column)
* 3. Per column: detect table grids from segments (grid detection + raycasting)
* 4. Render diagrams as PNG files (if output directory provided)
* 5. Render tables as markdown tables, free text as paragraphs/headings
*/
import * as path from "node:path";
import type { ConversionResult, Converter, StreamInfo } from "../../types";
import { detectColumns } from "./columns";
import { extractPages, renderImageRegion } from "./extract";
import { resolveTableGrids } from "./grid";
import { stripHeadersFooters } from "./headers";
import { renderPageContent } from "./render";
import type { Segment, TextBox } from "./types";
const EXTENSIONS = [".pdf"];
const MIMETYPES = ["application/pdf", "application/x-pdf"];
type ImageBlock = { topY: number; markdown: string };
/**
* Process a set of text boxes (one column or full page): run table detection,
* separate free text, and render to markdown.
*/
function processColumn(
pageNumber: number,
textBoxes: TextBox[],
segments: Segment[],
imageBlocks: ImageBlock[],
): string {
const { grids, consumedIds } = resolveTableGrids(pageNumber, textBoxes, segments);
const consumedSet = new Set(consumedIds);
const freeTextBoxes = textBoxes.filter(tb => !consumedSet.has(tb.id));
return renderPageContent(freeTextBoxes, grids, imageBlocks, textBoxes);
}
export class PdfConverter implements Converter {
name = "pdf";
accepts(streamInfo: StreamInfo): boolean {
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) {
return true;
}
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) {
return true;
}
return false;
}
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
const pdfBytes = new Uint8Array(input);
const pages = await extractPages(pdfBytes);
// Remove running headers/footers before processing.
stripHeadersFooters(pages);
const imageDir = streamInfo.imageDir;
const pageMarkdowns: string[] = [];
for (const page of pages) {
// Build image blocks for this page.
const imageBlocks: ImageBlock[] = [];
if (imageDir && page.images.length > 0) {
for (const img of page.images) {
const filename = `${img.id}.png`;
const filepath = path.join(imageDir, filename);
try {
const png = renderImageRegion(pdfBytes, img);
await Bun.write(filepath, png);
imageBlocks.push({ topY: img.topY, markdown: `![${img.id}](${filepath})` });
} catch {
// Image rendering failed — skip.
}
}
} else if (page.images.length > 0) {
for (const img of page.images) {
imageBlocks.push({
topY: img.topY,
markdown: `<!-- image: ${img.id} (page ${img.pageNumber}, ${img.bbox.w}x${img.bbox.h}pt) -->`,
});
}
}
// Detect column layout.
// If the page has vertical segments (tables), suppress column detection
// when one detected column is very narrow — that's a table's first column,
// not a page layout column.
const layout = detectColumns(page.textBoxes);
if (layout.columnCount > 1 && page.segments.some(s => Math.abs(s.x1 - s.x2) <= 0.8)) {
const pageXMin = Math.min(...page.textBoxes.map(tb => tb.bounds.left));
const pageXMax = Math.max(...page.textBoxes.map(tb => tb.bounds.right));
const pageWidth = pageXMax - pageXMin;
const minColFraction = 0.3;
const tooNarrow = layout.columns.some(col => {
const colXMin = Math.min(...col.map(tb => tb.bounds.left));
const colXMax = Math.max(...col.map(tb => tb.bounds.right));
return (colXMax - colXMin) / pageWidth < minColFraction;
});
if (tooNarrow) {
layout.columnCount = 1;
layout.columns = [page.textBoxes];
layout.boundaries = [];
}
}
if (layout.columnCount === 1) {
// Single column — process normally.
const md = processColumn(page.pageNumber, page.textBoxes, page.segments, imageBlocks);
if (md.length > 0) pageMarkdowns.push(md);
} else {
// Multi-column — process each column independently, then join.
const columnMarkdowns: string[] = [];
for (const colBoxes of layout.columns) {
// Filter segments to those within this column's X range.
const colXMin = Math.min(...colBoxes.map(tb => tb.bounds.left));
const colXMax = Math.max(...colBoxes.map(tb => tb.bounds.right));
const margin = 10;
const colSegments = page.segments.filter(seg => {
const segXMin = Math.min(seg.x1, seg.x2);
const segXMax = Math.max(seg.x1, seg.x2);
return segXMax >= colXMin - margin && segXMin <= colXMax + margin;
});
// Images go with the first column only (no X info to split by).
const md = processColumn(
page.pageNumber,
colBoxes,
colSegments,
columnMarkdowns.length === 0 ? imageBlocks : [],
);
if (md.length > 0) columnMarkdowns.push(md);
}
const joined = columnMarkdowns.join("\n\n");
if (joined.length > 0) pageMarkdowns.push(joined);
}
}
return { markdown: pageMarkdowns.join("\n\n") };
}
}
@@ -0,0 +1,501 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/**
* Markdown rendering for PDF pages.
*
* Converts table grids and free text boxes into markdown, handling:
* - Table grid → markdown table (`| col | col |`)
* - Free text → paragraphs with heading detection (by font size)
* - Content ordering (top-to-bottom via Y coordinate)
* - Paragraph wrap merging (lines broken across PDF line boundaries)
* - Page number removal
*
* Ported from @oharato/pdf2md-ts, stripped of CJK/TDnet-specific logic.
*/
import type { ContentBlock, TableGrid, TextBox } from "./types";
/** A free-text line grouped from horizontally adjacent text boxes. */
interface RenderLine {
text: string;
topY: number;
fontSize: number;
isBold: boolean;
isTabular: boolean;
}
/** A content block carrying the Y of its last wrapped line during merging. */
type WrapBlock = ContentBlock & { lastTopY: number };
// ---------------------------------------------------------------------------
// Utility
// ---------------------------------------------------------------------------
/** Convert full-width ASCII characters (A→A, !→! etc.) to normal ASCII. */
function normalizeFullWidthAscii(text: string): string {
return text.replace(/[!-~]/g, ch => String.fromCharCode(ch.charCodeAt(0) - 0xfee0));
}
function escapePipes(text: string): string {
return normalizeFullWidthAscii(text).replaceAll("|", "\\|").replaceAll("\n", "<br>");
}
/** Parse a markdown pipe-delimited row into cell strings. */
function parsePipeRow(line: string): string[] {
const trimmed = line.trim();
if (!trimmed.startsWith("|") || !trimmed.endsWith("|")) return [];
return trimmed
.slice(1, -1)
.split("|")
.map(cell => cell.trim());
}
// ---------------------------------------------------------------------------
// Table rendering
// ---------------------------------------------------------------------------
/**
* Render a TableGrid as a markdown table.
*/
export function renderTableToMarkdown(table: TableGrid): string {
if (table.rows === 0 || table.cols === 0) return "";
const matrix = Array.from({ length: table.rows }, () => Array.from({ length: table.cols }, () => ""));
for (const cell of table.cells) {
if (cell.row < table.rows && cell.col < table.cols) {
matrix[cell.row][cell.col] = escapePipes(cell.text.trim());
}
}
const normalized = normalizeShiftedSparseColumns(matrix);
const promoted = promoteSubHeaderPrefixes(normalized);
const header = `| ${promoted[0].join(" | ")} |`;
const divider = `| ${Array.from({ length: promoted[0].length }, () => "---").join(" | ")} |`;
const body = promoted
.slice(1)
.map(row => `| ${row.join(" | ")} |`)
.join("\n");
return [header, divider, body].filter(l => l.length > 0).join("\n");
}
/**
* Fix tables with ≥5 columns where sparse single-value columns are
* misaligned. Shifts those values to the adjacent dense column and
* removes the now-empty sparse columns.
*/
function normalizeShiftedSparseColumns(matrix: string[][]): string[][] {
if (matrix.length === 0 || matrix[0].length < 5) return matrix;
const _rows = matrix.length;
const cols = matrix[0].length;
const counts = Array.from({ length: cols }, (_, c) =>
matrix.reduce((n, row) => n + (row[c].trim().length > 0 ? 1 : 0), 0),
);
const denseCols = new Set(
counts
.map((count, col) => ({ count, col }))
.filter(({ col, count }) => col === 0 || count >= 2)
.map(({ col }) => col),
);
const sparseCols = counts
.map((count, col) => ({ count, col }))
.filter(({ col, count }) => col > 0 && col < cols - 1 && count === 1)
.map(({ col }) => col);
if (sparseCols.length < 2 || denseCols.size < 4) return matrix;
const moves: Array<{ from: number; to: number; row: number }> = [];
for (const from of sparseCols) {
const row = matrix.findIndex(r => r[from].trim().length > 0);
const to = from + 1;
if (row < 0) return matrix;
if (!denseCols.has(to)) return matrix;
if (matrix[row][to].trim().length > 0) return matrix;
moves.push({ from, to, row });
}
const copy = matrix.map(row => [...row]);
for (const { from, to, row } of moves) {
copy[row][to] = copy[row][to].trim().length > 0 ? `${copy[row][to]} ${copy[row][from]}` : copy[row][from];
copy[row][from] = "";
}
const keepCols = Array.from({ length: cols }, (_, c) => c).filter(c => copy.some(row => row[c].trim().length > 0));
if (keepCols.length === cols) return copy;
return copy.map(row => keepCols.map(c => row[c]));
}
/**
* When a data row has ≥2 parenthesized qualifiers in non-first columns
* (and the first column is empty), promote them into the header row.
*/
function promoteSubHeaderPrefixes(matrix: string[][]): string[][] {
if (matrix.length < 2) return matrix;
const PAREN_RE = /^\([^)]{1,40}\)$/;
const result = matrix.map(row => [...row]);
const cols = matrix[0].length;
const rowsToRemove = new Set<number>();
for (let r = 1; r < result.length; r++) {
if (rowsToRemove.has(r)) continue;
const promotable: Array<{ col: number; prefix: string; isFullCell: boolean }> = [];
for (let col = 1; col < cols; col++) {
const cell = (result[r][col] ?? "").trim();
if (!cell) continue;
const parts = cell.split("<br>");
if (parts.length === 1 && PAREN_RE.test(cell)) {
promotable.push({ col, prefix: cell, isFullCell: true });
} else if (parts.length >= 2 && PAREN_RE.test(parts[0].trim())) {
promotable.push({
col,
prefix: parts[0].trim(),
isFullCell: false,
});
}
}
if (promotable.length < 2) continue;
if (promotable.some(p => p.isFullCell) && result[r][0].trim().length > 0) continue;
for (const { col, prefix, isFullCell } of promotable) {
result[0][col] = result[0][col].trim() ? `${result[0][col]} ${prefix}` : prefix;
if (isFullCell) {
result[r][col] = "";
} else {
const parts = result[r][col].split("<br>");
result[r][col] = parts.slice(1).join("<br>");
}
}
if (result[r].every(cell => cell.trim().length === 0)) {
rowsToRemove.add(r);
}
}
return result.filter((_, r) => !rowsToRemove.has(r));
}
// ---------------------------------------------------------------------------
// Free text rendering
// ---------------------------------------------------------------------------
/** Y tolerance for grouping text boxes onto the same visual line. */
const TEXT_LINE_Y_TOLERANCE = 3;
/** Minimum X gap between adjacent boxes to mark line as tabular. */
const TABULAR_X_GAP = 30;
/**
* Minimum font size (pts) to consider when computing the modal body font.
* Tiny labels from diagrams, footnote markers, and superscripts are excluded
* so they don't skew the modal toward small sizes.
*/
const MIN_BODY_FONT_SIZE = 7;
/**
* Compute the most frequent font size among text boxes, ignoring very small
* text that likely comes from diagrams, footnotes, or superscripts.
*/
function modalFontSize(textBoxes: TextBox[]): number {
const counts = new Map<number, number>();
for (const tb of textBoxes) {
const size = Math.round((tb.fontSize ?? 0) * 10) / 10;
if (size < MIN_BODY_FONT_SIZE) continue;
counts.set(size, (counts.get(size) ?? 0) + 1);
}
let modal = 0;
let maxCount = 0;
for (const [size, count] of counts) {
if (count > maxCount) {
maxCount = count;
modal = size;
}
}
return modal;
}
/** Group free text boxes into horizontal lines, sorted top-to-bottom. */
function groupFreeTextIntoLines(textBoxes: TextBox[]): RenderLine[] {
if (textBoxes.length === 0) return [];
const sorted = [...textBoxes].sort((a, b) => {
const ya = (a.bounds.top + a.bounds.bottom) / 2;
const yb = (b.bounds.top + b.bounds.bottom) / 2;
const dy = yb - ya;
if (Math.abs(dy) > TEXT_LINE_Y_TOLERANCE) return dy;
return a.bounds.left - b.bounds.left;
});
const lines: RenderLine[] = [];
let curParts = [sorted[0].text];
let curBoxes = [sorted[0]];
let curY = (sorted[0].bounds.top + sorted[0].bounds.bottom) / 2;
let curTopY = curY;
let curFontSize = sorted[0].fontSize;
let curIsBold = sorted[0].isBold;
const finishLine = () => {
let isTabular = false;
for (let j = 1; j < curBoxes.length; j++) {
if (curBoxes[j].bounds.left - curBoxes[j - 1].bounds.right > TABULAR_X_GAP) {
isTabular = true;
break;
}
}
lines.push({
text: curParts.join(" "),
topY: curTopY,
fontSize: curFontSize,
isBold: curIsBold,
isTabular,
});
};
for (let i = 1; i < sorted.length; i++) {
const box = sorted[i];
const cy = (box.bounds.top + box.bounds.bottom) / 2;
if (Math.abs(cy - curY) <= TEXT_LINE_Y_TOLERANCE) {
curParts.push(box.text);
curBoxes.push(box);
curFontSize = Math.max(curFontSize, box.fontSize);
curIsBold = curIsBold || box.isBold;
} else {
finishLine();
curParts = [box.text];
curBoxes = [box];
curY = cy;
curTopY = cy;
curFontSize = box.fontSize;
curIsBold = box.isBold;
}
}
finishLine();
return lines;
}
/** Determine markdown heading prefix based on font size relative to body. */
function headingPrefix(fontSize: number, bodyFontSize: number, isBold: boolean): string {
if (bodyFontSize <= 0) return "";
const ratio = fontSize / bodyFontSize;
// Large headings (>2x body size)
if (ratio >= 2.0) return "# ";
// Medium headings (~1.5x body size)
if (ratio >= 1.4) return "## ";
// Small headings (bold and slightly larger)
if (ratio >= 1.1 && isBold) return "### ";
return "";
}
// ---------------------------------------------------------------------------
// Block merging
// ---------------------------------------------------------------------------
/** Merge consecutive blocks with the same heading prefix (wrapped headings). */
function mergeConsecutiveHeadings(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
if (blocks.length === 0) return [];
const HEADING_RE = /^(#{1,6} )/;
const maxGap = Math.max(bodyFS * 3, 30);
const merged: ContentBlock[] = [];
let cur: ContentBlock = { ...blocks[0] };
for (let i = 1; i < blocks.length; i++) {
const next = blocks[i];
const curMatch = cur.content.match(HEADING_RE);
const nextMatch = next.content.match(HEADING_RE);
const gap = cur.topY - next.topY;
if (curMatch && nextMatch && curMatch[1] === nextMatch[1] && gap <= maxGap) {
cur = {
topY: cur.topY,
content: `${cur.content} ${next.content.slice(nextMatch[1].length)}`,
isTabular: cur.isTabular || next.isTabular,
};
} else {
merged.push(cur);
cur = { ...next };
}
}
merged.push(cur);
return merged;
}
/**
* Merge consecutive plain-text blocks that are wrapped lines of the same paragraph.
*/
function mergeParagraphWraps(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
if (blocks.length === 0 || bodyFS <= 0) return blocks;
const HEADING_RE = /^#{1,6} /;
const SENTENCE_END_RE = /[.!?…)\]]\s*$/;
const maxGap = bodyFS * 2.0;
const MIN_WRAP_LENGTH = 25;
const merged: ContentBlock[] = [];
let cur: WrapBlock = { ...blocks[0], lastTopY: blocks[0].topY };
for (let i = 1; i < blocks.length; i++) {
const next = blocks[i];
const curIsBody = !HEADING_RE.test(cur.content) && !cur.content.startsWith("|");
const nextIsBody = !HEADING_RE.test(next.content) && !next.content.startsWith("|");
const gap = cur.lastTopY - next.topY;
const isWrap =
curIsBody &&
nextIsBody &&
!cur.isTabular &&
!next.isTabular &&
gap > 0 &&
gap <= maxGap &&
cur.content.length > MIN_WRAP_LENGTH &&
!SENTENCE_END_RE.test(cur.content);
if (isWrap) {
cur = {
topY: cur.topY,
lastTopY: next.topY,
content: `${cur.content.trimEnd()} ${next.content.trimStart()}`,
isTabular: false,
};
} else {
merged.push({ topY: cur.topY, content: cur.content });
cur = { ...next, lastTopY: next.topY };
}
}
merged.push({ topY: cur.topY, content: cur.content });
return merged;
}
/** Remove page number blocks near the bottom of the page. */
function removePageNumbers(blocks: ContentBlock[]): ContentBlock[] {
const PAGE_NUM_RE = /^(?:#{1,6}\s*)?\d+\s*$/;
const BOTTOM_Y = 120;
return blocks.filter((block, idx) => {
const isBottom = idx >= blocks.length - 3;
const isLowY = block.topY <= BOTTOM_Y;
const isPageNum = PAGE_NUM_RE.test(block.content.trim());
return !(isBottom && isLowY && isPageNum);
});
}
// ---------------------------------------------------------------------------
// Detached first-column table reconstruction
// ---------------------------------------------------------------------------
/**
* Fix tables where the first column was emitted as free text blocks
* around a markdown table containing only the right-side columns.
*
* Detects: a plain-text header line with (N+1) tokens above an N-column
* markdown table, plus short label lines whose count matches the table's
* logical row count. Reconstructs into a proper (N+1)-column table.
*/
function normalizeDetachedFirstColumnTables(blocks: ContentBlock[]): ContentBlock[] {
const HEADING_RE = /^#{1,6}\s/;
const isTableBlock = (text: string) => text.trimStart().startsWith("|");
const isPlainBlock = (text: string) => !HEADING_RE.test(text) && !isTableBlock(text);
const isShortLabel = (text: string) => {
const t = text.trim();
return t.length > 0 && t.length <= 40;
};
const splitTokens = (text: string) =>
text
.trim()
.split(/[ \t]+/)
.filter(Boolean);
const replacements = new Map<number, string>();
const remove = new Set<number>();
for (let tableIdx = 0; tableIdx < blocks.length; tableIdx++) {
if (remove.has(tableIdx)) continue;
const tableBlock = blocks[tableIdx];
if (!isTableBlock(tableBlock.content)) continue;
const tableLines = tableBlock.content
.split("\n")
.map(line => line.trim())
.filter(line => line.startsWith("|"));
const dataRows = tableLines
.filter(line => !/^\|\s*[-: ]+\|/.test(line))
.map(parsePipeRow)
.filter(row => row.length > 0);
if (dataRows.length === 0) continue;
const cols = dataRows[0].length;
if (cols < 2 || dataRows.some(row => row.length !== cols)) continue;
// Expand by <br> count to get logical row count
const logicalRows: string[][] = [];
for (const row of dataRows) {
const splitCells = row.map(cell => cell.split("<br>").map(p => p.trim()));
const rowSpan = Math.max(...splitCells.map(parts => parts.length));
for (let k = 0; k < rowSpan; k++) {
logicalRows.push(splitCells.map(parts => parts[k] ?? ""));
}
}
if (logicalRows.length < 2) continue;
// Find header with (cols + 1) non-numeric tokens
let headerIdx = -1;
let headerTokens: string[] = [];
for (let i = Math.max(0, tableIdx - 4); i <= tableIdx - 1; i++) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text)) continue;
const tokens = splitTokens(text);
if (tokens.length === cols + 1 && tokens.every(tok => !/[0-9]/.test(tok))) {
headerIdx = i;
headerTokens = tokens;
}
}
if (headerIdx < 0) continue;
// Collect short label lines above/below table
const aboveLabels: Array<{ idx: number; text: string }> = [];
for (let i = tableIdx - 1; i > headerIdx; i--) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text) || !isShortLabel(text)) break;
aboveLabels.push({ idx: i, text });
}
aboveLabels.reverse();
const belowLabels: Array<{ idx: number; text: string }> = [];
for (let i = tableIdx + 1; i < blocks.length; i++) {
const text = normalizeFullWidthAscii(blocks[i].content).trim();
if (!isPlainBlock(text) || !isShortLabel(text)) break;
belowLabels.push({ idx: i, text });
}
const labels = [...aboveLabels, ...belowLabels];
if (labels.length !== logicalRows.length) continue;
// Reconstruct the full table
const normalizedLines: string[] = [];
normalizedLines.push(`| ${headerTokens.join(" | ")} |`);
normalizedLines.push(`| ${Array.from({ length: cols + 1 }, () => "---").join(" | ")} |`);
for (let r = 0; r < logicalRows.length; r++) {
normalizedLines.push(`| ${labels[r].text} | ${logicalRows[r].join(" | ")} |`);
}
replacements.set(tableIdx, normalizedLines.join("\n"));
remove.add(headerIdx);
for (const label of labels) remove.add(label.idx);
}
if (replacements.size === 0 && remove.size === 0) return blocks;
const out: ContentBlock[] = [];
for (let i = 0; i < blocks.length; i++) {
if (remove.has(i)) continue;
const replaced = replacements.get(i);
if (replaced) {
out.push({ topY: blocks[i].topY, content: replaced });
} else {
out.push(blocks[i]);
}
}
return out;
}
// ---------------------------------------------------------------------------
// Public API
// ---------------------------------------------------------------------------
/**
* Render one page's content: free text and tables interleaved top-to-bottom.
*/
export function renderPageContent(
freeTextBoxes: TextBox[],
tables: TableGrid[],
imageBlocks: Array<{ topY: number; markdown: string }> = [],
allTextBoxes?: TextBox[],
): string {
const blocks: ContentBlock[] = [];
// Use ALL text boxes (before table/diagram filtering) for modal font size,
// so that diagram labels released as free text don't skew the body size.
const bodyFS = modalFontSize(allTextBoxes ?? freeTextBoxes);
// Free text lines
for (const line of groupFreeTextIntoLines(freeTextBoxes)) {
const prefix = headingPrefix(line.fontSize, bodyFS, line.isBold);
blocks.push({
topY: line.topY,
content: prefix + line.text,
isTabular: prefix === "" && line.isTabular,
});
}
// Tables
for (const table of tables) {
const md = renderTableToMarkdown(table);
if (md.length > 0) {
blocks.push({ topY: table.topY, content: md });
}
}
// Images
for (const img of imageBlocks) {
blocks.push({ topY: img.topY, content: img.markdown });
}
// Sort top-to-bottom (higher Y = higher on page = comes first)
blocks.sort((a, b) => b.topY - a.topY);
const cleaned = removePageNumbers(blocks);
const headingsMerged = mergeConsecutiveHeadings(cleaned, bodyFS);
const merged = mergeParagraphWraps(headingsMerged, bodyFS);
const normalized = normalizeDetachedFirstColumnTables(merged);
return normalized
.map(b => b.content)
.join("\n\n")
.trim();
}
@@ -0,0 +1,84 @@
// Adapted from markit-ai (MIT). See ../../NOTICE.
/** Bounding box in PDF coordinate space (origin = bottom-left). */
export type Bounds = {
left: number;
right: number;
/** Higher value = higher on the page. */
top: number;
bottom: number;
};
/** A text fragment with position and font metadata. */
export type TextBox = {
id: string;
text: string;
bounds: Bounds;
pageNumber: number;
/** Dominant font size in points. */
fontSize: number;
/** True if rendered bold (font name or rendering mode). */
isBold: boolean;
};
/** A horizontal or vertical line segment extracted from vector graphics. */
export type Segment = {
id: string;
x1: number;
y1: number;
x2: number;
y2: number;
};
/** A single cell in a resolved table grid. */
export type TableCell = {
row: number;
col: number;
text: string;
rowSpan: number;
colSpan: number;
};
/** A resolved table grid ready for markdown rendering. */
export type TableGrid = {
pageNumber: number;
rows: number;
cols: number;
cells: TableCell[];
warnings: string[];
/** Top Y coordinate (PDF space: larger = higher on page). */
topY: number;
/** True for tables detected without vector borders. */
isBorderless: boolean;
};
/** An image/diagram region detected on a page. */
export type ImageRegion = {
id: string;
pageNumber: number;
/** Bounding box in mupdf coordinates (top-left origin). */
bbox: {
x: number;
y: number;
w: number;
h: number;
};
/** Y position in PDF coordinates (bottom-left) for ordering. */
topY: number;
};
/** Result of extracting content from a single PDF page. */
export type PageContent = {
pageNumber: number;
textBoxes: TextBox[];
segments: Segment[];
images: ImageRegion[];
};
/** A block of rendered content (text paragraph or table). */
export type ContentBlock = {
topY: number;
content: string;
/** True if this line has wide gaps between text boxes (column headers). */
isTabular?: boolean;
};
@@ -0,0 +1,325 @@
// Adapted from markit-ai (MIT). See ../NOTICE.
import * as path from "node:path";
import { XMLParser } from "fast-xml-parser";
import { unzip, unzipText } from "../../utils/zip";
import type { ConversionResult, Converter, StreamInfo } from "../types";
const EXTENSIONS = [".pptx"];
const MIMETYPES = ["application/vnd.openxmlformats-officedocument.presentationml.presentation"];
/** A text value: bare string/number, or a `{ "#text" }` node when the element carries attributes. */
type XmlText = string | number | { "#text"?: string };
interface TextRun {
"a:t"?: XmlText;
}
interface Paragraph {
"a:r"?: TextRun | TextRun[];
}
interface TextBody {
"a:p"?: Paragraph | Paragraph[];
}
interface CNvPr {
"@_name": string;
}
interface Placeholder {
"@_type": string;
}
interface NvPr {
"p:ph"?: Placeholder;
}
interface NvSpPr {
"p:cNvPr"?: CNvPr;
"p:nvPr"?: NvPr;
}
interface NvPicPr {
"p:cNvPr"?: CNvPr;
}
interface Shape {
"p:txBody"?: TextBody;
"p:nvSpPr"?: NvSpPr;
}
interface Blip {
"@_r:embed": string;
}
interface BlipFill {
"a:blip"?: Blip;
}
interface Picture {
"p:blipFill"?: BlipFill;
"p:nvSpPr"?: NvSpPr;
"p:nvPicPr"?: NvPicPr;
}
interface TableCell {
"a:txBody"?: TextBody;
}
interface TableRow {
"a:tc"?: TableCell | TableCell[];
}
interface Table {
"a:tr"?: TableRow | TableRow[];
}
interface GraphicData {
"a:tbl"?: Table;
}
interface Graphic {
"a:graphicData"?: GraphicData;
}
interface GraphicFrame {
"a:graphic"?: Graphic;
}
interface SpTree {
"p:sp"?: Shape | Shape[];
"p:pic"?: Picture | Picture[];
"p:graphicFrame"?: GraphicFrame | GraphicFrame[];
}
interface CSld {
"p:spTree"?: SpTree;
}
interface SlideDoc {
"p:sld"?: { "p:cSld"?: CSld };
}
interface NotesDoc {
"p:notes"?: { "p:cSld"?: CSld };
}
interface SldId {
"@_r:id": string;
}
interface PresentationDoc {
"p:presentation"?: { "p:sldIdLst"?: { "p:sldId"?: SldId | SldId[] } };
}
interface Relationship {
"@_Id": string;
"@_Target": string;
}
interface RelationshipsDoc {
Relationships?: { Relationship?: Relationship | Relationship[] };
}
export class PptxConverter implements Converter {
name = "pptx";
accepts(streamInfo: StreamInfo): boolean {
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true;
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true;
return false;
}
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
const entries = unzip(input);
const parser = new XMLParser({
ignoreAttributes: false,
attributeNamePrefix: "@_",
textNodeName: "#text",
processEntities: { maxTotalExpansions: 1_000_000 },
});
// Get slide order from presentation.xml
const presXml = unzipText(entries, "ppt/presentation.xml");
if (!presXml) throw new Error("Invalid PPTX: missing presentation.xml");
const pres = parser.parse(presXml) as PresentationDoc;
const sldIdList = pres["p:presentation"]?.["p:sldIdLst"]?.["p:sldId"];
const sldIds = Array.isArray(sldIdList) ? sldIdList : sldIdList ? [sldIdList] : [];
// Get relationship mappings
const relsXml = unzipText(entries, "ppt/_rels/presentation.xml.rels");
const rels = relsXml ? (parser.parse(relsXml) as RelationshipsDoc) : null;
const relList = rels?.Relationships?.Relationship;
const relArray = Array.isArray(relList) ? relList : relList ? [relList] : [];
const relMap = new Map<string, string>();
for (const r of relArray) {
relMap.set(r["@_Id"], r["@_Target"]);
}
// Map slide IDs to file paths in order
const slidePaths: string[] = [];
for (const sld of sldIds) {
const rId = sld["@_r:id"];
const target = relMap.get(rId);
if (target) slidePaths.push(`ppt/${target}`);
}
// If we couldn't resolve from rels, fall back to finding slide files
if (slidePaths.length === 0) {
const slideFiles = Object.keys(entries)
.filter(f => /^ppt\/slides\/slide\d+\.xml$/.test(f))
.sort((a, b) => {
const na = parseInt(a.match(/slide(\d+)/)?.[1] || "0", 10);
const nb = parseInt(b.match(/slide(\d+)/)?.[1] || "0", 10);
return na - nb;
});
slidePaths.push(...slideFiles);
}
const imageDir = streamInfo.imageDir;
const sections: string[] = [];
let imageCount = 0;
for (let i = 0; i < slidePaths.length; i++) {
const slideXml = unzipText(entries, slidePaths[i]);
if (!slideXml) continue;
const slide = parser.parse(slideXml) as SlideDoc;
const spTree = slide["p:sld"]?.["p:cSld"]?.["p:spTree"];
if (!spTree) continue;
// Parse slide-level rels for image references
const slideRelsPath = `${slidePaths[i].replace("slides/slide", "slides/_rels/slide")}.rels`;
const slideRelsXml = unzipText(entries, slideRelsPath);
const slideRelMap = new Map<string, string>();
if (slideRelsXml) {
const slideRels = parser.parse(slideRelsXml) as RelationshipsDoc;
const relItems = toList(slideRels?.Relationships?.Relationship);
for (const r of relItems) {
slideRelMap.set(r["@_Id"], r["@_Target"]);
}
}
const slideLines = [`<!-- Slide ${i + 1} -->`];
const shapes = spTree["p:sp"];
const shapeList = Array.isArray(shapes) ? shapes : shapes ? [shapes] : [];
let isTitle = true;
for (const shape of shapeList) {
const text = this.extractText(shape);
if (!text) continue;
if (isTitle) {
slideLines.push(`# ${text}`);
isTitle = false;
} else {
slideLines.push(text);
}
}
// Extract embedded images
const pics = toList(spTree["p:pic"]);
for (const pic of pics) {
const blipFill = pic["p:blipFill"];
const rEmbed = blipFill?.["a:blip"]?.["@_r:embed"];
if (!rEmbed) continue;
const target = slideRelMap.get(rEmbed);
if (!target) continue;
// Resolve relative target against slide directory
const imagePath = target.startsWith("/") ? target.slice(1) : `ppt/slides/${target}`;
// Normalize path (e.g. ppt/slides/../media/image1.png → ppt/media/image1.png)
const normalizedPath = imagePath
.split("/")
.reduce<string[]>((parts, seg) => {
if (seg === "..") parts.pop();
else parts.push(seg);
return parts;
}, [])
.join("/");
const buf = entries[normalizedPath];
if (!buf) continue;
imageCount++;
const name =
pic["p:nvSpPr"]?.["p:cNvPr"]?.["@_name"] ||
pic["p:nvPicPr"]?.["p:cNvPr"]?.["@_name"] ||
`image_${imageCount}`;
if (imageDir) {
try {
const ext = normalizedPath.split(".").pop() || "png";
const filename = `slide${i + 1}_${imageCount}.${ext}`;
const filepath = path.join(imageDir, filename);
await Bun.write(filepath, buf);
slideLines.push(`![${name}](${filepath})`);
} catch {
slideLines.push(`<!-- image: ${name} (slide ${i + 1}) -->`);
}
} else {
slideLines.push(`<!-- image: ${name} (slide ${i + 1}) -->`);
}
}
// Tables
const graphicFrames = spTree["p:graphicFrame"];
const gfList = Array.isArray(graphicFrames) ? graphicFrames : graphicFrames ? [graphicFrames] : [];
for (const gf of gfList) {
const table = this.extractTable(gf);
if (table) slideLines.push(table);
}
// Slide notes
const noteFile = slidePaths[i].replace("slides/slide", "notesSlides/notesSlide");
const noteXml = unzipText(entries, noteFile);
if (noteXml) {
const note = parser.parse(noteXml) as NotesDoc;
const noteSpTree = note["p:notes"]?.["p:cSld"]?.["p:spTree"];
if (noteSpTree) {
const noteShapes = noteSpTree["p:sp"];
const noteList = Array.isArray(noteShapes) ? noteShapes : noteShapes ? [noteShapes] : [];
const noteTexts: string[] = [];
for (const ns of noteList) {
// Skip slide image placeholder
const phType = ns["p:nvSpPr"]?.["p:nvPr"]?.["p:ph"]?.["@_type"];
if (phType === "sldImg") continue;
const t = this.extractText(ns);
if (t) noteTexts.push(t);
}
if (noteTexts.length > 0) {
slideLines.push("\n### Notes:");
slideLines.push(noteTexts.join("\n"));
}
}
}
sections.push(slideLines.join("\n"));
}
return { markdown: sections.join("\n\n").trim() };
}
extractText(shape: Shape): string {
const txBody = shape["p:txBody"];
if (!txBody) return "";
const paragraphs = txBody["a:p"];
const pList = Array.isArray(paragraphs) ? paragraphs : paragraphs ? [paragraphs] : [];
const lines: string[] = [];
for (const p of pList) {
const runs = p["a:r"];
const rList = Array.isArray(runs) ? runs : runs ? [runs] : [];
const parts: string[] = [];
for (const r of rList) {
const t = r["a:t"];
if (t != null) parts.push(typeof t === "object" ? t["#text"] || "" : String(t));
}
if (parts.length > 0) lines.push(parts.join(""));
}
return lines.join("\n").trim();
}
extractTable(gf: GraphicFrame): string | null {
const tbl = gf?.["a:graphic"]?.["a:graphicData"]?.["a:tbl"];
if (!tbl) return null;
const rows = tbl["a:tr"];
const rowList = Array.isArray(rows) ? rows : rows ? [rows] : [];
if (rowList.length === 0) return null;
const mdRows: string[][] = [];
for (const row of rowList) {
const cells = row["a:tc"];
const cellList = Array.isArray(cells) ? cells : cells ? [cells] : [];
const cellTexts: string[] = [];
for (const cell of cellList) {
const txBody = cell["a:txBody"];
if (!txBody) {
cellTexts.push("");
continue;
}
const paragraphs = txBody["a:p"];
const pList = Array.isArray(paragraphs) ? paragraphs : paragraphs ? [paragraphs] : [];
const parts: string[] = [];
for (const p of pList) {
const runs = p["a:r"];
const rList = Array.isArray(runs) ? runs : runs ? [runs] : [];
for (const r of rList) {
const t = r["a:t"];
if (t != null) parts.push(typeof t === "object" ? t["#text"] || "" : String(t));
}
}
cellTexts.push(parts.join(" "));
}
mdRows.push(cellTexts);
}
if (mdRows.length === 0) return null;
const [header, ...body] = mdRows;
const lines: string[] = [];
lines.push(`| ${header.join(" | ")} |`);
lines.push(`| ${header.map(() => "---").join(" | ")} |`);
for (const row of body) {
while (row.length < header.length) row.push("");
lines.push(`| ${row.join(" | ")} |`);
}
return lines.join("\n");
}
}
function toList<T>(val: T | T[] | undefined): T[] {
if (!val) return [];
return Array.isArray(val) ? val : [val];
}
@@ -0,0 +1,173 @@
// Adapted from markit-ai (MIT). See ../NOTICE.
import { XMLParser } from "fast-xml-parser";
import { unzip, unzipText } from "../../utils/zip";
import type { ConversionResult, Converter, StreamInfo } from "../types";
const EXTENSIONS = [".xlsx"];
const MIMETYPES = ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"];
/** A text value: bare string/number, or a `{ "#text" }` node when the element carries attributes. */
type XmlText = string | number | { "#text"?: string };
interface RichTextRun {
t?: XmlText;
}
interface StringItem {
t?: XmlText;
r?: RichTextRun | RichTextRun[];
}
interface Cell {
"@_t"?: string;
v?: string | number;
is?: StringItem;
}
interface Row {
c?: Cell | Cell[];
}
interface WorksheetDoc {
worksheet?: { sheetData?: { row?: Row | Row[] } };
}
interface Sheet {
"@_name": string;
"@_r:id": string;
}
interface WorkbookDoc {
workbook?: { sheets?: { sheet?: Sheet | Sheet[] } };
}
interface SharedStringsDoc {
sst?: { si?: StringItem | StringItem[] };
}
interface Relationship {
"@_Id": string;
"@_Target": string;
}
interface RelationshipsDoc {
Relationships?: { Relationship?: Relationship | Relationship[] };
}
export class XlsxConverter implements Converter {
name = "xlsx";
accepts(streamInfo: StreamInfo): boolean {
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true;
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true;
return false;
}
async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
const entries = unzip(input);
const parser = new XMLParser({
ignoreAttributes: false,
attributeNamePrefix: "@_",
textNodeName: "#text",
processEntities: { maxTotalExpansions: 1_000_000 },
});
// Parse shared strings
const ssXml = unzipText(entries, "xl/sharedStrings.xml");
const ss = ssXml ? (parser.parse(ssXml) as SharedStringsDoc) : null;
const siList = ss?.sst?.si;
const shared = toArray(siList);
// Parse workbook for sheet names
const wbXml = unzipText(entries, "xl/workbook.xml");
if (!wbXml) throw new Error("Invalid XLSX: missing workbook.xml");
const wb = parser.parse(wbXml) as WorkbookDoc;
const sheets = toArray(wb.workbook?.sheets?.sheet);
// Parse workbook rels to map rIds to sheet files
const relsXml = unzipText(entries, "xl/_rels/workbook.xml.rels");
const rels = relsXml ? (parser.parse(relsXml) as RelationshipsDoc) : null;
const relList = toArray(rels?.Relationships?.Relationship);
const relMap = new Map<string, string>();
for (const r of relList) {
relMap.set(r["@_Id"], r["@_Target"]);
}
const sections: string[] = [];
for (const sheet of sheets) {
const sheetName = sheet["@_name"];
const rId = sheet["@_r:id"];
const target = relMap.get(rId);
if (!target) continue;
const sheetPath = target.startsWith("/") ? target.slice(1) : `xl/${target}`;
const sheetXml = unzipText(entries, sheetPath);
if (!sheetXml) continue;
const parsed = parser.parse(sheetXml) as WorksheetDoc;
const rows = toArray(parsed.worksheet?.sheetData?.row);
if (rows.length === 0) continue;
// Extract all rows as string arrays
const tableRows: string[][] = [];
for (const row of rows) {
const cells = toArray(row.c);
const values: string[] = [];
for (const cell of cells) {
values.push(this.getCellValue(cell, shared));
}
tableRows.push(values);
}
if (tableRows.length === 0) continue;
// Normalize column count
const maxCols = Math.max(...tableRows.map(r => r.length));
for (const row of tableRows) {
while (row.length < maxCols) row.push("");
}
sections.push(`## ${sheetName}`);
const [header, ...body] = tableRows;
const lines: string[] = [];
lines.push(`| ${header.join(" | ")} |`);
lines.push(`| ${header.map(() => "---").join(" | ")} |`);
for (const row of body) {
lines.push(`| ${row.join(" | ")} |`);
}
sections.push(lines.join("\n"));
}
return { markdown: sections.join("\n\n") };
}
getCellValue(cell: Cell, shared: StringItem[]): string {
// Shared string
if (cell["@_t"] === "s") {
return this.getSharedString(shared, Number(cell.v));
}
// Inline string
if (cell["@_t"] === "inlineStr") {
const is = cell.is;
if (!is) return "";
if (is.t != null) return textValue(is.t);
if (is.r)
return toArray(is.r)
.map(r => textValue(r.t))
.join("");
return "";
}
// Boolean
if (cell["@_t"] === "b") {
return cell.v === 1 || cell.v === "1" ? "TRUE" : "FALSE";
}
// Number or formula result
if (cell.v != null) return String(cell.v);
return "";
}
getSharedString(shared: StringItem[], idx: number): string {
const si = shared[idx];
if (!si) return "";
// Simple text
if (si.t != null) return textValue(si.t);
// Rich text runs
if (si.r) {
return toArray(si.r)
.map(r => textValue(r.t))
.join("");
}
return "";
}
}
function textValue(t: XmlText | undefined): string {
if (t == null) return "";
if (typeof t === "object") return t["#text"] || "";
return String(t);
}
function toArray<T>(val: T | T[] | undefined): T[] {
if (!val) return [];
return Array.isArray(val) ? val : [val];
}
@@ -0,0 +1,2 @@
export * from "./registry";
export * from "./types";
@@ -0,0 +1,59 @@
// Adapted from markit-ai (MIT). See ./NOTICE.
import * as path from "node:path";
import { DocxConverter } from "./converters/docx";
import { EpubConverter } from "./converters/epub";
import { PdfConverter } from "./converters/pdf";
import { PptxConverter } from "./converters/pptx";
import { XlsxConverter } from "./converters/xlsx";
import type { ConversionResult, Converter, MarkitOptions, StreamInfo } from "./types";
/**
* In-house document → markdown engine (replaces the `markit-ai` package).
*
* Only the document converters omp routes are registered (pdf, docx, pptx,
* xlsx, epub). The first converter whose `accepts()` returns true and whose
* `convert()` succeeds wins.
*/
export class Markit {
readonly #converters: readonly Converter[];
readonly #options: MarkitOptions;
constructor(options: MarkitOptions = {}) {
this.#options = options;
this.#converters = [
new PdfConverter(),
new DocxConverter(),
new PptxConverter(),
new XlsxConverter(),
new EpubConverter(),
];
}
async convertFile(filePath: string, extra?: { imageDir?: string }): Promise<ConversionResult> {
const buffer = Buffer.from(await Bun.file(filePath).arrayBuffer());
const streamInfo: StreamInfo = {
localPath: filePath,
extension: path.extname(filePath).toLowerCase(),
filename: path.basename(filePath),
...extra,
};
return this.convert(buffer, streamInfo);
}
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
const errors: { converter: string; error: Error }[] = [];
for (const converter of this.#converters) {
if (!converter.accepts(streamInfo)) continue;
try {
return await converter.convert(input, streamInfo, this.#options);
} catch (err) {
errors.push({ converter: converter.name, error: err instanceof Error ? err : new Error(String(err)) });
}
}
if (errors.length > 0) {
const details = errors.map(e => ` ${e.converter}: ${e.error.message}`).join("\n");
throw new Error(`Conversion failed:\n${details}`);
}
throw new Error(`Unsupported format: ${streamInfo.extension || streamInfo.mimetype || "unknown"}`);
}
}
+35
View File
@@ -0,0 +1,35 @@
// Adapted from markit-ai (MIT). See ./NOTICE.
export interface StreamInfo {
mimetype?: string;
extension?: string;
charset?: string;
filename?: string;
localPath?: string;
url?: string;
/** Directory to write extracted images/diagrams. */
imageDir?: string;
}
export interface ConversionResult {
markdown: string;
title?: string;
}
export interface MarkitOptions {
/** Describe an image, return markdown. Receives raw bytes and mimetype. */
describe?: (image: Buffer, mimetype: string) => Promise<string>;
/** Transcribe audio, return text. Receives raw bytes and mimetype. */
transcribe?: (audio: Buffer, mimetype: string) => Promise<string>;
/** Extra instructions appended to the image description prompt. */
prompt?: string;
}
export interface Converter {
/** Human-readable name for error messages. */
name: string;
/** Quick check: can this converter handle the given stream? */
accepts(streamInfo: StreamInfo): boolean;
/** Convert the source to markdown. */
convert(input: Buffer, streamInfo: StreamInfo, options?: MarkitOptions): Promise<ConversionResult>;
}
@@ -1,7 +1,7 @@
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { inflateSync, strFromU8 } from "fflate";
import { bytesToText, inflateRaw } from "../utils/zip";
import { formatBytes } from "./render-utils";
import { ToolError } from "./tool-errors";
@@ -417,7 +417,7 @@ function parseZipCentralDirectory(
throw new ToolError("Invalid ZIP archive: truncated central directory entry");
}
const rawPath = strFromU8(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0);
const rawPath = bytesToText(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0);
const normalizedPath = normalizeArchiveEntryPath(rawPath);
if (normalizedPath) {
const values = readZip64EntryValues(
@@ -490,7 +490,7 @@ async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number):
}
try {
return inflateSync(compressedBytes, { out: new Uint8Array(uncompressedSize) });
return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize));
} catch (error) {
throw new ToolError(error instanceof Error ? error.message : String(error));
}
+10 -57
View File
@@ -51,34 +51,9 @@ const CONVERTIBLE_MIMES = new Set([
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
"application/rtf",
"application/epub+zip",
"image/png",
"image/jpeg",
"image/gif",
"image/webp",
"audio/mpeg",
"audio/wav",
"audio/ogg",
]);
const CONVERTIBLE_EXTENSIONS = new Set([
".pdf",
".doc",
".docx",
".ppt",
".pptx",
".xls",
".xlsx",
".rtf",
".epub",
".png",
".jpg",
".jpeg",
".gif",
".webp",
".mp3",
".wav",
".ogg",
]);
const CONVERTIBLE_EXTENSIONS = new Set([".pdf", ".doc", ".docx", ".ppt", ".pptx", ".xls", ".xlsx", ".rtf", ".epub"]);
const NOTEBOOK_MIMES = new Set(["application/x-ipynb+json"]);
const NOTEBOOK_EXTENSIONS = new Set([".ipynb"]);
@@ -1144,45 +1119,25 @@ async function renderUrl(
notes.push(
`Image MIME type ${imageMimeType} is unsupported for inline model serialization; returning text metadata only`,
);
const shouldTryConvertibleFallback = isConvertible(mime, extHint);
if (shouldTryConvertibleFallback) {
notes.push("Attempting binary conversion fallback for unsupported image MIME type");
} else {
notes.push("Falling back to textual rendering from initial response");
}
skipConvertibleBinaryRetry = !shouldTryConvertibleFallback;
notes.push("Falling back to textual rendering from initial response");
skipConvertibleBinaryRetry = true;
} else {
const binary = await fetchBinary(finalUrl, timeout, signal);
if (binary.ok) {
notes.push("Fetched image binary");
const conversionExtension = getExtensionHint(finalUrl, binary.contentDisposition) || extHint;
let convertedText: string | null = null;
const converted = await convertWithMarkit(binary.buffer, conversionExtension, timeout, signal);
if (converted.ok) {
if (converted.content.trim().length > 50) {
notes.push("Converted with markit");
convertedText = converted.content;
} else {
notes.push("markit conversion produced no usable output");
}
} else if (converted.error) {
notes.push(`markit conversion failed: ${converted.error}`);
} else {
notes.push("markit conversion failed");
}
if (binary.buffer.byteLength > MAX_INLINE_IMAGE_SOURCE_BYTES) {
notes.push(
`Image exceeds inline source limit (${binary.buffer.byteLength} bytes > ${MAX_INLINE_IMAGE_SOURCE_BYTES} bytes)`,
);
const output = finalizeOutput(
convertedText ?? `Fetched image content (${imageMimeType}), but it is too large to inline render.`,
`Fetched image content (${imageMimeType}), but it is too large to inline render.`,
);
return {
url,
finalUrl,
contentType: imageMimeType,
method: convertedText ? "markit" : "image-too-large",
method: "image-too-large",
content: output.content,
fetchedAt,
truncated: output.truncated,
@@ -1199,15 +1154,13 @@ async function renderUrl(
if (!isDecodedImage) {
notes.push(`Fetched payload could not be decoded as ${imageMimeType}; returning text metadata only`);
const output = finalizeOutput(
convertedText ??
rawContent ??
`Fetched payload was labeled ${imageMimeType}, but bytes were not a valid image.`,
rawContent ?? `Fetched payload was labeled ${imageMimeType}, but bytes were not a valid image.`,
);
return {
url,
finalUrl,
contentType: imageMimeType,
method: convertedText ? "markit" : "image-invalid",
method: "image-invalid",
content: output.content,
fetchedAt,
truncated: output.truncated,
@@ -1219,13 +1172,13 @@ async function renderUrl(
`Image exceeds inline output limit after resize (${resized.buffer.length} bytes > ${MAX_INLINE_IMAGE_OUTPUT_BYTES} bytes)`,
);
const output = finalizeOutput(
convertedText ?? `Fetched image content (${imageMimeType}), but it is too large to inline render.`,
`Fetched image content (${imageMimeType}), but it is too large to inline render.`,
);
return {
url,
finalUrl,
contentType: imageMimeType,
method: convertedText ? "markit" : "image-too-large",
method: "image-too-large",
content: output.content,
fetchedAt,
truncated: output.truncated,
@@ -1234,7 +1187,7 @@ async function renderUrl(
}
const dimensionNote = formatDimensionNote(resized);
let imageSummary = convertedText ?? `Fetched image content (${resized.mimeType}).`;
let imageSummary = `Fetched image content (${resized.mimeType}).`;
if (dimensionNote) {
imageSummary += `\n${dimensionNote}`;
}
+4 -10
View File
@@ -65,12 +65,6 @@ import { toolResult } from "./tool-result";
const LOOSE_HASHLINE_HEADER_RE = /^\s*\[[^#\r\n]+#[^ \t\r\n]*\]\s*$/;
const EXECUTABLE_NOTICE = "[Notice: Made executable via chmod +x]";
let fflateModulePromise: Promise<typeof import("fflate")> | undefined;
async function loadFflate(): Promise<typeof import("fflate")> {
if (!fflateModulePromise) fflateModulePromise = import("fflate");
return fflateModulePromise;
}
const writeSchema = type({
path: type("string").describe("file path"),
content: type("string").describe("file content"),
@@ -387,8 +381,8 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
if (resolvedArchivePath.exists) {
try {
const bytes = await Bun.file(resolvedArchivePath.absolutePath).bytes();
const { unzipSync } = await loadFflate();
const existing = unzipSync(new Uint8Array(bytes));
const { unzip } = await import("../utils/zip");
const existing = unzip(new Uint8Array(bytes));
for (const [entryPath, data] of Object.entries(existing)) {
zipEntries[entryPath.replace(/\\/g, "/")] = data;
}
@@ -400,8 +394,8 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
zipEntries[resolvedArchivePath.archiveSubPath] = new TextEncoder().encode(content);
try {
const { zipSync } = await loadFflate();
const zipBuffer = zipSync(zipEntries);
const { zip } = await import("../utils/zip");
const zipBuffer = zip(zipEntries);
await Bun.write(tmpPath, zipBuffer);
await fs.rename(tmpPath, finalPath);
} catch (error) {
+10 -9
View File
@@ -1,5 +1,5 @@
import { logger, untilAborted } from "@oh-my-pi/pi-utils";
import type { Markit, StreamInfo } from "markit-ai";
import type { Markit, StreamInfo } from "../markit";
import { ToolAbortError } from "../tools/tool-errors";
export interface MarkitConversionResult {
@@ -23,26 +23,27 @@ interface MuPdfWasmModuleConfig {
printErr?: (...values: unknown[]) => void;
}
declare global {
var $libmupdf_wasm_Module: MuPdfWasmModuleConfig | undefined;
}
function logMuPdfWasmOutput(stream: "stdout" | "stderr", values: unknown[]): void {
const message = values.length === 1 && typeof values[0] === "string" ? values[0] : values.map(String).join(" ");
logger.debug("mupdf wasm output", { stream, message });
}
// `$libmupdf_wasm_Module` is declared globally (as `any`) by the mupdf package.
// Install print hooks before the WASM module initializes so its stdout/stderr
// route to the file logger instead of corrupting the TUI.
function installMuPdfWasmLogger(): void {
const moduleConfig = globalThis.$libmupdf_wasm_Module ?? {};
moduleConfig.print = (...values) => logMuPdfWasmOutput("stdout", values);
moduleConfig.printErr = (...values) => logMuPdfWasmOutput("stderr", values);
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
moduleConfig.print = (...values: unknown[]) => logMuPdfWasmOutput("stdout", values);
moduleConfig.printErr = (...values: unknown[]) => logMuPdfWasmOutput("stderr", values);
globalThis.$libmupdf_wasm_Module = moduleConfig;
}
installMuPdfWasmLogger();
let markit: () => Markit | Promise<Markit> = async () => {
const promise = import("markit-ai").then(({ Markit }) => {
// Lazy: keep the document engine (mammoth/fflate/mupdf) off the startup
// import graph — it loads only when a document is first converted.
const promise = import("../markit").then(({ Markit }) => {
const instance = new Markit();
markit = () => instance;
return instance;
@@ -0,0 +1,83 @@
import TurndownService from "turndown";
import { gfm } from "turndown-plugin-gfm";
type TurndownListParent = {
nodeName: string;
getAttribute(name: string): string | null;
children: ArrayLike<unknown>;
};
/**
* Build a Turndown instance configured for GFM with the fixes omp relies on:
* `~~strikethrough~~`, unescaped heading periods, and single-space list markers.
*
* Shared by the web scrapers (HTML → markdown) and the markit document engine
* (`src/markit`). The rule set must stay identical across both call sites.
*/
export function createTurndown(): TurndownService {
const turndown = new TurndownService({
headingStyle: "atx",
codeBlockStyle: "fenced",
bulletListMarker: "-",
});
turndown.use(gfm);
// GFM spec uses ~~ (double tilde), not ~ (single)
turndown.addRule("strikethrough", {
filter: ["del", "s", "strike"],
replacement(content) {
return `~~${content}~~`;
},
});
// Unescape the backslash turndown inserts before periods in headings ("1." -> "1\.")
turndown.addRule("heading", {
filter: ["h1", "h2", "h3", "h4", "h5", "h6"],
replacement(content, node) {
const level = Number(node.nodeName.charAt(1));
const prefix = "#".repeat(level);
const cleaned = content.replace(/\\([.])/g, "$1").trim();
return `\n\n${prefix} ${cleaned}\n\n`;
},
});
// Single space after the marker (turndown hardcodes three)
turndown.addRule("listItem", {
filter: "li",
replacement(content, node, options) {
const body = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n ");
const parent = node.parentNode as unknown as TurndownListParent | null;
let prefix = `${options.bulletListMarker} `;
if (parent?.nodeName === "OL") {
const start = parent.getAttribute("start");
const index = Array.prototype.indexOf.call(parent.children, node);
prefix = `${(start ? Number(start) : 1) + index}. `;
}
return prefix + body + (node.nextSibling ? "\n" : "");
},
});
return turndown;
}
/**
* Normalize HTML tables so turndown-plugin-gfm can render them:
* - strip `<p>` tags inside `<td>`/`<th>` cells (joining paragraphs with a space)
* - wrap the first row in `<thead>` when missing
*/
export function normalizeTablesHtml(html: string): string {
let result = html.replace(
/<(td|th)([^>]*)>([\s\S]*?)<\/(td|th)>/gi,
(_match, tag: string, attrs: string, inner: string, closeTag: string) => {
const stripped = inner
.replace(/^\s*<p>/i, "")
.replace(/<\/p>\s*$/i, "")
.replace(/<\/p>\s*<p>/gi, " ");
return `<${tag}${attrs}>${stripped}</${closeTag}>`;
},
);
result = result.replace(
/<table([^>]*)>\s*(?:<tbody>\s*)?(<tr[\s\S]*?<\/tr>)([\s\S]*?)<\/(?:tbody>\s*<\/)?table>/gi,
(_match, attrs: string, firstRow: string, rest: string) => {
const theadRow = firstRow.replace(/<td/gi, "<th").replace(/<\/td>/gi, "</th>");
return `<table${attrs}><thead>${theadRow}</thead><tbody>${rest}</tbody></table>`;
},
);
return result;
}
+29
View File
@@ -0,0 +1,29 @@
// The single ZIP/DEFLATE boundary for the codebase. This is the ONLY module
// that imports `fflate`; the markit document converters, the write tool, and
// the archive reader all go through here so there is exactly one ZIP
// implementation to reason about. Do not import `fflate` (or another archive
// library) anywhere else.
import type { Unzipped } from "fflate";
import { inflateSync, strFromU8 } from "fflate";
export type { Unzipped } from "fflate";
export { unzipSync as unzip, zipSync as zip } from "fflate";
/** Read a single ZIP entry as UTF-8 text, or `undefined` when the entry is absent. */
export function unzipText(entries: Unzipped, entryPath: string): string | undefined {
const data = entries[entryPath];
return data ? strFromU8(data) : undefined;
}
/**
* Inflate a raw DEFLATE stream (a single deflate-compressed ZIP member). Pass a
* preallocated `into` buffer when the uncompressed size is known up front.
*/
export function inflateRaw(bytes: Uint8Array, into?: Uint8Array): Uint8Array {
return into ? inflateSync(bytes, { out: into }) : inflateSync(bytes);
}
/** Decode raw bytes as text — UTF-8 by default, latin1 when `latin1` is set. */
export function bytesToText(bytes: Uint8Array, latin1?: boolean): string {
return strFromU8(bytes, latin1);
}
@@ -243,58 +243,15 @@ export async function loadPage(url: string, options: LoadPageOptions = {}): Prom
/** Module-level Turndown instance — built lazily on first use. */
let turndownPromise: Promise<TurndownService> | undefined;
type TurndownListParent = {
nodeName: string;
getAttribute(name: string): string | null;
children: ArrayLike<unknown>;
};
function getTurndown(): Promise<TurndownService> {
turndownPromise ||= initTurndown();
return turndownPromise;
}
async function initTurndown(): Promise<TurndownService> {
const [{ default: TurndownService }, { gfm }] = await Promise.all([
import("turndown"),
import("turndown-plugin-gfm"),
]);
const turndown = new TurndownService({
headingStyle: "atx",
codeBlockStyle: "fenced",
bulletListMarker: "-",
});
turndown.use(gfm);
turndown.addRule("strikethrough", {
filter: ["del", "s", "strike"],
replacement(content) {
return `~~${content}~~`;
},
});
turndown.addRule("heading", {
filter: ["h1", "h2", "h3", "h4", "h5", "h6"],
replacement(content, node) {
const level = Number(node.nodeName.charAt(1));
const prefix = "#".repeat(level);
const cleaned = content.replace(/\\([.])/g, "$1").trim();
return `\n\n${prefix} ${cleaned}\n\n`;
},
});
turndown.addRule("listItem", {
filter: "li",
replacement(content, node, options) {
content = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n ");
const parent = node.parentNode as unknown as TurndownListParent | null;
let prefix = `${options.bulletListMarker} `;
if (parent?.nodeName === "OL") {
const start = parent.getAttribute("start");
const index = Array.prototype.indexOf.call(parent.children, node);
prefix = `${(start ? Number(start) : 1) + index}. `;
}
return prefix + content + (node.nextSibling ? "\n" : "");
},
});
return turndown;
// Lazy import keeps turndown/turndown-plugin-gfm off the startup graph.
const { createTurndown } = await import("../../utils/turndown");
return createTurndown();
}
/**
@@ -0,0 +1,160 @@
/**
* Runtime coverage for the in-house markit document engine (src/markit), which
* replaced the `markit-ai` package. Each format is generated in-memory via the
* shared zip util (src/utils/zip) — no external fixtures — and converted through
* the public wrapper (src/utils/markit), locking: docx text, xlsx tables, pptx
* slides, epub metadata+spine, shared HTML-table normalization, image
* extraction, the nested/relative zip path resolution (the JSZip→fflate
* regression surface), and the unsupported-format error contract.
*/
import { describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { convertBufferWithMarkit, convertFileWithMarkit } from "@oh-my-pi/pi-coding-agent/utils/markit";
import { zip } from "@oh-my-pi/pi-coding-agent/utils/zip";
const enc = (s: string): Uint8Array => new TextEncoder().encode(s);
const WML = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
function makeDocx(bodyXml: string): Uint8Array {
return zip({
"[Content_Types].xml": enc(
`<?xml version="1.0"?><Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"><Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/><Override PartName="/word/document.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"/></Types>`,
),
"_rels/.rels": enc(
`<?xml version="1.0"?><Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships"><Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="word/document.xml"/></Relationships>`,
),
"word/document.xml": enc(
`<?xml version="1.0"?><w:document xmlns:w="${WML}"><w:body>${bodyXml}</w:body></w:document>`,
),
});
}
describe("markit converters", () => {
it("converts docx paragraphs to markdown", async () => {
const docx = makeDocx(
`<w:p><w:r><w:t>First paragraph.</w:t></w:r></w:p><w:p><w:r><w:t>Second paragraph.</w:t></w:r></w:p>`,
);
const result = await convertBufferWithMarkit(docx, ".docx");
expect(result.ok).toBe(true);
expect(result.content).toBe("First paragraph.\n\nSecond paragraph.");
});
it("converts xlsx sheets to markdown tables", async () => {
const xlsx = zip({
"xl/workbook.xml": enc(
`<?xml version="1.0"?><workbook xmlns:r="r"><sheets><sheet name="People" sheetId="1" r:id="rId1"/></sheets></workbook>`,
),
"xl/_rels/workbook.xml.rels": enc(
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="worksheets/sheet1.xml"/></Relationships>`,
),
"xl/worksheets/sheet1.xml": enc(
`<?xml version="1.0"?><worksheet><sheetData><row><c t="inlineStr"><is><t>Name</t></is></c><c t="inlineStr"><is><t>Age</t></is></c></row><row><c t="inlineStr"><is><t>Alice</t></is></c><c><v>30</v></c></row></sheetData></worksheet>`,
),
});
const result = await convertBufferWithMarkit(xlsx, ".xlsx");
expect(result.ok).toBe(true);
expect(result.content).toContain("## People");
expect(result.content).toContain("| Name | Age |");
expect(result.content).toContain("| --- | --- |");
expect(result.content).toContain("| Alice | 30 |");
});
it("reads an xlsx worksheet through an absolute (/-prefixed) rel target", async () => {
const xlsx = zip({
"xl/workbook.xml": enc(
`<?xml version="1.0"?><workbook xmlns:r="r"><sheets><sheet name="S" sheetId="1" r:id="rId1"/></sheets></workbook>`,
),
"xl/_rels/workbook.xml.rels": enc(
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="/xl/worksheets/sheet1.xml"/></Relationships>`,
),
"xl/worksheets/sheet1.xml": enc(
`<?xml version="1.0"?><worksheet><sheetData><row><c t="inlineStr"><is><t>Header</t></is></c></row><row><c><v>42</v></c></row></sheetData></worksheet>`,
),
});
const result = await convertBufferWithMarkit(xlsx, ".xlsx");
expect(result.ok).toBe(true);
expect(result.content).toContain("| Header |");
expect(result.content).toContain("| 42 |");
});
it("converts pptx slides with a title heading and body text", async () => {
const pptx = zip({
"ppt/presentation.xml": enc(
`<?xml version="1.0"?><p:presentation xmlns:p="p" xmlns:r="r"><p:sldIdLst><p:sldId id="256" r:id="rId1"/></p:sldIdLst></p:presentation>`,
),
"ppt/_rels/presentation.xml.rels": enc(
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="slides/slide1.xml"/></Relationships>`,
),
"ppt/slides/slide1.xml": enc(
`<?xml version="1.0"?><p:sld xmlns:p="p" xmlns:a="a"><p:cSld><p:spTree><p:sp><p:txBody><a:p><a:r><a:t>The Title</a:t></a:r></a:p></p:txBody></p:sp><p:sp><p:txBody><a:p><a:r><a:t>Body line</a:t></a:r></a:p></p:txBody></p:sp></p:spTree></p:cSld></p:sld>`,
),
});
const result = await convertBufferWithMarkit(pptx, ".pptx");
expect(result.ok).toBe(true);
expect(result.content).toContain("# The Title");
expect(result.content).toContain("Body line");
});
it("extracts a pptx image through a ../media relative rel target into imageDir", async () => {
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "markit-pptx-"));
try {
const pptx = zip({
"ppt/presentation.xml": enc(
`<?xml version="1.0"?><p:presentation xmlns:p="p" xmlns:r="r"><p:sldIdLst><p:sldId id="256" r:id="rId1"/></p:sldIdLst></p:presentation>`,
),
"ppt/_rels/presentation.xml.rels": enc(
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="slides/slide1.xml"/></Relationships>`,
),
"ppt/slides/slide1.xml": enc(
`<?xml version="1.0"?><p:sld xmlns:p="p" xmlns:a="a"><p:cSld><p:spTree><p:pic><p:nvPicPr><p:cNvPr name="Pic1"/></p:nvPicPr><p:blipFill><a:blip r:embed="rId2"/></p:blipFill></p:pic></p:spTree></p:cSld></p:sld>`,
),
"ppt/slides/_rels/slide1.xml.rels": enc(
`<?xml version="1.0"?><Relationships><Relationship Id="rId2" Target="../media/image1.png"/></Relationships>`,
),
"ppt/media/image1.png": new Uint8Array([137, 80, 78, 71, 13, 10, 26, 10]),
});
const pptxPath = path.join(dir, "deck.pptx");
const imageDir = path.join(dir, "imgs");
await Bun.write(pptxPath, pptx);
const result = await convertFileWithMarkit(pptxPath, undefined, { imageDir });
expect(result.ok).toBe(true);
const written = await fs.readdir(imageDir);
expect(written).toHaveLength(1);
expect(result.content).toContain(`](${path.join(imageDir, written[0]!)})`);
} finally {
await fs.rm(dir, { recursive: true, force: true });
}
});
it("converts epub spine, normalizes HTML tables, and resolves a non-root OPF basePath", async () => {
const epub = zip({
"META-INF/container.xml": enc(
`<?xml version="1.0"?><container><rootfiles><rootfile full-path="OEBPS/content.opf"/></rootfiles></container>`,
),
"OEBPS/content.opf": enc(
`<?xml version="1.0"?><package><metadata xmlns:dc="dc"><dc:title>Nested Book</dc:title><dc:creator>Ada</dc:creator></metadata><manifest><item id="c1" href="text/ch1.xhtml"/></manifest><spine><itemref idref="c1"/></spine></package>`,
),
"OEBPS/text/ch1.xhtml": enc(
`<html><body><h2>Chapter One</h2><p>Body text.</p><table><tr><td>A</td><td>B</td></tr><tr><td>1</td><td>2</td></tr></table></body></html>`,
),
});
const result = await convertBufferWithMarkit(epub, ".epub");
expect(result.ok).toBe(true);
expect(result.content).toContain("**Title:** Nested Book");
expect(result.content).toContain("**Authors:** Ada");
expect(result.content).toContain("## Chapter One");
expect(result.content).toContain("Body text.");
// normalizeTablesHtml promotes the first row to a header so GFM renders a table.
expect(result.content).toContain("| A | B |");
expect(result.content).toContain("| --- | --- |");
});
it("reports an unsupported format instead of emitting garbage", async () => {
const rtf = enc("{\\rtf1\\ansi binary-ish}");
const result = await convertBufferWithMarkit(rtf, ".rtf");
expect(result.ok).toBe(false);
expect(result.error).toContain("Unsupported format");
});
});
+2 -2
View File
@@ -18,8 +18,8 @@ import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
import { DEFAULT_FILE_LIMIT, MULTI_FILE_PER_FILE_MATCHES, SearchTool } from "@oh-my-pi/pi-coding-agent/tools/search";
import * as toolTimeouts from "@oh-my-pi/pi-coding-agent/tools/tool-timeouts";
import { WriteTool } from "@oh-my-pi/pi-coding-agent/tools/write";
import { unzip } from "@oh-my-pi/pi-coding-agent/utils/zip";
import { $which, Snowflake } from "@oh-my-pi/pi-utils";
import { unzipSync } from "fflate";
// Helper to extract text from content blocks
function getTextOutput(result: any): string {
@@ -931,7 +931,7 @@ describe("Coding Agent Tools", () => {
`Successfully wrote ${content.length} bytes to ${path.basename(archivePath)}:pkg/README.md`,
);
const unzipped = unzipSync(new Uint8Array(fs.readFileSync(archivePath)));
const unzipped = unzip(new Uint8Array(fs.readFileSync(archivePath)));
expect(new TextDecoder().decode(unzipped["pkg/README.md"])).toBe(content);
expect(new TextDecoder().decode(unzipped["pkg/src/index.ts"])).toBe("export const archiveValue = 1;\n");
});
@@ -7,10 +7,10 @@ import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
import { zip } from "@oh-my-pi/pi-coding-agent/utils/zip";
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { zipSync } from "fflate";
function makeSession(testDir: string): ToolSession {
const sessionFile = path.join(testDir, "session.jsonl");
@@ -112,7 +112,7 @@ describe("read URL binary dispatch", () => {
});
it("lists a remote zip instead of dumping decoded bytes", async () => {
const zipBytes = zipSync({
const zipBytes = zip({
"root.txt": Buffer.from("root file\n"),
"nested/data.txt": Buffer.from("nested file\n"),
});
@@ -167,11 +167,6 @@ describe("read tool URL handling", () => {
ok: true,
buffer: imageBytes,
});
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
ok: false,
content: "",
error: "markit unavailable",
});
vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
buffer: imageBytes,
mimeType: "image/png",
@@ -222,11 +217,7 @@ describe("read tool URL handling", () => {
ok: true,
buffer: new Uint8Array([137, 80, 78, 71]),
});
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
ok: false,
content: "",
error: "markit unavailable",
});
const convertSpy = vi.spyOn(scraperUtils, "convertWithMarkit");
const result = await tool.execute("fetch-image-resized", { path: "https://example.com/image.png" });
const imageBlock = result.content.find(
@@ -240,52 +231,9 @@ describe("read tool URL handling", () => {
expect(imageBlock?.data).toBe("cmVzaXplZA==");
expect(textBlock?.type).toBe("text");
expect(textBlock?.text).toContain("displayed at 1000x500");
expect(convertSpy).not.toHaveBeenCalled();
});
it("keeps markit extracted text for image responses", async () => {
const session = createSession();
const tool = new ReadTool(session);
const extractedText = "Converted image text content that is definitely longer than fifty characters.";
vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
buffer: new Uint8Array([1, 2, 3]),
mimeType: "image/png",
originalWidth: 100,
originalHeight: 100,
width: 100,
height: 100,
wasResized: false,
get data() {
return "aW1hZ2U=";
},
});
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
ok: true,
status: 200,
contentType: "image/png",
finalUrl: "https://example.com/image.png",
content: "",
});
vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({
ok: true,
buffer: new Uint8Array([137, 80, 78, 71]),
});
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
ok: true,
content: extractedText,
});
const result = await tool.execute("fetch-image-with-ocr", { path: "https://example.com/image.png" });
const textBlock = result.content.find(content => content.type === "text");
const imageBlock = result.content.find(
(content): content is { type: "image"; data: string; mimeType: string } => content.type === "image",
);
expect(result.details?.method).toBe("image");
expect(textBlock?.type).toBe("text");
expect(textBlock?.text).toContain(extractedText);
expect(imageBlock?.mimeType).toBe("image/png");
expect(imageBlock?.data).toBe("aW1hZ2U=");
});
it("falls back to text-only output for unsupported image MIME types", async () => {
const session = createSession();
const tool = new ReadTool(session);
@@ -309,39 +257,6 @@ describe("read tool URL handling", () => {
expect(textBlock?.text).toContain("<svg></svg>");
});
it("uses binary conversion fallback for unsupported image MIME when extension is convertible", async () => {
const session = createSession();
const tool = new ReadTool(session);
const convertedText = "Converted image text from markit fallback with sufficient length to pass threshold.";
const fetchBinarySpy = vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({
ok: true,
buffer: new Uint8Array([255, 216, 255, 224]),
});
const convertSpy = vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
ok: true,
content: convertedText,
});
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
ok: true,
status: 200,
contentType: "image/jpg",
finalUrl: "https://example.com/image.jpg",
content: "\u0000\u0001garbage",
});
const result = await tool.execute("fetch-image-jpg-fallback", { path: "https://example.com/image.jpg" });
const imageBlock = result.content.find(content => content.type === "image");
const textBlock = result.content.find(content => content.type === "text");
expect(result.details?.method).toBe("markit");
expect(fetchBinarySpy).toHaveBeenCalledTimes(1);
expect(convertSpy).toHaveBeenCalledTimes(1);
expect(result.details?.notes).toContain("Attempting binary conversion fallback for unsupported image MIME type");
expect(imageBlock).toBeUndefined();
expect(textBlock?.type).toBe("text");
expect(textBlock?.text).toContain(convertedText);
});
it("does not treat text/html at .png paths as inline images", async () => {
const session = createSession();
const tool = new ReadTool(session);
@@ -404,11 +319,6 @@ describe("read tool URL handling", () => {
ok: true,
buffer: new Uint8Array([60, 104, 116, 109, 108]),
});
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
ok: false,
content: "",
error: "conversion failed",
});
vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
buffer: new Uint8Array([60, 104, 116, 109, 108]),
mimeType: "image/png",