feat(coding-agent): vendor relevant parts of markit
This commit is contained in:
@@ -55,7 +55,6 @@ out.html
|
||||
pi-*.html
|
||||
|
||||
# Generated files
|
||||
packages/coding-agent/src/internal-urls/docs-index.generated.ts
|
||||
packages/coding-agent/src/export/html/tool-views.generated.js
|
||||
packages/natives/npm/
|
||||
/runs/
|
||||
|
||||
@@ -60,7 +60,6 @@
|
||||
"!**/vendor/**/*",
|
||||
"!**/node_modules/**/*",
|
||||
"!**/test-sessions.ts",
|
||||
"!**/docs-index.generated.ts",
|
||||
"!**/agent_pb.ts",
|
||||
"!.worktrees/**/*",
|
||||
"!.wt/**/*"
|
||||
|
||||
@@ -98,11 +98,13 @@
|
||||
"arktype": "catalog:",
|
||||
"chalk": "catalog:",
|
||||
"diff": "catalog:",
|
||||
"fast-xml-parser": "catalog:",
|
||||
"fflate": "catalog:",
|
||||
"handlebars": "catalog:",
|
||||
"linkedom": "catalog:",
|
||||
"lru-cache": "catalog:",
|
||||
"markit-ai": "catalog:",
|
||||
"mammoth": "catalog:",
|
||||
"mupdf": "catalog:",
|
||||
"puppeteer-core": "catalog:",
|
||||
"turndown": "catalog:",
|
||||
"turndown-plugin-gfm": "catalog:",
|
||||
@@ -277,7 +279,6 @@
|
||||
"version": "16.0.7",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
"beautiful-mermaid": "catalog:",
|
||||
"handlebars": "catalog:",
|
||||
"winston": "catalog:",
|
||||
"winston-daily-rotate-file": "catalog:",
|
||||
@@ -311,7 +312,6 @@
|
||||
},
|
||||
"patchedDependencies": {
|
||||
"@ark/schema@0.56.0": "patches/@ark%2Fschema@0.56.0.patch",
|
||||
"beautiful-mermaid@1.1.3": "patches/beautiful-mermaid@1.1.3.patch",
|
||||
},
|
||||
"catalog": {
|
||||
"@agentclientprotocol/sdk": "0.25.0",
|
||||
@@ -355,11 +355,11 @@
|
||||
"@typescript/native-preview": "7.0.0-dev.20260609.1",
|
||||
"@xterm/headless": "^6.0.0",
|
||||
"arktype": "^2.2.0",
|
||||
"beautiful-mermaid": "^1.1.3",
|
||||
"chalk": "^5.6.2",
|
||||
"chart.js": "^4.5.1",
|
||||
"date-fns": "^4.4.0",
|
||||
"diff": "^9.0.0",
|
||||
"fast-xml-parser": "^5.9.0",
|
||||
"fastembed": "2.1.0",
|
||||
"fflate": "0.8.3",
|
||||
"ghostty-web": "^0.4.0",
|
||||
@@ -368,8 +368,9 @@
|
||||
"lint-staged": "^17.0.7",
|
||||
"lru-cache": "11.5.1",
|
||||
"lucide-react": "^1.17.0",
|
||||
"mammoth": "^1.12.0",
|
||||
"marked": "^18.0.5",
|
||||
"markit-ai": "0.5.3",
|
||||
"mupdf": "^1.27.0",
|
||||
"onnxruntime-node": "1.26.0",
|
||||
"partial-json": "^0.1.7",
|
||||
"postcss": "^8.5.15",
|
||||
@@ -460,8 +461,6 @@
|
||||
|
||||
"@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.5.0", "", { "os": "win32", "cpu": "x64" }, "sha512-VT/lF+GId+67j8aDfLkxdxNoVApsPSTbyAtB3jJq0IWTrY77WXfbPfpngxq0bA6JCEv/7k8C9qWjDRKRznDlyw=="],
|
||||
|
||||
"@borewit/text-codec": ["@borewit/text-codec@0.2.2", "", {}, "sha512-DDaRehssg1aNrH4+2hnj1B7vnUGEjU6OIlyRdkMd0aUdIUvKXrJfXsy8LVtXAy7DRvYVluWbMspsRhz2lcW0mQ=="],
|
||||
|
||||
"@bufbuild/protobuf": ["@bufbuild/protobuf@2.12.0", "", {}, "sha512-B/XlCaFIP8LOwzo+bz5uFzATYokcwCKQcghqnlfwSmM5eX/qTkvDBnDPs+gXtX/RyjxJ4DRikECcPJbyALA8FA=="],
|
||||
|
||||
"@colors/colors": ["@colors/colors@1.6.0", "", {}, "sha512-Ir+AOibqzrIsL6ajt3Rz3LskB7OiMVHqltZmspbW/TJuTVuyOMirVqAkjfY6JISiLHgyNqicAC8AyHHGzNd/dA=="],
|
||||
@@ -860,10 +859,6 @@
|
||||
|
||||
"@tailwindcss/vite": ["@tailwindcss/vite@4.3.1", "", { "dependencies": { "@tailwindcss/node": "4.3.1", "@tailwindcss/oxide": "4.3.1", "tailwindcss": "4.3.1" }, "peerDependencies": { "vite": "^5.2.0 || ^6 || ^7 || ^8" } }, "sha512-hItDHuIIlEV61R+faXu66s1K36aTurO/Qw0e45Vskz57gXl9pWOT6eg3zmcEui6CZXddbN7zd41bwmvag4JGwQ=="],
|
||||
|
||||
"@tokenizer/inflate": ["@tokenizer/inflate@0.4.1", "", { "dependencies": { "debug": "^4.4.3", "token-types": "^6.1.1" } }, "sha512-2mAv+8pkG6GIZiF1kNg1jAjh27IDxEPKwdGul3snfztFerfPGI1LjDezZp3i7BElXompqEtPmoPx6c2wgtWsOA=="],
|
||||
|
||||
"@tokenizer/token": ["@tokenizer/token@0.3.0", "", {}, "sha512-OvjF+z51L3ov0OyAU0duzsYuvO01PH7x4t6DJx+guahgTnBHkhJdG7soQeTSFLWN3efnHyibZ4Z8l2EuWwJN3A=="],
|
||||
|
||||
"@ts-morph/common": ["@ts-morph/common@0.29.0", "", { "dependencies": { "minimatch": "^10.0.1", "path-browserify": "^1.0.1", "tinyglobby": "^0.2.14" } }, "sha512-35oUmphHbJvQ/+UTwFNme/t2p3FoKiGJ5auTjjpNTop2dyREspirjMy82PLSC1pnDJ8ah1GU98hwpVt64YXQsg=="],
|
||||
|
||||
"@tybys/wasm-util": ["@tybys/wasm-util@0.10.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg=="],
|
||||
@@ -936,8 +931,6 @@
|
||||
|
||||
"baseline-browser-mapping": ["baseline-browser-mapping@2.10.37", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-girxaJ7WZssDOFhzCGZTDKoTa1gk6A1TbflaYTpykLJ4UU9Fz9kx1aREM8JCuoVHbL8X8T/mJg7w2oYSq72Oig=="],
|
||||
|
||||
"beautiful-mermaid": ["beautiful-mermaid@1.1.3", "", { "dependencies": { "elkjs": "^0.11.0", "entities": "^7.0.1" } }, "sha512-TItrtrAyHp1vwFfFVYauWGrquouk/6SS21Aq3RsxindSYZODcN4xYrPZD6BiZRU+o5mKJzDPz9MUSMvELdylyg=="],
|
||||
|
||||
"before-after-hook": ["before-after-hook@4.0.0", "", {}, "sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ=="],
|
||||
|
||||
"bluebird": ["bluebird@3.4.7", "", {}, "sha512-iD3898SR7sWVRHbiQv+sHUtHnMvC1o3nW5rAcqnq3uOn07DSAppZYUkIGslDz6gXC7HfunPe7YVBgoEJASPcHA=="],
|
||||
@@ -988,8 +981,6 @@
|
||||
|
||||
"colorette": ["colorette@2.0.20", "", {}, "sha512-IfEDxwoWIjkeXL1eXcDiow4UbKjhLdq6/EuSVR9GMN7KVH3r9gQ83e73hsz1Nd1T3ijd5xv1wcWRYO+D6kCI2w=="],
|
||||
|
||||
"commander": ["commander@14.0.3", "", {}, "sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw=="],
|
||||
|
||||
"content-type": ["content-type@2.0.0", "", {}, "sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ=="],
|
||||
|
||||
"convert-source-map": ["convert-source-map@2.0.0", "", {}, "sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg=="],
|
||||
@@ -1034,8 +1025,6 @@
|
||||
|
||||
"electron-to-chromium": ["electron-to-chromium@1.5.372", "", {}, "sha512-M3yhbAlilnwqC8D21t28UCDGHyitShTmmLRU/H+b74P6Ski16Nb9HONYEaVpMj/pwC7BEo5B95FpjODLCWbtfA=="],
|
||||
|
||||
"elkjs": ["elkjs@0.11.1", "", {}, "sha512-zxxR9k+rx5ktMwT/FwyLdPCrq7xN6e4VGGHH8hA01vVYKjTFik7nHOxBnAYtrgYUB1RpAiLvA1/U2YraWxyKKg=="],
|
||||
|
||||
"emnapi": ["emnapi@1.11.1", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-kSRjhIcxjMFsBqk7ORvoc9aA5SBKDmecrtF5RMcmOTao0kD/zamaxsuTxMI8C1//wGUuvE7a+19pCE7AEhGVnA=="],
|
||||
|
||||
"emoji-regex": ["emoji-regex@8.0.0", "", {}, "sha512-MSjYzcWNOA0ewAHpz0MxpYFvwg6yjy1NG3xteoqz644VCo/RPgnr1/GGt+ic3iJTzQ8Eu3TdM14SawnVUmGE6A=="],
|
||||
@@ -1062,8 +1051,6 @@
|
||||
|
||||
"eventemitter3": ["eventemitter3@5.0.4", "", {}, "sha512-mlsTRyGaPBjPedk6Bvw+aqbsXDtoAyAzm5MO7JgU+yVRyMQ5O8bD4Kcci7BS85f93veegeCPkL8R4GLClnjLFw=="],
|
||||
|
||||
"exifr": ["exifr@7.1.3", "", {}, "sha512-g/aje2noHivrRSLbAUtBPWFbxKdKhgj/xr1vATDdUXPOFYJlQ62Ft0oy+72V6XLIpDJfHs6gXLbBLAolqOXYRw=="],
|
||||
|
||||
"fast-string-truncated-width": ["fast-string-truncated-width@3.0.3", "", {}, "sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g=="],
|
||||
|
||||
"fast-string-width": ["fast-string-width@3.0.2", "", { "dependencies": { "fast-string-truncated-width": "^3.0.2" } }, "sha512-gX8LrtNEI5hq8DVUfRQMbr5lpaS4nMIWV+7XEbXk2b8kiQIizgnlr12B4dA3ZEx3308ze0O4Q1R+cHts8kyUJg=="],
|
||||
@@ -1084,8 +1071,6 @@
|
||||
|
||||
"file-stream-rotator": ["file-stream-rotator@0.6.1", "", { "dependencies": { "moment": "^2.29.1" } }, "sha512-u+dBid4PvZw17PmDeRcNOtCP9CCK/9lRN2w+r1xIS7yOL9JFrIBKTvrYsxT4P0pGtThYTn++QS5ChHaUov3+zQ=="],
|
||||
|
||||
"file-type": ["file-type@21.3.4", "", { "dependencies": { "@tokenizer/inflate": "^0.4.1", "strtok3": "^10.3.4", "token-types": "^6.1.1", "uint8array-extras": "^1.4.0" } }, "sha512-Ievi/yy8DS3ygGvT47PjSfdFoX+2isQueoYP1cntFW1JLYAuS4GD7NUPGg4zv2iZfV52uDyk5w5Z0TdpRS6Q1g=="],
|
||||
|
||||
"flatbuffers": ["flatbuffers@25.9.23", "", {}, "sha512-MI1qs7Lo4Syw0EOzUl0xjs2lsoeqFku44KpngfIduHBYvzm8h2+7K8YMQh1JtVVVrUvhLpNwqVi4DERegUJhPQ=="],
|
||||
|
||||
"fn.name": ["fn.name@1.1.0", "", {}, "sha512-GRnmB5gPyJpAhTQdSZTSp9uaPSvl09KoYcMQtsB9rQoOmzs9dH6ffeccH+Z+cv6P68Hu5bC6JjRh4Ah/mHSNRw=="],
|
||||
@@ -1126,8 +1111,6 @@
|
||||
|
||||
"iconv-lite": ["iconv-lite@0.7.2", "", { "dependencies": { "safer-buffer": ">= 2.1.2 < 3.0.0" } }, "sha512-im9DjEDQ55s9fL4EYzOAv0yMqmMBSZp6G0VvFyTMPKWxiSBHUj9NW/qqLmXUwXrrM7AvqSlTCfvqRb0cM8yYqw=="],
|
||||
|
||||
"ieee754": ["ieee754@1.2.1", "", {}, "sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA=="],
|
||||
|
||||
"immediate": ["immediate@3.0.6", "", {}, "sha512-XXOFtyqDjNDAQxVfYxuF7g9Il/IbWmmlQg2MYKOH8ExIT1qg6xc4zyS3HaEEATgs1btfzxq15ciUiY7gjSXRGQ=="],
|
||||
|
||||
"inherits": ["inherits@2.0.4", "", {}, "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ=="],
|
||||
@@ -1210,12 +1193,8 @@
|
||||
|
||||
"marked": ["marked@18.0.5", "", { "bin": { "marked": "bin/marked.js" } }, "sha512-S6GcvALHg6K4ohtu4E7x0a1AqhAjp6cV8KhLSyN9qVapnzJkusVBxZRcIU9AeYsbe6P1hKDusSbEOzGyyuce6w=="],
|
||||
|
||||
"markit-ai": ["markit-ai@0.5.3", "", { "dependencies": { "chalk": "^5.6.2", "commander": "^14.0.3", "exifr": "^7.1.3", "fast-xml-parser": "^5.5.9", "jszip": "^3.10.1", "mammoth": "^1.9.0", "mupdf": "^1.27.0", "music-metadata": "^11.12.3", "rss-parser": "^3.13.0", "turndown": "^7.2.0", "turndown-plugin-gfm": "^1.0.2" }, "bin": { "markit": "dist/main.js" } }, "sha512-h4nhn6a/SNXEdc3kLVtL37TspxjUNCNL0OM7LRWxd389ZByI/B7bjNNgxFdVAT0O+H7ZekSwLdVe/lws1l2AZQ=="],
|
||||
|
||||
"matcher": ["matcher@4.0.0", "", { "dependencies": { "escape-string-regexp": "^4.0.0" } }, "sha512-S6x5wmcDmsDRRU/c2dkccDwQPXoFczc5+HpQ2lON8pnvHlnvHAHj5WlLVvw6n6vNyHuVugYrFohYxbS+pvFpKQ=="],
|
||||
|
||||
"media-typer": ["media-typer@2.0.0", "", {}, "sha512-kOy3OxT2HH39N70UnKgu4NWDZjLOz8W/mfyvniHjRH/DrL3f2pOfvWQ4p60offbbtDAnXWp0v9LfMIqMec269Q=="],
|
||||
|
||||
"merge-anything": ["merge-anything@5.1.7", "", { "dependencies": { "is-what": "^4.1.8" } }, "sha512-eRtbOb1N5iyH0tkQDAoQ4Ipsp/5qSR79Dzrz8hEPxRX10RWWR/iQXdoKmBSRCThY1Fh5EhISDtpSc93fpxUniQ=="],
|
||||
|
||||
"mimic-function": ["mimic-function@5.0.1", "", {}, "sha512-VP79XUPxV2CigYP3jWwAUFSku2aKqBH7uTAapFWCBqutsbmDo96KY5o8uh6U+/YSIn5OxJnXp73beVkpqMIGhA=="],
|
||||
@@ -1240,8 +1219,6 @@
|
||||
|
||||
"mupdf": ["mupdf@1.27.0", "", {}, "sha512-vEPUYwZeu5NgiFLz4e20R7Vp2pNY7szirGEvTxHyQQpQs6ab4DeGdonwT6sH1JZG5EhyHSrojZrZn2/0ee6qZQ=="],
|
||||
|
||||
"music-metadata": ["music-metadata@11.13.0", "", { "dependencies": { "@borewit/text-codec": "^0.2.2", "@tokenizer/token": "^0.3.0", "content-type": "^2.0.0", "debug": "^4.4.3", "file-type": "^21.3.4", "media-typer": "^2.0.0", "strtok3": "^10.3.5", "token-types": "^6.1.2", "uint8array-extras": "^1.5.0", "win-guid": "^0.2.1" } }, "sha512-uXRaov9dfjSpQufXIU7sMxVZnh+FilCQv2mXn+K5EJ/decP3dTWrgvPYa5r6MtRbieNSCE708Da4J0u1UGfQIw=="],
|
||||
|
||||
"mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="],
|
||||
|
||||
"nanoid": ["nanoid@3.3.12", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-ZB9RH/39qpq5Vu6Y+NmUaFhQR6pp+M2Xt76XBnEwDaGcVAqhlvxrl3B2bKS5D3NH3QR76v3aSrKaF/Kiy7lEtQ=="],
|
||||
@@ -1322,16 +1299,12 @@
|
||||
|
||||
"rolldown": ["rolldown@1.0.3", "", { "dependencies": { "@oxc-project/types": "=0.133.0", "@rolldown/pluginutils": "^1.0.0" }, "optionalDependencies": { "@rolldown/binding-android-arm64": "1.0.3", "@rolldown/binding-darwin-arm64": "1.0.3", "@rolldown/binding-darwin-x64": "1.0.3", "@rolldown/binding-freebsd-x64": "1.0.3", "@rolldown/binding-linux-arm-gnueabihf": "1.0.3", "@rolldown/binding-linux-arm64-gnu": "1.0.3", "@rolldown/binding-linux-arm64-musl": "1.0.3", "@rolldown/binding-linux-ppc64-gnu": "1.0.3", "@rolldown/binding-linux-s390x-gnu": "1.0.3", "@rolldown/binding-linux-x64-gnu": "1.0.3", "@rolldown/binding-linux-x64-musl": "1.0.3", "@rolldown/binding-openharmony-arm64": "1.0.3", "@rolldown/binding-wasm32-wasi": "1.0.3", "@rolldown/binding-win32-arm64-msvc": "1.0.3", "@rolldown/binding-win32-x64-msvc": "1.0.3" }, "bin": { "rolldown": "./bin/cli.mjs" } }, "sha512-i00lAJ2ks1BYr7rjNjKC7BcqAS7nVfiT3QX1SI5aY+AFHblCmaUf9OE9dbdzDvW6dJxbi2ZCZiy9v3CcwOiX3g=="],
|
||||
|
||||
"rss-parser": ["rss-parser@3.13.0", "", { "dependencies": { "entities": "^2.0.3", "xml2js": "^0.5.0" } }, "sha512-7jWUBV5yGN3rqMMj7CZufl/291QAhvrrGpDNE4k/02ZchL0npisiYYqULF71jCEKoIiHvK/Q2e6IkDwPziT7+w=="],
|
||||
|
||||
"safe-buffer": ["safe-buffer@5.1.2", "", {}, "sha512-Gd2UZBJDkXlY7GbJxfsE8/nvKkUEU1G38c1siN6QP6a9PT9MmHB8GnpscSmMJSoF8LOIrt8ud/wPtojys4G6+g=="],
|
||||
|
||||
"safe-stable-stringify": ["safe-stable-stringify@2.5.0", "", {}, "sha512-b3rppTKm9T+PsVCBEOUR46GWI7fdOs00VKZ1+9c1EWDaDMvjQc6tUwuFyIprgGgTcWoVHSKrU8H31ZHA2e0RHA=="],
|
||||
|
||||
"safer-buffer": ["safer-buffer@2.1.2", "", {}, "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg=="],
|
||||
|
||||
"sax": ["sax@1.6.0", "", {}, "sha512-6R3J5M4AcbtLUdZmRv2SygeVaM7IhrLXu9BmnOGmmACak8fiUtOsYNWUS4uK7upbmHIBbLBeFeI//477BKLBzA=="],
|
||||
|
||||
"scheduler": ["scheduler@0.27.0", "", {}, "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q=="],
|
||||
|
||||
"semver": ["semver@7.8.4", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-rUCObTnP32Q08R2uuIrt7r9PlEonuTmtuXYcW6s5kjdlj3xbnwe+21yXptAUYcMAABLkYYTtnmzb3w3EDZfueA=="],
|
||||
@@ -1390,8 +1363,6 @@
|
||||
|
||||
"strnum": ["strnum@2.4.0", "", { "dependencies": { "anynum": "^1.0.0" } }, "sha512-sHrVyWWdq28RbhjuJdZsA1SnGRJV6NiXbk6AXBxDOsgAcA+lmpUZCYjOdLBxkXMwis6RRe7dlZt4VlIWFVzkmg=="],
|
||||
|
||||
"strtok3": ["strtok3@10.3.5", "", { "dependencies": { "@tokenizer/token": "^0.3.0" } }, "sha512-ki4hZQfh5rX0QDLLkOCj+h+CVNkqmp/CMf8v8kZpkNVK6jGQooMytqzLZYUVYIZcFZ6yDB70EfD8POcFXiF5oA=="],
|
||||
|
||||
"tailwindcss": ["tailwindcss@4.3.1", "", {}, "sha512-hk+TB1m+K8CYNrP6rjQaq/Y+4Zylwpa87mLYBKCunwnnQ9p+fHb7kmSfGqyEJoxF/O6CDyABWVFEafNSYKll+Q=="],
|
||||
|
||||
"tapable": ["tapable@2.3.3", "", {}, "sha512-uxc/zpqFg6x7C8vOE7lh6Lbda8eEL9zmVm/PLeTPBRhh1xCgdWaQ+J1CUieGpIfm2HdtsUpRv+HshiasBMcc6A=="],
|
||||
@@ -1404,8 +1375,6 @@
|
||||
|
||||
"tinyglobby": ["tinyglobby@0.2.17", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g=="],
|
||||
|
||||
"token-types": ["token-types@6.1.2", "", { "dependencies": { "@borewit/text-codec": "^0.2.1", "@tokenizer/token": "^0.3.0", "ieee754": "^1.2.1" } }, "sha512-dRXchy+C0IgK8WPC6xvCHFRIWYUbqqdEIKPaKo/AcTUNzwLTK6AH7RjdLWsEZcAN/TBdtfUw3PYEgPr5VPr6ww=="],
|
||||
|
||||
"triple-beam": ["triple-beam@1.4.1", "", {}, "sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg=="],
|
||||
|
||||
"ts-morph": ["ts-morph@28.0.0", "", { "dependencies": { "@ts-morph/common": "~0.29.0", "code-block-writer": "^13.0.3" } }, "sha512-Wp3tnZ2bzwxyTZMtgWVzXDfm7lB1Drz+y9DmmYH/L702PQhPyVrp3pkou3yIz4qjS14GY9kcpmLiOOMvl8oG1g=="],
|
||||
@@ -1428,8 +1397,6 @@
|
||||
|
||||
"uhyphen": ["uhyphen@0.2.0", "", {}, "sha512-qz3o9CHXmJJPGBdqzab7qAYuW8kQGKNEuoHFYrBwV6hWIMcpAmxDLXojcHfFr9US1Pe6zUswEIJIbLI610fuqA=="],
|
||||
|
||||
"uint8array-extras": ["uint8array-extras@1.5.0", "", {}, "sha512-rvKSBiC5zqCCiDZ9kAOszZcDvdAHwwIKJG33Ykj43OKcWsnmcBRL09YTU4nOeHZ8Y2a7l1MgTd08SBe9A8Qj6A=="],
|
||||
|
||||
"underscore": ["underscore@1.13.8", "", {}, "sha512-DXtD3ZtEQzc7M8m4cXotyHR+FAS18C64asBYY5vqZexfYryNNnDc02W4hKg3rdQuqOYas1jkseX0+nZXjTXnvQ=="],
|
||||
|
||||
"undici-types": ["undici-types@7.24.6", "", {}, "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg=="],
|
||||
@@ -1448,8 +1415,6 @@
|
||||
|
||||
"webdriver-bidi-protocol": ["webdriver-bidi-protocol@0.4.2", "", {}, "sha512-VSV+fzfChirL3e7jay2yUC7B4HQCGtEWEg/MSSQbK+qWbqeGlRLlXTzPpYr3XGUvbpDHumWZBJxgesg4N7dbtA=="],
|
||||
|
||||
"win-guid": ["win-guid@0.2.1", "", {}, "sha512-gEIQU4mkgl2OPeoNrWflcJFJ3Ae2BPd4eCsHHA/XikslkIVms/nHhvnvzIZV7VLmBvtFlDOzLt9rrZT+n6D67A=="],
|
||||
|
||||
"winston": ["winston@3.19.0", "", { "dependencies": { "@colors/colors": "^1.6.0", "@dabh/diagnostics": "^2.0.8", "async": "^3.2.3", "is-stream": "^2.0.0", "logform": "^2.7.0", "one-time": "^1.0.0", "readable-stream": "^3.4.0", "safe-stable-stringify": "^2.3.1", "stack-trace": "0.0.x", "triple-beam": "^1.3.0", "winston-transport": "^4.9.0" } }, "sha512-LZNJgPzfKR+/J3cHkxcpHKpKKvGfDZVPS4hfJCc4cCG0CgYzvlD6yE/S3CIL/Yt91ak327YCpiF/0MyeZHEHKA=="],
|
||||
|
||||
"winston-daily-rotate-file": ["winston-daily-rotate-file@5.0.0", "", { "dependencies": { "file-stream-rotator": "^0.6.1", "object-hash": "^3.0.0", "triple-beam": "^1.4.1", "winston-transport": "^4.7.0" }, "peerDependencies": { "winston": "^3" } }, "sha512-JDjiXXkM5qvwY06733vf09I2wnMXpZEhxEVOSPenZMii+g7pcDcTBt2MRugnoi8BwVSuCT2jfRXBUy+n1Zz/Yw=="],
|
||||
@@ -1464,8 +1429,6 @@
|
||||
|
||||
"xml-naming": ["xml-naming@0.1.0", "", {}, "sha512-k8KO9hrMyNk6tUWqUfkTEZbezRRpONVOzUTnc97VnCvyj6Tf9lyUR9EDAIeiVLv56jsMcoXEwjW8Kv5yPY52lw=="],
|
||||
|
||||
"xml2js": ["xml2js@0.5.0", "", { "dependencies": { "sax": ">=0.6.0", "xmlbuilder": "~11.0.0" } }, "sha512-drPFnkQJik/O+uPKpqSgr22mpuFHqKdbS835iAQrUC73L2F5WkboIRd63ai/2Yg6I1jzifPFKH2NTK+cfglkIA=="],
|
||||
|
||||
"xmlbuilder": ["xmlbuilder@10.1.1", "", {}, "sha512-OyzrcFLL/nb6fMGHbiRDuPup9ljBycsdCypwuyg5AAHvyWzGfChJpCXMG88AGTIMFhGZ9RccFN1e6lhg3hkwKg=="],
|
||||
|
||||
"y18n": ["y18n@5.0.8", "", {}, "sha512-0pfFzegeDWJHJIAmTLRP2DwHjdF5s7jo9tuztdQxAhINCdvS+3nGINqPd00AphqJR/0LhANUS6/+7SCb98YOfA=="],
|
||||
@@ -1560,8 +1523,6 @@
|
||||
|
||||
"roarr/sprintf-js": ["sprintf-js@1.1.3", "", {}, "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA=="],
|
||||
|
||||
"rss-parser/entities": ["entities@2.2.0", "", {}, "sha512-p92if5Nz619I0w+akJrLZH0MX0Pb5DX39XOwQTtXSdQQOaYH03S1uIQp4mhOZtAXrxq4ViO67YTiLBo2638o9A=="],
|
||||
|
||||
"slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="],
|
||||
|
||||
"string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="],
|
||||
@@ -1570,8 +1531,6 @@
|
||||
|
||||
"wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="],
|
||||
|
||||
"xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="],
|
||||
|
||||
"@babel/helper-compilation-targets/lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="],
|
||||
|
||||
"@huggingface/transformers/onnxruntime-node/global-agent": ["global-agent@3.0.0", "", { "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", "matcher": "^3.0.0", "roarr": "^2.15.3", "semver": "^7.3.2", "serialize-error": "^7.0.1" } }, "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q=="],
|
||||
|
||||
+4
-2
@@ -60,14 +60,16 @@
|
||||
"diff": "^9.0.0",
|
||||
"fflate": "0.8.3",
|
||||
"fastembed": "2.1.0",
|
||||
"fast-xml-parser": "^5.9.0",
|
||||
"ghostty-web": "^0.4.0",
|
||||
"handlebars": "^4.7.9",
|
||||
"linkedom": "^0.18.12",
|
||||
"lint-staged": "^17.0.7",
|
||||
"lru-cache": "11.5.1",
|
||||
"lucide-react": "^1.17.0",
|
||||
"mammoth": "^1.12.0",
|
||||
"marked": "^18.0.5",
|
||||
"markit-ai": "0.5.3",
|
||||
"mupdf": "^1.27.0",
|
||||
"onnxruntime-node": "1.26.0",
|
||||
"partial-json": "^0.1.7",
|
||||
"postcss": "^8.5.15",
|
||||
@@ -169,7 +171,7 @@
|
||||
"lint:py": "ruff check python && ruff format --check python",
|
||||
"fix:py": "ruff check --fix python && ruff format python",
|
||||
"prepublishOnly": "bun run check",
|
||||
"prepare": "bun run generate-docs-index && bun run build-tool-views",
|
||||
"prepare": "bun run build-tool-views",
|
||||
"publish": "bun run prepublishOnly && npm publish -ws --access public",
|
||||
"publish:dry": "bun run prepublishOnly && npm publish -ws --access public --dry-run",
|
||||
"release": "bun scripts/release.ts",
|
||||
|
||||
@@ -1,14 +1,23 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
|
||||
- Updated internal image processing to no longer include metadata text for fetched images
|
||||
- Optimized `omp://` documentation indexing by compressing doc bodies into a lazily-inflated blob
|
||||
- Changed Mermaid fenced-block ASCII rendering to use the first-party vendored renderer in `@oh-my-pi/pi-utils` (`src/vendor/mermaid-ascii`), dropping the `beautiful-mermaid` npm package, its transitive `elkjs` (~3.13MB), and the `beautiful-mermaid` `bun patch`; CJK/emoji width handling and the layout-direction override are preserved.
|
||||
- Changed `omp://` documentation embedding to a gzipped base64 index (`docs-index.generated.txt`, populated at build time and reset afterward) inflated on first read, instead of a ~1.6MB raw TypeScript map. The compiled binary / npm bundle drops ~0.9MB; the dev tree and source checkouts read `docs/` from disk.
|
||||
- Replaced the `markit-ai` package with a vendored in-house document engine (`src/markit`) for the document formats it converts: PDF (via `mupdf`), DOCX (via `mammoth`), and PPTX/XLSX/EPUB (via `fast-xml-parser`). Conversion output is preserved, including the PDF column/table-detection pipeline and HTML-table normalization; `markit-ai`'s unused converters and their dependency tail are gone. Legacy `.doc`/`.ppt`/`.xls`/`.rtf` remain unsupported (a conversion error), as before.
|
||||
- Centralized ZIP handling behind a single `src/utils/zip.ts` (`fflate`): the new document converters, the `write` tool's in-place archive editing, and the `read` tool's ranged archive reader now share one ZIP implementation instead of mixing `jszip` and `fflate`.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the `markit-ai`, `exifr`, `music-metadata`, and direct `jszip` dependencies. As a side effect, fetching an image or audio URL no longer appends EXIF/audio metadata text; image inlining and resizing are unchanged, and document conversion (PDF/DOCX/PPTX/XLSX/EPUB) is unaffected.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed auto-retry after transient model-stream socket closes to replay text/thinking-only partial assistant output, including turns where incomplete tool-call arguments were dropped; completed tool calls still block retry to avoid duplicating tool execution.
|
||||
- Fixed `omp update` reporting `EPERM: operation not permitted, unlink '<binary>.bak'` on Windows when self-replacing a standalone binary, even though the new binary had already been installed. The backed-up old executable is still the running process image and cannot be unlinked until the process exits, so the post-verify backup cleanup is now best-effort, backups use a unique per-attempt name, and stale backups are swept on the next update ([#845](https://github.com/can1357/oh-my-pi/issues/845)).
|
||||
- Fixed plan-mode `Refine plan` so the internal approval abort is hidden and the editor is ready for a follow-up prompt instead of showing `Operation aborted` ([#2971](https://github.com/can1357/oh-my-pi/issues/2971)).
|
||||
- Fixed TUI prompts beginning with shell-style variables such as `$HOME` being misrouted to Python eval; Python shortcuts now require `$ <code>` or `$$ <code>`. ([#2944](https://github.com/can1357/oh-my-pi/issues/2944))
|
||||
|
||||
@@ -35,7 +35,7 @@
|
||||
"check": "biome check . && bun run check:types",
|
||||
"check:types": "tsgo -p tsconfig.json --noEmit",
|
||||
"lint": "biome lint .",
|
||||
"test": "bun test --parallel=4",
|
||||
"test": "bun test --parallel=4 test src",
|
||||
"fix": "biome check --write --unsafe . && bun run format-prompts",
|
||||
"fmt": "biome format --write . && bun run format-prompts",
|
||||
"format-prompts": "bun scripts/format-prompts.ts",
|
||||
@@ -71,11 +71,13 @@
|
||||
"arktype": "catalog:",
|
||||
"chalk": "catalog:",
|
||||
"diff": "catalog:",
|
||||
"fast-xml-parser": "catalog:",
|
||||
"fflate": "catalog:",
|
||||
"handlebars": "catalog:",
|
||||
"linkedom": "catalog:",
|
||||
"lru-cache": "catalog:",
|
||||
"markit-ai": "catalog:",
|
||||
"mammoth": "catalog:",
|
||||
"mupdf": "catalog:",
|
||||
"puppeteer-core": "catalog:",
|
||||
"turndown": "catalog:",
|
||||
"turndown-plugin-gfm": "catalog:",
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
This directory contains an in-house document-to-markdown engine adapted from
|
||||
markit-ai (https://github.com/Michaelliv/markit), used under the MIT License.
|
||||
|
||||
Copyright (c) 2026 Michael Liv
|
||||
|
||||
Only the converters for the document formats omp supports are ported (pdf,
|
||||
docx, pptx, xlsx, epub); the CLI, plugin/provider, and unused converters
|
||||
(html, image, audio, plain-text, rss, github, wikipedia, csv, json, yaml,
|
||||
ipynb, iwork, zip, xml) were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and
|
||||
`.rtf` are routed by the read/fetch tools but have no converter — they surface
|
||||
a conversion error, exactly as upstream markit did. Logic is ported faithfully
|
||||
so conversion output matches the upstream package.
|
||||
|
||||
MIT License
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
@@ -0,0 +1,56 @@
|
||||
// Adapted from markit-ai (MIT). See ../NOTICE.
|
||||
import * as path from "node:path";
|
||||
import mammoth from "mammoth";
|
||||
import { createTurndown, normalizeTablesHtml } from "../../utils/turndown";
|
||||
import type { ConversionResult, Converter, StreamInfo } from "../types";
|
||||
|
||||
const EXTENSIONS = [".docx"];
|
||||
const MIMETYPES = ["application/vnd.openxmlformats-officedocument.wordprocessingml.document"];
|
||||
|
||||
export class DocxConverter implements Converter {
|
||||
name = "docx";
|
||||
|
||||
accepts(streamInfo: StreamInfo): boolean {
|
||||
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) {
|
||||
return true;
|
||||
}
|
||||
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const imageDir = streamInfo.imageDir;
|
||||
let imageCount = 0;
|
||||
const convertImage = imageDir
|
||||
? mammoth.images.imgElement(image => {
|
||||
imageCount++;
|
||||
const ext = (image.contentType?.split("/")[1] || "png").replace("jpeg", "jpg");
|
||||
const filename = `image_${imageCount}.${ext}`;
|
||||
const filepath = path.join(imageDir, filename);
|
||||
return image.read("base64").then(async base64 => {
|
||||
await Bun.write(filepath, Buffer.from(base64, "base64"));
|
||||
return { src: filepath, alt: `image_${imageCount}` };
|
||||
});
|
||||
})
|
||||
: mammoth.images.imgElement(image => {
|
||||
imageCount++;
|
||||
const contentType = image.contentType || "image/png";
|
||||
return image.read("base64").then(base64 => {
|
||||
return {
|
||||
src: `data:${contentType};base64,${base64.slice(0, 0)}`,
|
||||
alt: `image_${imageCount}`,
|
||||
};
|
||||
});
|
||||
});
|
||||
const { value: html } = await mammoth.convertToHtml({ buffer: input }, { convertImage });
|
||||
const turndown = createTurndown();
|
||||
let markdown = turndown.turndown(normalizeTablesHtml(html));
|
||||
// Replace data URI images with comment placeholders when no imageDir
|
||||
if (!imageDir) {
|
||||
markdown = markdown.replace(/!\[([^\]]*)\]\(data:[^)]*\)/g, "<!-- image: $1 -->");
|
||||
}
|
||||
return { markdown: markdown.trim() };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,136 @@
|
||||
// Adapted from markit-ai (MIT). See ../NOTICE.
|
||||
import { XMLParser } from "fast-xml-parser";
|
||||
import { createTurndown, normalizeTablesHtml } from "../../utils/turndown";
|
||||
import { unzip, unzipText } from "../../utils/zip";
|
||||
import type { ConversionResult, Converter, StreamInfo } from "../types";
|
||||
|
||||
const EXTENSIONS = [".epub"];
|
||||
const MIMETYPES = ["application/epub", "application/epub+zip", "application/x-epub+zip"];
|
||||
|
||||
/** A metadata value: a bare string, or a node carrying `#text` and/or array children. */
|
||||
type MetaValue = string | MetaNode;
|
||||
interface MetaNode {
|
||||
"#text"?: string;
|
||||
[index: number]: MetaValue;
|
||||
}
|
||||
interface Metadata {
|
||||
"dc:title"?: MetaValue;
|
||||
"dc:creator"?: MetaValue;
|
||||
"dc:language"?: MetaValue;
|
||||
"dc:publisher"?: MetaValue;
|
||||
"dc:date"?: MetaValue;
|
||||
"dc:description"?: MetaValue;
|
||||
}
|
||||
interface ManifestItem {
|
||||
"@_id": string;
|
||||
"@_href": string;
|
||||
}
|
||||
interface SpineItem {
|
||||
"@_idref": string;
|
||||
}
|
||||
interface OpfDoc {
|
||||
package?: {
|
||||
metadata?: Metadata;
|
||||
manifest?: { item?: ManifestItem | ManifestItem[] };
|
||||
spine?: { itemref?: SpineItem | SpineItem[] };
|
||||
};
|
||||
}
|
||||
interface Rootfile {
|
||||
"@_full-path": string;
|
||||
}
|
||||
interface ContainerDoc {
|
||||
container?: { rootfiles?: { rootfile?: Rootfile | Rootfile[] } };
|
||||
}
|
||||
|
||||
export class EpubConverter implements Converter {
|
||||
name = "epub";
|
||||
|
||||
accepts(streamInfo: StreamInfo): boolean {
|
||||
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true;
|
||||
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const entries = unzip(input);
|
||||
const parser = new XMLParser({
|
||||
ignoreAttributes: false,
|
||||
attributeNamePrefix: "@_",
|
||||
textNodeName: "#text",
|
||||
processEntities: { maxTotalExpansions: 1_000_000 },
|
||||
});
|
||||
// Find content.opf path from container.xml
|
||||
const containerXml = unzipText(entries, "META-INF/container.xml");
|
||||
if (!containerXml) throw new Error("Invalid EPUB: missing container.xml");
|
||||
const container = parser.parse(containerXml) as ContainerDoc;
|
||||
const rootfile = container.container?.rootfiles?.rootfile;
|
||||
const opfPath = Array.isArray(rootfile) ? rootfile[0]["@_full-path"] : rootfile?.["@_full-path"];
|
||||
if (!opfPath) throw new Error("Invalid EPUB: missing rootfile path");
|
||||
// Parse content.opf
|
||||
const opfXml = unzipText(entries, opfPath);
|
||||
if (!opfXml) throw new Error("Invalid EPUB: missing content.opf");
|
||||
const opf = parser.parse(opfXml) as OpfDoc;
|
||||
// Extract metadata
|
||||
const meta: Metadata = opf.package?.metadata ?? {};
|
||||
const metadata: Record<string, string | undefined> = {
|
||||
title: this.getText(meta["dc:title"]),
|
||||
authors: this.getTextArray(meta["dc:creator"]).join(", ") || undefined,
|
||||
language: this.getText(meta["dc:language"]),
|
||||
publisher: this.getText(meta["dc:publisher"]),
|
||||
date: this.getText(meta["dc:date"]),
|
||||
description: this.getText(meta["dc:description"]),
|
||||
};
|
||||
// Build manifest map (id → href)
|
||||
const manifestItems = opf.package?.manifest?.item;
|
||||
const itemList = Array.isArray(manifestItems) ? manifestItems : manifestItems ? [manifestItems] : [];
|
||||
const manifest = new Map<string, string>();
|
||||
for (const item of itemList) {
|
||||
manifest.set(item["@_id"], item["@_href"]);
|
||||
}
|
||||
// Get spine order
|
||||
const spineItems = opf.package?.spine?.itemref;
|
||||
const spineList = Array.isArray(spineItems) ? spineItems : spineItems ? [spineItems] : [];
|
||||
const spineOrder = spineList.map(s => s["@_idref"]);
|
||||
// Resolve file paths
|
||||
const basePath = opfPath.includes("/") ? opfPath.substring(0, opfPath.lastIndexOf("/")) : "";
|
||||
const turndown = createTurndown();
|
||||
const sections: string[] = [];
|
||||
// Add metadata header
|
||||
const metaLines: string[] = [];
|
||||
for (const key in metadata) {
|
||||
const value = metadata[key];
|
||||
if (value) metaLines.push(`**${key.charAt(0).toUpperCase() + key.slice(1)}:** ${value}`);
|
||||
}
|
||||
if (metaLines.length > 0) sections.push(metaLines.join("\n"));
|
||||
// Convert spine files
|
||||
for (const idref of spineOrder) {
|
||||
const href = manifest.get(idref);
|
||||
if (!href) continue;
|
||||
const filePath = basePath ? `${basePath}/${href}` : href;
|
||||
const html = unzipText(entries, filePath);
|
||||
if (!html) continue;
|
||||
// Strip script/style, convert to markdown
|
||||
const cleaned = html.replace(/<script[\s\S]*?<\/script>/gi, "").replace(/<style[\s\S]*?<\/style>/gi, "");
|
||||
const md = turndown.turndown(normalizeTablesHtml(cleaned)).trim();
|
||||
if (md) sections.push(md);
|
||||
}
|
||||
return {
|
||||
markdown: sections.join("\n\n").trim(),
|
||||
title: metadata.title,
|
||||
};
|
||||
}
|
||||
|
||||
getText(node: MetaValue | undefined): string | undefined {
|
||||
if (!node) return undefined;
|
||||
if (typeof node === "string") return node;
|
||||
if (node["#text"]) return String(node["#text"]);
|
||||
if (Array.isArray(node)) return this.getText(node[0]);
|
||||
return undefined;
|
||||
}
|
||||
|
||||
getTextArray(node: MetaValue | undefined): (string | undefined)[] {
|
||||
if (!node) return [];
|
||||
const list = Array.isArray(node) ? node : [node];
|
||||
return list.map(n => this.getText(n)).filter(Boolean);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
// Minimal ambient types for `mammoth` (ships no types). Declares only what
|
||||
// DocxConverter uses. See ../NOTICE.
|
||||
declare module "mammoth" {
|
||||
interface MammothImage {
|
||||
contentType?: string;
|
||||
read(encoding: "base64"): Promise<string>;
|
||||
}
|
||||
interface ImgAttributes {
|
||||
src: string;
|
||||
alt?: string;
|
||||
}
|
||||
type ConvertImageHandler = (image: MammothImage) => Promise<ImgAttributes>;
|
||||
interface ConvertOptions {
|
||||
convertImage?: ConvertImageHandler;
|
||||
}
|
||||
interface ConvertResult {
|
||||
value: string;
|
||||
messages: unknown[];
|
||||
}
|
||||
export const images: { imgElement(fn: ConvertImageHandler): ConvertImageHandler };
|
||||
export function convertToHtml(input: { buffer: Buffer }, options?: ConvertOptions): Promise<ConvertResult>;
|
||||
const _default: { convertToHtml: typeof convertToHtml; images: typeof images };
|
||||
export default _default;
|
||||
}
|
||||
@@ -0,0 +1,103 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Multi-column layout detection and text box reordering.
|
||||
*
|
||||
* Many PDFs (legal documents, datasheets, academic papers) use two-column
|
||||
* layouts. Without column detection, text boxes are ordered by Y position
|
||||
* only, interleaving left and right column content.
|
||||
*
|
||||
* Algorithm:
|
||||
* 1. Collect left edges of all text boxes on the page
|
||||
* 2. Find the largest horizontal gap between consecutive left edges
|
||||
* 3. If gap > MIN_GAP_RATIO of the text width and both sides have
|
||||
* enough boxes → multi-column detected
|
||||
* 4. Assign each text box to a column based on its center X
|
||||
* 5. Return columns in reading order (left-to-right, top-to-bottom)
|
||||
*
|
||||
* This only detects the column structure. The caller is responsible for
|
||||
* processing each column's text boxes independently (table detection,
|
||||
* rendering, etc.).
|
||||
*/
|
||||
import type { TextBox } from "./types";
|
||||
|
||||
export interface ColumnLayout {
|
||||
/** Number of columns detected (1 = single column, 2+ = multi-column). */
|
||||
columnCount: number;
|
||||
/** Text boxes grouped by column, in reading order (left to right). */
|
||||
columns: TextBox[][];
|
||||
/** X positions of column boundaries (between columns). */
|
||||
boundaries: number[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimum gap as a fraction of the total text width to consider a column
|
||||
* boundary. A two-column layout typically has ~50% gap; we use a lower
|
||||
* threshold to catch asymmetric columns.
|
||||
*/
|
||||
const MIN_GAP_RATIO = 0.15;
|
||||
/** Minimum number of text boxes on each side of the gap. */
|
||||
const MIN_BOXES_PER_COLUMN = 4;
|
||||
/** Minimum gap in absolute points to avoid splitting on small whitespace. */
|
||||
const MIN_GAP_PTS = 40;
|
||||
|
||||
/**
|
||||
* Detect column layout and return text boxes grouped by column.
|
||||
*
|
||||
* For single-column pages, returns all boxes in one group.
|
||||
* For multi-column pages, returns boxes split by column in reading order.
|
||||
*/
|
||||
export function detectColumns(textBoxes: TextBox[]): ColumnLayout {
|
||||
if (textBoxes.length < MIN_BOXES_PER_COLUMN * 2) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
// Collect unique left edges (rounded to avoid float noise)
|
||||
const lefts = [...new Set(textBoxes.map(tb => Math.round(tb.bounds.left)))].sort((a, b) => a - b);
|
||||
if (lefts.length < 2) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
const textXMin = lefts[0];
|
||||
const textXMax = Math.max(...textBoxes.map(tb => Math.round(tb.bounds.right)));
|
||||
const textWidth = textXMax - textXMin;
|
||||
if (textWidth <= 0) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
// Find the largest gap between consecutive left-edge positions
|
||||
let maxGap = 0;
|
||||
let gapLeft = 0;
|
||||
let gapRight = 0;
|
||||
for (let i = 1; i < lefts.length; i++) {
|
||||
const gap = lefts[i] - lefts[i - 1];
|
||||
if (gap > maxGap) {
|
||||
maxGap = gap;
|
||||
gapLeft = lefts[i - 1];
|
||||
gapRight = lefts[i];
|
||||
}
|
||||
}
|
||||
const gapRatio = maxGap / textWidth;
|
||||
if (gapRatio < MIN_GAP_RATIO || maxGap < MIN_GAP_PTS) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
// Split point is the midpoint of the gap
|
||||
const splitX = (gapLeft + gapRight) / 2;
|
||||
// Assign boxes to columns based on center X
|
||||
const leftCol: TextBox[] = [];
|
||||
const rightCol: TextBox[] = [];
|
||||
for (const tb of textBoxes) {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
if (cx < splitX) {
|
||||
leftCol.push(tb);
|
||||
} else {
|
||||
rightCol.push(tb);
|
||||
}
|
||||
}
|
||||
// Validate both columns have enough content
|
||||
if (leftCol.length < MIN_BOXES_PER_COLUMN || rightCol.length < MIN_BOXES_PER_COLUMN) {
|
||||
return { columnCount: 1, columns: [textBoxes], boundaries: [] };
|
||||
}
|
||||
return {
|
||||
columnCount: 2,
|
||||
columns: [leftCol, rightCol],
|
||||
boundaries: [splitX],
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,557 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* PDF content extraction using mupdf.
|
||||
*
|
||||
* Extracts text boxes (with position, font size, bold) and vector line
|
||||
* segments (table borders) from each page. Uses mupdf's native WASM
|
||||
* engine for fast parsing, and reads raw content streams for vector graphics.
|
||||
*
|
||||
* Coordinate system: PDF native (origin = bottom-left, Y increases upward).
|
||||
*/
|
||||
import * as mupdf from "mupdf";
|
||||
import type { ImageRegion, PageContent, Segment, TextBox } from "./types";
|
||||
|
||||
/** mupdf structured-text JSON bounding box (top-left origin). */
|
||||
interface StextBBox {
|
||||
x: number;
|
||||
y: number;
|
||||
w: number;
|
||||
h: number;
|
||||
}
|
||||
|
||||
/** Font metadata attached to a structured-text line. */
|
||||
interface StextFont {
|
||||
size?: number;
|
||||
weight?: string;
|
||||
name?: string;
|
||||
}
|
||||
|
||||
/** A line within a text block in mupdf structured-text JSON. */
|
||||
interface StextLine {
|
||||
text?: string;
|
||||
font?: StextFont;
|
||||
bbox: StextBBox;
|
||||
}
|
||||
|
||||
/** A block (text or image) in mupdf structured-text JSON. */
|
||||
interface StextBlock {
|
||||
type: string;
|
||||
bbox: StextBBox;
|
||||
lines: StextLine[];
|
||||
}
|
||||
|
||||
/** Parsed mupdf structured-text JSON for a page. */
|
||||
interface StructuredTextJSON {
|
||||
blocks: StextBlock[];
|
||||
}
|
||||
|
||||
/** A raw text fragment before merging into word/phrase boxes. */
|
||||
interface RawTextItem {
|
||||
text: string;
|
||||
x: number;
|
||||
y: number;
|
||||
width: number;
|
||||
height: number;
|
||||
fontSize: number;
|
||||
isBold: boolean;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Text extraction
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Y tolerance for merging text fragments on the same visual line. */
|
||||
const SAME_LINE_Y_TOLERANCE = 2;
|
||||
/** Max horizontal gap (pts) to merge adjacent fragments into one text box. */
|
||||
const MAX_MERGE_GAP = 14;
|
||||
|
||||
/**
|
||||
* Merge horizontally adjacent raw text items on the same visual line into
|
||||
* word/phrase-level text boxes.
|
||||
*/
|
||||
function mergeIntoWords(raws: RawTextItem[]): RawTextItem[] {
|
||||
if (raws.length === 0) return [];
|
||||
// Sort by Y descending (top-first in bottom-left coords), then X ascending
|
||||
const sorted = [...raws].sort((a, b) => {
|
||||
const dy = b.y - a.y;
|
||||
return Math.abs(dy) > SAME_LINE_Y_TOLERANCE ? dy : a.x - b.x;
|
||||
});
|
||||
const merged: RawTextItem[] = [];
|
||||
let cur = { ...sorted[0] };
|
||||
for (let i = 1; i < sorted.length; i++) {
|
||||
const next = sorted[i];
|
||||
const sameY = Math.abs(next.y - cur.y) <= SAME_LINE_Y_TOLERANCE;
|
||||
const close = next.x <= cur.x + cur.width + MAX_MERGE_GAP;
|
||||
if (sameY && close) {
|
||||
const gap = next.x - (cur.x + cur.width);
|
||||
const sep = gap > 1 ? " " : "";
|
||||
cur.text += sep + next.text;
|
||||
cur.width = next.x + next.width - cur.x;
|
||||
cur.height = Math.max(cur.height, next.height);
|
||||
cur.fontSize = Math.max(cur.fontSize, next.fontSize);
|
||||
cur.isBold = cur.isBold || next.isBold;
|
||||
} else {
|
||||
merged.push(cur);
|
||||
cur = { ...next };
|
||||
}
|
||||
}
|
||||
merged.push(cur);
|
||||
return merged;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract text boxes from a mupdf page using structured text output.
|
||||
*
|
||||
* mupdf's structured text JSON uses top-left origin; we convert to
|
||||
* bottom-left (standard PDF coordinates) using the page height.
|
||||
*/
|
||||
function extractTextBoxes(
|
||||
page: mupdf.Page,
|
||||
pageNumber: number,
|
||||
pageHeight: number,
|
||||
stext?: StructuredTextJSON,
|
||||
): TextBox[] {
|
||||
if (!stext) {
|
||||
stext = JSON.parse(page.toStructuredText("preserve-whitespace").asJSON()) as StructuredTextJSON;
|
||||
}
|
||||
const raws: RawTextItem[] = [];
|
||||
for (const block of stext.blocks) {
|
||||
if (block.type !== "text") continue;
|
||||
for (const line of block.lines) {
|
||||
const text = line.text?.trim();
|
||||
if (!text) continue;
|
||||
const fontSize = line.font?.size ?? 0;
|
||||
const weight = line.font?.weight ?? "normal";
|
||||
const fontName = line.font?.name ?? "";
|
||||
const isBold = weight === "bold" || /bold/i.test(fontName) || /Black|Heavy/i.test(fontName);
|
||||
// mupdf bbox: {x, y, w, h} in top-left coords
|
||||
// Convert to bottom-left: pdfY = pageHeight - (bbox.y + bbox.h)
|
||||
const bboxY = line.bbox.y;
|
||||
const bboxH = line.bbox.h;
|
||||
const pdfY = pageHeight - (bboxY + bboxH);
|
||||
raws.push({
|
||||
text,
|
||||
x: line.bbox.x,
|
||||
y: pdfY,
|
||||
width: line.bbox.w,
|
||||
height: bboxH,
|
||||
fontSize,
|
||||
isBold,
|
||||
});
|
||||
}
|
||||
}
|
||||
const words = mergeIntoWords(raws);
|
||||
return words
|
||||
.map((w, i) => ({
|
||||
id: `p${pageNumber}-t${i}`,
|
||||
text: w.text.trim(),
|
||||
pageNumber,
|
||||
fontSize: w.fontSize,
|
||||
isBold: w.isBold,
|
||||
bounds: {
|
||||
left: w.x,
|
||||
right: w.x + w.width,
|
||||
bottom: w.y,
|
||||
top: w.y + w.height,
|
||||
},
|
||||
}))
|
||||
.filter(b => b.text.length > 0);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Vector segment extraction from raw content stream
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Minimum aspect ratio for a filled rect to be considered a line. */
|
||||
const LINE_ASPECT_THRESHOLD = 6;
|
||||
/** Minimum length (pts) for a segment to count. */
|
||||
const MIN_LENGTH = 2;
|
||||
/** Maximum thickness (pts) for a border line (filters out filled areas). */
|
||||
const MAX_THICKNESS = 3;
|
||||
|
||||
/**
|
||||
* Convert a thin filled rectangle to a horizontal or vertical segment.
|
||||
* Returns null if the rect doesn't look like a border line.
|
||||
*/
|
||||
function thinRectToSegment(id: string, x: number, y: number, w: number, h: number): Segment | null {
|
||||
const aw = Math.abs(w);
|
||||
const ah = Math.abs(h);
|
||||
if (aw > ah * LINE_ASPECT_THRESHOLD && aw >= MIN_LENGTH && ah <= MAX_THICKNESS) {
|
||||
// Horizontal line
|
||||
const cy = y + ah / 2;
|
||||
return { id, x1: x, y1: cy, x2: x + aw, y2: cy };
|
||||
}
|
||||
if (ah > aw * LINE_ASPECT_THRESHOLD && ah >= MIN_LENGTH && aw <= MAX_THICKNESS) {
|
||||
// Vertical line
|
||||
const cx = x + aw / 2;
|
||||
return { id, x1: cx, y1: y, x2: cx, y2: y + ah };
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Emit 4 edge segments from a stroked rectangle.
|
||||
*/
|
||||
function pushStrokedRectEdges(segments: Segment[], id: string, x: number, y: number, w: number, h: number): void {
|
||||
const aw = Math.abs(w);
|
||||
const ah = Math.abs(h);
|
||||
const base = id;
|
||||
if (aw >= MIN_LENGTH) {
|
||||
segments.push({ id: `${base}-b`, x1: x, y1: y, x2: x + aw, y2: y });
|
||||
segments.push({
|
||||
id: `${base}-t`,
|
||||
x1: x,
|
||||
y1: y + ah,
|
||||
x2: x + aw,
|
||||
y2: y + ah,
|
||||
});
|
||||
}
|
||||
if (ah >= MIN_LENGTH) {
|
||||
segments.push({ id: `${base}-l`, x1: x, y1: y, x2: x, y2: y + ah });
|
||||
segments.push({
|
||||
id: `${base}-r`,
|
||||
x1: x + aw,
|
||||
y1: y,
|
||||
x2: x + aw,
|
||||
y2: y + ah,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const CTM_IDENTITY = [1, 0, 0, 1, 0, 0];
|
||||
|
||||
/** Concatenate two affine matrices: result = parent × child. */
|
||||
function ctmConcat(p: number[], c: number[]): number[] {
|
||||
return [
|
||||
p[0] * c[0] + p[2] * c[1],
|
||||
p[1] * c[0] + p[3] * c[1],
|
||||
p[0] * c[2] + p[2] * c[3],
|
||||
p[1] * c[2] + p[3] * c[3],
|
||||
p[0] * c[4] + p[2] * c[5] + p[4],
|
||||
p[1] * c[4] + p[3] * c[5] + p[5],
|
||||
];
|
||||
}
|
||||
|
||||
function ctmApply(m: number[], x: number, y: number): [number, number] {
|
||||
return [m[0] * x + m[2] * y + m[4], m[1] * x + m[3] * y + m[5]];
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Content stream parsing
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Parse a PDF content stream and extract line segments from thin filled
|
||||
* rectangles (re+f), stroked rectangles (re+S), and explicit lines (m/l+S).
|
||||
* Tracks the CTM via q/Q/cm operators so coordinates are in page space.
|
||||
*/
|
||||
function extractSegmentsFromContentStream(raw: string, pageNumber: number): Segment[] {
|
||||
const segments: Segment[] = [];
|
||||
const tokens = tokenizeContentStream(raw);
|
||||
let idx = 0;
|
||||
let strokeWidth = 1.0;
|
||||
// Graphics state stack (q/Q): saves CTM + strokeWidth
|
||||
let ctm = [...CTM_IDENTITY];
|
||||
const stateStack: Array<{ ctm: number[]; strokeWidth: number }> = [];
|
||||
// State for path building (in user coordinates, pre-CTM)
|
||||
let curX = 0;
|
||||
let curY = 0;
|
||||
let pathStartX = 0;
|
||||
let pathStartY = 0;
|
||||
const pendingRects: Array<{ x: number; y: number; w: number; h: number }> = [];
|
||||
const pendingLines: Array<{ x1: number; y1: number; x2: number; y2: number }> = [];
|
||||
function flushPath(mode: "fill" | "stroke"): void {
|
||||
const sid = () => `p${pageNumber}-s${segments.length}`;
|
||||
if (mode === "fill") {
|
||||
for (const r of pendingRects) {
|
||||
// Transform the rect corners through CTM, then check if it's a thin line
|
||||
const [x0, y0] = ctmApply(ctm, r.x, r.y);
|
||||
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
|
||||
const seg = thinRectToSegment(
|
||||
sid(),
|
||||
Math.min(x0, x1),
|
||||
Math.min(y0, y1),
|
||||
Math.abs(x1 - x0),
|
||||
Math.abs(y1 - y0),
|
||||
);
|
||||
if (seg) segments.push(seg);
|
||||
}
|
||||
} else if (mode === "stroke" && strokeWidth <= MAX_THICKNESS) {
|
||||
for (const r of pendingRects) {
|
||||
const [x0, y0] = ctmApply(ctm, r.x, r.y);
|
||||
const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h);
|
||||
pushStrokedRectEdges(
|
||||
segments,
|
||||
sid(),
|
||||
Math.min(x0, x1),
|
||||
Math.min(y0, y1),
|
||||
Math.abs(x1 - x0),
|
||||
Math.abs(y1 - y0),
|
||||
);
|
||||
}
|
||||
for (const l of pendingLines) {
|
||||
const [lx1, ly1] = ctmApply(ctm, l.x1, l.y1);
|
||||
const [lx2, ly2] = ctmApply(ctm, l.x2, l.y2);
|
||||
const dx = Math.abs(lx2 - lx1);
|
||||
const dy = Math.abs(ly2 - ly1);
|
||||
// Only keep H/V lines
|
||||
if ((dx >= MIN_LENGTH && dy < 1) || (dy >= MIN_LENGTH && dx < 1)) {
|
||||
segments.push({ id: sid(), x1: lx1, y1: ly1, x2: lx2, y2: ly2 });
|
||||
}
|
||||
}
|
||||
}
|
||||
pendingRects.length = 0;
|
||||
pendingLines.length = 0;
|
||||
}
|
||||
while (idx < tokens.length) {
|
||||
const t = tokens[idx];
|
||||
if (t === "q") {
|
||||
stateStack.push({ ctm: [...ctm], strokeWidth });
|
||||
} else if (t === "Q") {
|
||||
const saved = stateStack.pop();
|
||||
if (saved) {
|
||||
ctm = saved.ctm;
|
||||
strokeWidth = saved.strokeWidth;
|
||||
}
|
||||
} else if (t === "cm" && idx >= 6) {
|
||||
const a = Number(tokens[idx - 6]);
|
||||
const b = Number(tokens[idx - 5]);
|
||||
const c = Number(tokens[idx - 4]);
|
||||
const d = Number(tokens[idx - 3]);
|
||||
const e = Number(tokens[idx - 2]);
|
||||
const f = Number(tokens[idx - 1]);
|
||||
ctm = ctmConcat(ctm, [a, b, c, d, e, f]);
|
||||
} else if (t === "w" && idx >= 1) {
|
||||
strokeWidth = Number(tokens[idx - 1]) || strokeWidth;
|
||||
} else if (t === "re" && idx >= 4) {
|
||||
const x = Number(tokens[idx - 4]);
|
||||
const y = Number(tokens[idx - 3]);
|
||||
const w = Number(tokens[idx - 2]);
|
||||
const h = Number(tokens[idx - 1]);
|
||||
if (Number.isFinite(x + y + w + h)) {
|
||||
pendingRects.push({ x, y, w, h });
|
||||
}
|
||||
} else if (t === "m" && idx >= 2) {
|
||||
curX = Number(tokens[idx - 2]);
|
||||
curY = Number(tokens[idx - 1]);
|
||||
pathStartX = curX;
|
||||
pathStartY = curY;
|
||||
} else if (t === "l" && idx >= 2) {
|
||||
const x2 = Number(tokens[idx - 2]);
|
||||
const y2 = Number(tokens[idx - 1]);
|
||||
pendingLines.push({ x1: curX, y1: curY, x2, y2 });
|
||||
curX = x2;
|
||||
curY = y2;
|
||||
} else if (t === "h") {
|
||||
// closePath: line back to start
|
||||
if (curX !== pathStartX || curY !== pathStartY) {
|
||||
pendingLines.push({
|
||||
x1: curX,
|
||||
y1: curY,
|
||||
x2: pathStartX,
|
||||
y2: pathStartY,
|
||||
});
|
||||
}
|
||||
curX = pathStartX;
|
||||
curY = pathStartY;
|
||||
} else if (t === "f" || t === "F" || t === "f*") {
|
||||
flushPath("fill");
|
||||
} else if (t === "S" || t === "s") {
|
||||
if (t === "s") {
|
||||
// closeStroke: implicit closePath
|
||||
if (curX !== pathStartX || curY !== pathStartY) {
|
||||
pendingLines.push({
|
||||
x1: curX,
|
||||
y1: curY,
|
||||
x2: pathStartX,
|
||||
y2: pathStartY,
|
||||
});
|
||||
}
|
||||
}
|
||||
flushPath("stroke");
|
||||
} else if (t === "B" || t === "B*" || t === "b" || t === "b*") {
|
||||
// fill + stroke combined
|
||||
flushPath("fill");
|
||||
flushPath("stroke");
|
||||
} else if (t === "n") {
|
||||
// end path without painting — discard
|
||||
pendingRects.length = 0;
|
||||
pendingLines.length = 0;
|
||||
}
|
||||
idx++;
|
||||
}
|
||||
return segments;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fast tokenizer for PDF content streams.
|
||||
* Splits on whitespace, skipping comments and string literals.
|
||||
*/
|
||||
function tokenizeContentStream(raw: string): string[] {
|
||||
const tokens: string[] = [];
|
||||
const len = raw.length;
|
||||
let i = 0;
|
||||
while (i < len) {
|
||||
const ch = raw.charCodeAt(i);
|
||||
// Skip whitespace
|
||||
if (ch <= 32) {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
// Skip comments
|
||||
if (ch === 37 /* % */) {
|
||||
while (i < len && raw.charCodeAt(i) !== 10) i++;
|
||||
continue;
|
||||
}
|
||||
// Skip string literals (...)
|
||||
if (ch === 40 /* ( */) {
|
||||
let depth = 1;
|
||||
i++;
|
||||
while (i < len && depth > 0) {
|
||||
const c = raw.charCodeAt(i);
|
||||
if (c === 92 /* \ */) {
|
||||
i++;
|
||||
} else if (c === 40) {
|
||||
depth++;
|
||||
} else if (c === 41) {
|
||||
depth--;
|
||||
}
|
||||
i++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
// Skip hex strings <...>
|
||||
if (ch === 60 /* < */ && i + 1 < len && raw.charCodeAt(i + 1) !== 60) {
|
||||
i++;
|
||||
while (i < len && raw.charCodeAt(i) !== 62) i++;
|
||||
i++; // skip >
|
||||
continue;
|
||||
}
|
||||
// Skip dict delimiters << >>
|
||||
if (ch === 60 && i + 1 < len && raw.charCodeAt(i + 1) === 60) {
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
if (ch === 62 && i + 1 < len && raw.charCodeAt(i + 1) === 62) {
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
// Regular token: read until whitespace or delimiter
|
||||
const start = i;
|
||||
while (i < len) {
|
||||
const c = raw.charCodeAt(i);
|
||||
if (c <= 32 || c === 40 || c === 41 || c === 60 || c === 62 || c === 37) break;
|
||||
i++;
|
||||
}
|
||||
if (i > start) {
|
||||
tokens.push(raw.substring(start, i));
|
||||
}
|
||||
}
|
||||
return tokens;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Image region detection
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Minimum area (pts²) for an image to be considered a diagram, not an icon. */
|
||||
const MIN_IMAGE_AREA = 5000;
|
||||
|
||||
function extractImageRegions(stext: StructuredTextJSON, pageNumber: number, pageHeight: number): ImageRegion[] {
|
||||
const regions: ImageRegion[] = [];
|
||||
for (const block of stext.blocks) {
|
||||
if (block.type !== "image") continue;
|
||||
const { x, y, w, h } = block.bbox;
|
||||
if (w * h < MIN_IMAGE_AREA) continue; // skip tiny icons
|
||||
// Convert Y from mupdf (top-left) to PDF (bottom-left) for ordering
|
||||
const pdfTopY = pageHeight - y;
|
||||
regions.push({
|
||||
id: `p${pageNumber}-img${regions.length}`,
|
||||
pageNumber,
|
||||
bbox: { x, y, w, h },
|
||||
topY: pdfTopY,
|
||||
});
|
||||
}
|
||||
return regions;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Render an image region from a PDF page as a PNG buffer.
|
||||
* Uses mupdf's DrawDevice to render just the cropped area at 2x resolution.
|
||||
*/
|
||||
export function renderImageRegion(input: Uint8Array, region: ImageRegion): Uint8Array {
|
||||
const doc = mupdf.Document.openDocument(input, "application/pdf");
|
||||
const page = doc.loadPage(region.pageNumber - 1);
|
||||
const pad = 10;
|
||||
const bx = region.bbox.x - pad;
|
||||
const by = region.bbox.y - pad;
|
||||
const bw = region.bbox.w + 2 * pad;
|
||||
const bh = region.bbox.h + 2 * pad;
|
||||
const scale = 2;
|
||||
const pw = Math.round(bw * scale);
|
||||
const ph = Math.round(bh * scale);
|
||||
const pix = new mupdf.Pixmap(mupdf.ColorSpace.DeviceRGB, [0, 0, pw, ph], false);
|
||||
pix.clear(255);
|
||||
const matrix: mupdf.Matrix = [scale, 0, 0, scale, -bx * scale, -by * scale];
|
||||
const dl = page.toDisplayList();
|
||||
const dev = new mupdf.DrawDevice(matrix, pix);
|
||||
dl.run(dev, mupdf.Matrix.identity);
|
||||
dev.close();
|
||||
return pix.asPNG();
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract text boxes and vector segments from all pages of a PDF buffer.
|
||||
*/
|
||||
export async function extractPages(input: Uint8Array): Promise<PageContent[]> {
|
||||
const doc = mupdf.Document.openDocument(input, "application/pdf");
|
||||
const pages: PageContent[] = [];
|
||||
for (let i = 0; i < doc.countPages(); i++) {
|
||||
const pageNumber = i + 1;
|
||||
const page = doc.loadPage(i);
|
||||
const bounds = page.getBounds();
|
||||
const pageHeight = bounds[3] - bounds[1];
|
||||
// Single structured text pass with both flags
|
||||
const stext = JSON.parse(
|
||||
page.toStructuredText("preserve-whitespace,preserve-images").asJSON(),
|
||||
) as StructuredTextJSON;
|
||||
// Extract text boxes and image regions from the same parse
|
||||
const textBoxes = extractTextBoxes(page, pageNumber, pageHeight, stext);
|
||||
const images = extractImageRegions(stext, pageNumber, pageHeight);
|
||||
// Extract vector segments from raw content stream
|
||||
let segments: Segment[] = [];
|
||||
try {
|
||||
const pageObj = (page as mupdf.PDFPage).getObject();
|
||||
const contents = pageObj.get("Contents");
|
||||
if (contents) {
|
||||
let rawBytes: Uint8Array;
|
||||
if (contents.isArray()) {
|
||||
// Multiple content streams — concatenate
|
||||
const parts: Uint8Array[] = [];
|
||||
const len = contents.length ?? 0;
|
||||
for (let j = 0; j < len; j++) {
|
||||
const stream = contents.get(j);
|
||||
if (stream?.readStream) {
|
||||
parts.push(stream.readStream().asUint8Array());
|
||||
}
|
||||
}
|
||||
const totalLen = parts.reduce((s, p) => s + p.length, 0);
|
||||
rawBytes = new Uint8Array(totalLen);
|
||||
let offset = 0;
|
||||
for (const part of parts) {
|
||||
rawBytes.set(part, offset);
|
||||
offset += part.length;
|
||||
}
|
||||
} else {
|
||||
rawBytes = contents.readStream().asUint8Array();
|
||||
}
|
||||
const raw = new TextDecoder().decode(rawBytes);
|
||||
segments = extractSegmentsFromContentStream(raw, pageNumber);
|
||||
}
|
||||
} catch {
|
||||
// Content stream extraction failed — proceed with text only
|
||||
}
|
||||
pages.push({ pageNumber, textBoxes, segments, images });
|
||||
}
|
||||
return pages;
|
||||
}
|
||||
@@ -0,0 +1,780 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Table grid detection from vector segments and text boxes.
|
||||
*
|
||||
* Ported from @oharato/pdf2md-ts with TypeScript types and without
|
||||
* CJK-specific borderless table heuristics. The core algorithm:
|
||||
*
|
||||
* 1. Classify segments as horizontal or vertical lines
|
||||
* 2. Group horizontal Y-lines into table groups (split by vertical gaps)
|
||||
* 3. For each group:
|
||||
* a. Full grid (H+V lines): build cells from grid intersections,
|
||||
* place text via raycasting
|
||||
* b. H-line only (no V lines): infer columns from text X positions
|
||||
* 4. Prune empty rows/cols
|
||||
*
|
||||
* Coordinate system: PDF native (bottom-left origin, Y increases upward).
|
||||
*/
|
||||
import type { Segment, TableCell, TableGrid, TextBox } from "./types";
|
||||
|
||||
export interface GridResult {
|
||||
grids: TableGrid[];
|
||||
consumedIds: string[];
|
||||
}
|
||||
|
||||
type RayDirection = "up" | "down" | "left" | "right";
|
||||
|
||||
interface Ray {
|
||||
direction: RayDirection;
|
||||
segmentId: string | null;
|
||||
distance: number;
|
||||
}
|
||||
|
||||
interface Interval {
|
||||
min: number;
|
||||
max: number;
|
||||
}
|
||||
|
||||
function castRaysForTextBox(textBox: TextBox, segments: Segment[]): Ray[] {
|
||||
const cx = (textBox.bounds.left + textBox.bounds.right) / 2;
|
||||
const cy = (textBox.bounds.top + textBox.bounds.bottom) / 2;
|
||||
let up: Ray = { direction: "up", segmentId: null, distance: Infinity };
|
||||
let down: Ray = { direction: "down", segmentId: null, distance: Infinity };
|
||||
let left: Ray = { direction: "left", segmentId: null, distance: Infinity };
|
||||
let right: Ray = {
|
||||
direction: "right",
|
||||
segmentId: null,
|
||||
distance: Infinity,
|
||||
};
|
||||
for (const seg of segments) {
|
||||
const isH = Math.abs(seg.y1 - seg.y2) < 0.5;
|
||||
const isV = Math.abs(seg.x1 - seg.x2) < 0.5;
|
||||
if (isH) {
|
||||
const minX = Math.min(seg.x1, seg.x2);
|
||||
const maxX = Math.max(seg.x1, seg.x2);
|
||||
if (cx >= minX && cx <= maxX) {
|
||||
const d = seg.y1 - cy;
|
||||
if (d >= 0 && d < up.distance) up = { direction: "up", segmentId: seg.id, distance: d };
|
||||
const dd = cy - seg.y1;
|
||||
if (dd >= 0 && dd < down.distance) down = { direction: "down", segmentId: seg.id, distance: dd };
|
||||
}
|
||||
}
|
||||
if (isV) {
|
||||
const minY = Math.min(seg.y1, seg.y2);
|
||||
const maxY = Math.max(seg.y1, seg.y2);
|
||||
if (cy >= minY && cy <= maxY) {
|
||||
const d = cx - seg.x1;
|
||||
if (d >= 0 && d < left.distance) left = { direction: "left", segmentId: seg.id, distance: d };
|
||||
const rd = seg.x1 - cx;
|
||||
if (rd >= 0 && rd < right.distance) right = { direction: "right", segmentId: seg.id, distance: rd };
|
||||
}
|
||||
}
|
||||
}
|
||||
return [up, down, left, right];
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Utility
|
||||
// ---------------------------------------------------------------------------
|
||||
const AXIS_EPSILON = 0.8;
|
||||
const PAGE_MARGIN = 20;
|
||||
|
||||
function uniqueSorted(values: number[]): number[] {
|
||||
const sorted = [...values].sort((a, b) => a - b);
|
||||
const result: number[] = [];
|
||||
for (const v of sorted) {
|
||||
if (result.length === 0 || Math.abs(result[result.length - 1] - v) > 1) result.push(v);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Y-line group splitting
|
||||
// ---------------------------------------------------------------------------
|
||||
function chainCoversRange(intervals: Interval[], lowerY: number, upperY: number, eps: number): boolean {
|
||||
const sorted = [...intervals].sort((a, b) => a.min - b.min);
|
||||
let covered = lowerY;
|
||||
for (const iv of sorted) {
|
||||
if (iv.min > covered + eps) break;
|
||||
if (iv.max > covered) covered = iv.max;
|
||||
if (covered >= upperY - eps) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function countBridgingVLineCols(upperY: number, lowerY: number, verticals: Segment[]): number {
|
||||
const eps = 1.5;
|
||||
const byX = new Map<number, Interval[]>();
|
||||
for (const seg of verticals) {
|
||||
const rx = Math.round(seg.x1);
|
||||
if (!byX.has(rx)) byX.set(rx, []);
|
||||
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
|
||||
}
|
||||
let count = 0;
|
||||
for (const intervals of byX.values()) {
|
||||
if (chainCoversRange(intervals, lowerY, upperY, eps)) count++;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
function bridgingXSet(upperY: number, lowerY: number, verticals: Segment[]): Set<number> {
|
||||
const eps = 1.5;
|
||||
const xs = new Set<number>();
|
||||
const byX = new Map<number, Interval[]>();
|
||||
for (const seg of verticals) {
|
||||
const rx = Math.round(seg.x1);
|
||||
if (!byX.has(rx)) byX.set(rx, []);
|
||||
byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) });
|
||||
}
|
||||
for (const [rx, intervals] of byX) {
|
||||
if (chainCoversRange(intervals, lowerY, upperY, eps)) xs.add(rx);
|
||||
}
|
||||
return xs;
|
||||
}
|
||||
|
||||
const MIN_RICH_BRIDGING_COLS = 3;
|
||||
|
||||
function splitYLinesIntoGroups(yLines: number[], verticals: Segment[]): number[][] {
|
||||
if (yLines.length === 0) return [];
|
||||
const eps = 1.5;
|
||||
const allX = verticals.map(s => Math.round(s.x1));
|
||||
const globalXMin = allX.length > 0 ? Math.min(...allX) : 0;
|
||||
const globalXMax = allX.length > 0 ? Math.max(...allX) : 0;
|
||||
const groups: number[][] = [];
|
||||
let currentGroup = [yLines[0]];
|
||||
let prevBridgingCols = -1;
|
||||
for (let i = 1; i < yLines.length; i++) {
|
||||
const upperY = yLines[i - 1];
|
||||
const lowerY = yLines[i];
|
||||
const cols = countBridgingVLineCols(upperY, lowerY, verticals);
|
||||
if (cols === 0) {
|
||||
groups.push(currentGroup);
|
||||
currentGroup = [yLines[i]];
|
||||
prevBridgingCols = -1;
|
||||
continue;
|
||||
}
|
||||
if (prevBridgingCols >= MIN_RICH_BRIDGING_COLS && cols < MIN_RICH_BRIDGING_COLS) {
|
||||
const bxs = bridgingXSet(upperY, lowerY, verticals);
|
||||
const isOuterFrameOnly = [...bxs].every(
|
||||
x => Math.abs(x - globalXMin) <= eps || Math.abs(x - globalXMax) <= eps,
|
||||
);
|
||||
if (!isOuterFrameOnly) {
|
||||
groups.push(currentGroup);
|
||||
currentGroup = [yLines[i - 1], yLines[i]];
|
||||
prevBridgingCols = cols;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
currentGroup.push(yLines[i]);
|
||||
prevBridgingCols = cols;
|
||||
}
|
||||
groups.push(currentGroup);
|
||||
return groups;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Sub-row Y-cluster expansion
|
||||
// ---------------------------------------------------------------------------
|
||||
const Y_CLUSTER_GAP = 10;
|
||||
const MIN_COLS_IN_TOP_CLUSTER = 2;
|
||||
|
||||
function assignToYCluster(y: number, clusters: number[]): number {
|
||||
let closest = 0;
|
||||
let closestDist = Math.abs(y - clusters[0]);
|
||||
for (let k = 1; k < clusters.length; k++) {
|
||||
const d = Math.abs(y - clusters[k]);
|
||||
if (d < closestDist) {
|
||||
closestDist = d;
|
||||
closest = k;
|
||||
}
|
||||
}
|
||||
return closest;
|
||||
}
|
||||
|
||||
function expandSubRowsByYClusters(
|
||||
originalRows: number,
|
||||
cols: number,
|
||||
cells: TableCell[],
|
||||
cellBoxes: Map<TableCell, TextBox[]>,
|
||||
): number {
|
||||
let addedRows = 0;
|
||||
for (let origRow = 0; origRow < originalRows; origRow++) {
|
||||
const currentRow = origRow + addedRows;
|
||||
const rowCellInfos: Array<{ cell: TableCell; col: number; boxes: TextBox[] }> = [];
|
||||
for (let col = 0; col < cols; col++) {
|
||||
const cell = cells.find(c => c.row === currentRow && c.col === col);
|
||||
if (!cell) continue;
|
||||
const boxes = cellBoxes.get(cell);
|
||||
if (boxes && boxes.length > 0) rowCellInfos.push({ cell, col, boxes });
|
||||
}
|
||||
if (rowCellInfos.length === 0) continue;
|
||||
const allMidYs = rowCellInfos.flatMap(({ boxes }) => boxes.map(b => (b.bounds.top + b.bounds.bottom) / 2));
|
||||
const sortedY = [...new Set(allMidYs.map(y => Math.round(y * 10) / 10))].sort((a, b) => b - a);
|
||||
const clusters = [sortedY[0]];
|
||||
for (let i = 1; i < sortedY.length; i++) {
|
||||
if (clusters[clusters.length - 1] - sortedY[i] > Y_CLUSTER_GAP) {
|
||||
clusters.push(sortedY[i]);
|
||||
}
|
||||
}
|
||||
if (clusters.length < 2) continue;
|
||||
const colsInTopCluster = new Set<number>();
|
||||
const totalNonEmptyCols = new Set<number>();
|
||||
for (const { col, boxes } of rowCellInfos) {
|
||||
totalNonEmptyCols.add(col);
|
||||
if (boxes.some(b => assignToYCluster((b.bounds.top + b.bounds.bottom) / 2, clusters) === 0)) {
|
||||
colsInTopCluster.add(col);
|
||||
}
|
||||
}
|
||||
if (colsInTopCluster.size < MIN_COLS_IN_TOP_CLUSTER) continue;
|
||||
if (colsInTopCluster.size >= totalNonEmptyCols.size) continue;
|
||||
const sparseColsHaveMultipleBoxes = rowCellInfos.some(
|
||||
({ col, boxes }) => !colsInTopCluster.has(col) && boxes.length > 1,
|
||||
);
|
||||
if (!sparseColsHaveMultipleBoxes) continue;
|
||||
const numSubRows = clusters.length;
|
||||
const numNewRows = numSubRows - 1;
|
||||
for (const cell of cells) {
|
||||
if (cell.row > currentRow) cell.row += numNewRows;
|
||||
}
|
||||
for (let subRow = 1; subRow < numSubRows; subRow++) {
|
||||
for (let col = 0; col < cols; col++) {
|
||||
cells.push({
|
||||
row: currentRow + subRow,
|
||||
col,
|
||||
text: "",
|
||||
rowSpan: 1,
|
||||
colSpan: 1,
|
||||
});
|
||||
}
|
||||
}
|
||||
for (const { cell: origCell, col, boxes } of rowCellInfos) {
|
||||
const subRowBoxGroups: TextBox[][] = Array.from({ length: numSubRows }, () => []);
|
||||
for (const box of boxes) {
|
||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
||||
subRowBoxGroups[assignToYCluster(cy, clusters)].push(box);
|
||||
}
|
||||
cellBoxes.set(origCell, subRowBoxGroups[0]);
|
||||
if (subRowBoxGroups[0].length === 0) cellBoxes.delete(origCell);
|
||||
for (let subRow = 1; subRow < numSubRows; subRow++) {
|
||||
if (subRowBoxGroups[subRow].length > 0) {
|
||||
const newCell = cells.find(c => c.row === currentRow + subRow && c.col === col);
|
||||
if (newCell) cellBoxes.set(newCell, subRowBoxGroups[subRow]);
|
||||
}
|
||||
}
|
||||
}
|
||||
addedRows += numNewRows;
|
||||
}
|
||||
return originalRows + addedRows;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Cross-column text box splitting
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Find which column a horizontal position falls into.
|
||||
* Returns -1 if outside the grid.
|
||||
*/
|
||||
function findCol(x: number, xLines: number[]): number {
|
||||
for (let i = 0; i < xLines.length - 1; i++) {
|
||||
if (x >= xLines[i] && x <= xLines[i + 1]) return i;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/**
|
||||
* When a text box spans across one or more vertical column boundaries,
|
||||
* split it into multiple virtual text boxes — one per column — with the
|
||||
* text divided proportionally by width.
|
||||
*
|
||||
* We split at word boundaries closest to the proportional split point
|
||||
* so we don't chop words in half.
|
||||
*/
|
||||
function splitCrossColumnBoxes(textBoxes: TextBox[], xLines: number[]): TextBox[] {
|
||||
const result: TextBox[] = [];
|
||||
const MARGIN = 5; // allow small overlap before considering it cross-column
|
||||
for (const tb of textBoxes) {
|
||||
const leftCol = findCol(tb.bounds.left + MARGIN, xLines);
|
||||
const rightCol = findCol(tb.bounds.right - MARGIN, xLines);
|
||||
// Not spanning columns, or outside grid — keep as-is
|
||||
if (leftCol < 0 || rightCol < 0 || leftCol === rightCol) {
|
||||
result.push(tb);
|
||||
continue;
|
||||
}
|
||||
// Text box spans from leftCol to rightCol — split it
|
||||
const totalWidth = tb.bounds.right - tb.bounds.left;
|
||||
if (totalWidth <= 0) {
|
||||
result.push(tb);
|
||||
continue;
|
||||
}
|
||||
const words = tb.text.split(/\s+/);
|
||||
if (words.length <= 1) {
|
||||
// Single word spanning columns — just assign to whichever col has more overlap
|
||||
result.push(tb);
|
||||
continue;
|
||||
}
|
||||
// For each column boundary crossing, find the best word-boundary split
|
||||
let remainingWords = [...words];
|
||||
let currentLeft = tb.bounds.left;
|
||||
for (let col = leftCol; col <= rightCol && remainingWords.length > 0; col++) {
|
||||
const colRight = col < xLines.length - 1 ? xLines[col + 1] : tb.bounds.right;
|
||||
const segmentRight = Math.min(colRight, tb.bounds.right);
|
||||
if (col === rightCol) {
|
||||
// Last column — take all remaining words
|
||||
result.push({
|
||||
...tb,
|
||||
id: `${tb.id}-split${col}`,
|
||||
text: remainingWords.join(" "),
|
||||
bounds: {
|
||||
...tb.bounds,
|
||||
left: currentLeft,
|
||||
right: tb.bounds.right,
|
||||
},
|
||||
});
|
||||
remainingWords = [];
|
||||
} else {
|
||||
// Find how many words fit in this column segment proportionally
|
||||
const segmentWidth = segmentRight - currentLeft;
|
||||
const fractionOfTotal = segmentWidth / totalWidth;
|
||||
const approxChars = Math.round(fractionOfTotal * tb.text.length);
|
||||
// Walk words to find the split closest to the proportional point
|
||||
let charCount = 0;
|
||||
let splitIdx = 0;
|
||||
for (let w = 0; w < remainingWords.length; w++) {
|
||||
const nextCount = charCount + remainingWords[w].length + (w > 0 ? 1 : 0);
|
||||
if (nextCount > approxChars && splitIdx > 0) break;
|
||||
charCount = nextCount;
|
||||
splitIdx = w + 1;
|
||||
}
|
||||
if (splitIdx === 0) splitIdx = 1; // take at least one word
|
||||
if (splitIdx >= remainingWords.length) {
|
||||
// All remaining words fit here
|
||||
result.push({
|
||||
...tb,
|
||||
id: `${tb.id}-split${col}`,
|
||||
text: remainingWords.join(" "),
|
||||
bounds: {
|
||||
...tb.bounds,
|
||||
left: currentLeft,
|
||||
right: segmentRight,
|
||||
},
|
||||
});
|
||||
remainingWords = [];
|
||||
} else {
|
||||
const partWords = remainingWords.slice(0, splitIdx);
|
||||
result.push({
|
||||
...tb,
|
||||
id: `${tb.id}-split${col}`,
|
||||
text: partWords.join(" "),
|
||||
bounds: {
|
||||
...tb.bounds,
|
||||
left: currentLeft,
|
||||
right: segmentRight,
|
||||
},
|
||||
});
|
||||
remainingWords = remainingWords.slice(splitIdx);
|
||||
currentLeft = segmentRight;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Full grid table (H + V lines)
|
||||
// ---------------------------------------------------------------------------
|
||||
function buildCells(rows: number, cols: number): TableCell[] {
|
||||
const cells: TableCell[] = [];
|
||||
for (let row = 0; row < rows; row++) {
|
||||
for (let col = 0; col < cols; col++) {
|
||||
cells.push({ row, col, text: "", rowSpan: 1, colSpan: 1 });
|
||||
}
|
||||
}
|
||||
return cells;
|
||||
}
|
||||
|
||||
function buildTableGrid(
|
||||
pageNumber: number,
|
||||
yLines: number[],
|
||||
xLines: number[],
|
||||
filteredSegments: Segment[],
|
||||
textBoxes: TextBox[],
|
||||
): { grid: TableGrid; consumedIds: string[] } {
|
||||
let rows = yLines.length - 1;
|
||||
const cols = xLines.length - 1;
|
||||
const cells = buildCells(rows, cols);
|
||||
const consumedIds: string[] = [];
|
||||
const yMin = yLines[yLines.length - 1];
|
||||
const yMax = yLines[0];
|
||||
const xMin = xLines[0];
|
||||
const xMax = xLines[xLines.length - 1];
|
||||
// Split text boxes that span multiple columns before placement
|
||||
const splitBoxes = splitCrossColumnBoxes(textBoxes, xLines);
|
||||
// Track which split piece IDs get placed in cells, so we can consume
|
||||
// the original (unsplit) text box IDs too.
|
||||
const placedSplitIds = new Set<string>();
|
||||
// Look for header text boxes just above the grid.
|
||||
// Use the ORIGINAL (unsplit) text boxes for header detection so that
|
||||
// wide paragraph text isn't falsely split into column-sized header chunks.
|
||||
// Reject boxes wider than 1.5 columns — those are paragraph text, not headers.
|
||||
const avgColWidth = (xMax - xMin) / cols;
|
||||
const maxHeaderBoxWidth = avgColWidth * 1.5;
|
||||
const headerBoxes = textBoxes.filter(tb => {
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const boxWidth = tb.bounds.right - tb.bounds.left;
|
||||
return cy > yMax && cy <= yMax + 20 && cx >= xMin && cx <= xMax && boxWidth <= maxHeaderBoxWidth;
|
||||
});
|
||||
if (headerBoxes.length > 0) {
|
||||
rows += 1;
|
||||
for (const cell of cells) cell.row += 1;
|
||||
for (let col = 0; col < cols; col++) {
|
||||
cells.push({ row: 0, col, text: "", rowSpan: 1, colSpan: 1 });
|
||||
}
|
||||
for (const tb of headerBoxes) {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const col = xLines.findIndex((lineX, idx) => {
|
||||
const next = xLines[idx + 1];
|
||||
return next !== undefined && cx >= lineX && cx <= next;
|
||||
});
|
||||
if (col >= 0 && col < cols) {
|
||||
const cell = cells.find(c => c.row === 0 && c.col === col);
|
||||
if (cell) {
|
||||
cell.text = cell.text.length === 0 ? tb.text : `${cell.text} ${tb.text}`;
|
||||
consumedIds.push(tb.id);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
const cellBoxes = new Map<TableCell, TextBox[]>();
|
||||
for (const tb of splitBoxes) {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (cy < yMin || cy > yMax || cx < xMin || cx > xMax) continue;
|
||||
const rays = castRaysForTextBox(tb, filteredSegments);
|
||||
const rayConfidence = rays.filter(r => r.segmentId !== null).length;
|
||||
let row = yLines.findIndex((lineY, idx) => {
|
||||
const next = yLines[idx + 1];
|
||||
return next !== undefined && cy <= lineY && cy >= next;
|
||||
});
|
||||
if (row < 0 || row >= (headerBoxes.length > 0 ? rows - 1 : rows)) continue;
|
||||
if (headerBoxes.length > 0) row += 1;
|
||||
const col = xLines.findIndex((lineX, idx) => {
|
||||
const next = xLines[idx + 1];
|
||||
return next !== undefined && cx >= lineX && cx <= next;
|
||||
});
|
||||
if (col < 0 || col >= cols) continue;
|
||||
if (rayConfidence === 0) continue;
|
||||
const cell = cells.find(c => c.row === row && c.col === col);
|
||||
if (!cell) continue;
|
||||
if (!cellBoxes.has(cell)) cellBoxes.set(cell, []);
|
||||
cellBoxes.get(cell)?.push(tb);
|
||||
consumedIds.push(tb.id);
|
||||
if (tb.id.includes("-split")) placedSplitIds.add(tb.id);
|
||||
}
|
||||
rows = expandSubRowsByYClusters(rows, cols, cells, cellBoxes);
|
||||
// Merge text boxes within each cell into cell text
|
||||
for (const [cell, boxes] of cellBoxes.entries()) {
|
||||
boxes.sort((a, b) => b.bounds.top - a.bounds.top);
|
||||
const lines: string[] = [];
|
||||
let currentLine: string[] = [];
|
||||
let currentY = boxes[0].bounds.top;
|
||||
for (const box of boxes) {
|
||||
if (Math.abs(box.bounds.top - currentY) > 5) {
|
||||
lines.push(currentLine.join(" "));
|
||||
currentLine = [box.text];
|
||||
currentY = box.bounds.top;
|
||||
} else {
|
||||
currentLine.push(box.text);
|
||||
}
|
||||
}
|
||||
if (currentLine.length > 0) lines.push(currentLine.join(" "));
|
||||
cell.text = lines.join("<br>");
|
||||
}
|
||||
const grid = pruneEmptyRowsAndCols({
|
||||
pageNumber,
|
||||
rows,
|
||||
cols,
|
||||
cells,
|
||||
warnings: [],
|
||||
topY: yLines[0],
|
||||
isBorderless: false,
|
||||
});
|
||||
// Also consume the original (unsplit) text box IDs when any of their
|
||||
// split pieces were placed in a cell.
|
||||
for (const splitId of placedSplitIds) {
|
||||
const origId = splitId.replace(/-split\d+$/, "");
|
||||
if (!consumedIds.includes(origId)) {
|
||||
consumedIds.push(origId);
|
||||
}
|
||||
}
|
||||
return { grid, consumedIds };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// H-line-only table (inferred columns)
|
||||
// ---------------------------------------------------------------------------
|
||||
const COL_GAP_THRESHOLD = 20;
|
||||
const HONLY_ROW_GAP = 30;
|
||||
const HONLY_ROW_TOLERANCE = 8;
|
||||
const MIN_TABLE_HEIGHT = 24;
|
||||
const MIN_LEFT_SPREAD = 50;
|
||||
|
||||
function inferXLinesFromBoxes(textBoxes: TextBox[], xMin: number, xMax: number): number[] {
|
||||
const centers = textBoxes.map(tb => (tb.bounds.left + tb.bounds.right) / 2).sort((a, b) => a - b);
|
||||
if (centers.length === 0) return [xMin, xMax];
|
||||
const boundaries = [xMin];
|
||||
for (let i = 1; i < centers.length; i++) {
|
||||
if (centers[i] - centers[i - 1] >= COL_GAP_THRESHOLD) {
|
||||
boundaries.push((centers[i - 1] + centers[i]) / 2);
|
||||
}
|
||||
}
|
||||
boundaries.push(xMax);
|
||||
return boundaries;
|
||||
}
|
||||
|
||||
function buildHLineOnlyTable(
|
||||
pageNumber: number,
|
||||
yLines: number[],
|
||||
xMin: number,
|
||||
xMax: number,
|
||||
textBoxes: TextBox[],
|
||||
alreadyConsumed: Set<string>,
|
||||
): { grid: TableGrid; consumedIds: string[] } | null {
|
||||
const yMax = yLines[0];
|
||||
const yMin = yLines[yLines.length - 1];
|
||||
const candidates = textBoxes.filter(tb => !alreadyConsumed.has(tb.id));
|
||||
const BOX_LEFT_TOLERANCE = 30;
|
||||
const inRange = candidates.filter(tb => {
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
return (
|
||||
tb.bounds.left >= xMin - BOX_LEFT_TOLERANCE &&
|
||||
tb.bounds.right <= xMax + BOX_LEFT_TOLERANCE &&
|
||||
cy >= yMin &&
|
||||
cy <= yMax
|
||||
);
|
||||
});
|
||||
// Extend downward below yMin
|
||||
const belowYMin = candidates
|
||||
.filter(tb => {
|
||||
const cx = (tb.bounds.left + tb.bounds.right) / 2;
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
return cx >= xMin && cx <= xMax && cy < yMin;
|
||||
})
|
||||
.sort((a, b) => (b.bounds.top + b.bounds.bottom) / 2 - (a.bounds.top + a.bounds.bottom) / 2);
|
||||
const extensionBoxes: TextBox[] = [];
|
||||
let lastY = yMin;
|
||||
for (const tb of belowYMin) {
|
||||
const cy = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (lastY - cy > HONLY_ROW_GAP) break;
|
||||
extensionBoxes.push(tb);
|
||||
lastY = cy;
|
||||
}
|
||||
const allBoxes = [...inRange, ...extensionBoxes];
|
||||
if (allBoxes.length === 0) return null;
|
||||
const leftEdges = allBoxes.map(tb => tb.bounds.left);
|
||||
if (Math.max(...leftEdges) - Math.min(...leftEdges) < MIN_LEFT_SPREAD) return null;
|
||||
const xLines = inferXLinesFromBoxes(allBoxes, xMin, xMax);
|
||||
if (xLines.length < 2) return null;
|
||||
const cols = xLines.length - 1;
|
||||
// Build visual rows
|
||||
const visualRows: Array<{ midY: number; boxes: TextBox[] }> = [];
|
||||
const sortedBoxes = [...allBoxes].sort((a, b) => {
|
||||
const ya = (a.bounds.top + a.bounds.bottom) / 2;
|
||||
const yb = (b.bounds.top + b.bounds.bottom) / 2;
|
||||
if (Math.abs(ya - yb) > 0.5) return yb - ya;
|
||||
return a.bounds.left - b.bounds.left;
|
||||
});
|
||||
for (const box of sortedBoxes) {
|
||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
||||
const last = visualRows[visualRows.length - 1];
|
||||
if (last && Math.abs(last.midY - cy) <= HONLY_ROW_TOLERANCE) {
|
||||
last.boxes.push(box);
|
||||
} else {
|
||||
visualRows.push({ midY: cy, boxes: [box] });
|
||||
}
|
||||
}
|
||||
if (visualRows.length === 0) return null;
|
||||
const cells: TableCell[] = [];
|
||||
const consumedIds: string[] = [];
|
||||
for (let rowIdx = 0; rowIdx < visualRows.length; rowIdx++) {
|
||||
const vrow = visualRows[rowIdx];
|
||||
const colBoxes = new Map<number, TextBox[]>();
|
||||
for (const box of vrow.boxes) {
|
||||
const cx = (box.bounds.left + box.bounds.right) / 2;
|
||||
const col = xLines.findIndex((lineX, idx) => {
|
||||
const next = xLines[idx + 1];
|
||||
return next !== undefined && cx >= lineX && cx <= next;
|
||||
});
|
||||
if (col >= 0 && col < cols) {
|
||||
if (!colBoxes.has(col)) colBoxes.set(col, []);
|
||||
colBoxes.get(col)?.push(box);
|
||||
}
|
||||
}
|
||||
for (let c = 0; c < cols; c++) {
|
||||
const cbs = (colBoxes.get(c) ?? []).sort((a, b) => a.bounds.left - b.bounds.left);
|
||||
cells.push({
|
||||
row: rowIdx,
|
||||
col: c,
|
||||
text: cbs.map(b => b.text).join(" "),
|
||||
rowSpan: 1,
|
||||
colSpan: 1,
|
||||
});
|
||||
consumedIds.push(...cbs.map(b => b.id));
|
||||
}
|
||||
}
|
||||
const contentTopY = visualRows.length > 0 ? visualRows[0].midY : yMax;
|
||||
const grid = pruneEmptyRowsAndCols({
|
||||
pageNumber,
|
||||
rows: visualRows.length,
|
||||
cols,
|
||||
cells,
|
||||
warnings: [],
|
||||
topY: contentTopY,
|
||||
isBorderless: false,
|
||||
});
|
||||
return { grid, consumedIds };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Pruning
|
||||
// ---------------------------------------------------------------------------
|
||||
function pruneEmptyRowsAndCols(table: TableGrid): TableGrid {
|
||||
const occupiedRows = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.row));
|
||||
const occupiedCols = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.col));
|
||||
if (occupiedRows.size === 0) return table;
|
||||
const rowMap = new Map<number, number>();
|
||||
let newRow = 0;
|
||||
for (let r = 0; r < table.rows; r++) {
|
||||
if (occupiedRows.has(r)) rowMap.set(r, newRow++);
|
||||
}
|
||||
const colMap = new Map<number, number>();
|
||||
let newCol = 0;
|
||||
for (let c = 0; c < table.cols; c++) {
|
||||
if (occupiedCols.has(c)) colMap.set(c, newCol++);
|
||||
}
|
||||
const prunedCells = table.cells
|
||||
.filter(c => occupiedRows.has(c.row) && occupiedCols.has(c.col))
|
||||
.map(c => ({
|
||||
...c,
|
||||
row: rowMap.get(c.row) ?? c.row,
|
||||
col: colMap.get(c.col) ?? c.col,
|
||||
}));
|
||||
return { ...table, rows: newRow, cols: newCol, cells: prunedCells };
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Diagram vs table discrimination
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Maximum column count for a plausible data table. */
|
||||
const MAX_TABLE_COLS = 25;
|
||||
|
||||
/**
|
||||
* Returns true if a grid looks like a vector diagram rather than a data table.
|
||||
*
|
||||
* Heuristics (any match → diagram):
|
||||
* 1. Column count > 25 (diagrams create many X-lines from box edges)
|
||||
* 2. Fill ratio < 25% (most cells empty — scattered boxes)
|
||||
* 3. Fill < 50% AND duplicate text ratio > 30% (repeating labels in a
|
||||
* diagram layout, e.g. "Hash", "Transaction" appearing in each column)
|
||||
* 4. Fill < 50% AND cols >= 6 (moderate sparseness with wide grid)
|
||||
*/
|
||||
function isDiagram(grid: TableGrid): boolean {
|
||||
const totalCells = grid.rows * grid.cols;
|
||||
if (totalCells === 0) return true;
|
||||
const filled = grid.cells.filter(c => c.text.trim().length > 0);
|
||||
const fillRatio = filled.length / totalCells;
|
||||
// Very high column count
|
||||
if (grid.cols > MAX_TABLE_COLS) return true;
|
||||
// Very sparse
|
||||
if (fillRatio < 0.25) return true;
|
||||
// Compute duplicate text ratio among non-trivial cells.
|
||||
// Exclude short values (≤3 chars) like "—", "V", "YES", "NO" which
|
||||
// naturally repeat in real data tables.
|
||||
const substantive = filled.filter(c => c.text.trim().length > 3);
|
||||
const uniqueTexts = new Set(substantive.map(c => c.text.trim())).size;
|
||||
const dupRatio = substantive.length > 2 ? 1 - uniqueTexts / substantive.length : 0;
|
||||
// Sparse + highly duplicated substantive text → repeating diagram
|
||||
if (fillRatio < 0.5 && dupRatio > 0.3) return true;
|
||||
// High duplication + wide grid → repeating diagram even at moderate fill
|
||||
if (dupRatio > 0.4 && grid.cols >= 6) return true;
|
||||
// Sparse + wide grid with no substantive text to judge
|
||||
if (fillRatio < 0.4 && grid.cols >= 6) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect all table grids on a single page from its text boxes and segments.
|
||||
*/
|
||||
export function resolveTableGrids(pageNumber: number, textBoxes: TextBox[], segments: Segment[]): GridResult {
|
||||
const vertical = segments.filter(s => Math.abs(s.x1 - s.x2) <= AXIS_EPSILON);
|
||||
const horizontal = segments.filter(s => Math.abs(s.y1 - s.y2) <= AXIS_EPSILON);
|
||||
// Filter segments to the text's visible area
|
||||
const textYValues = textBoxes.flatMap(t => [t.bounds.bottom, t.bounds.top]);
|
||||
const textYMin = textYValues.length > 0 ? Math.min(...textYValues) - PAGE_MARGIN : -Infinity;
|
||||
const textYMax = textYValues.length > 0 ? Math.max(...textYValues) + PAGE_MARGIN : Infinity;
|
||||
const textXValues = textBoxes.flatMap(t => [t.bounds.left, t.bounds.right]);
|
||||
const textXMin = textXValues.length > 0 ? Math.min(...textXValues) - 100 : -Infinity;
|
||||
const textXMax = textXValues.length > 0 ? Math.max(...textXValues) + 100 : Infinity;
|
||||
const filteredH = horizontal.filter(
|
||||
s => s.y1 >= textYMin && s.y1 <= textYMax && s.x1 <= textXMax && s.x2 >= textXMin,
|
||||
);
|
||||
const hMaxX2 = filteredH.length > 0 ? Math.max(...filteredH.map(s => s.x2)) : textXMax;
|
||||
const vSegXMax = Math.max(textXMax, hMaxX2 + PAGE_MARGIN);
|
||||
const filteredV = vertical.filter(s => {
|
||||
const segMin = Math.min(s.y1, s.y2);
|
||||
const segMax = Math.max(s.y1, s.y2);
|
||||
return segMax >= textYMin && segMin <= textYMax && s.x1 >= textXMin && s.x1 <= vSegXMax;
|
||||
});
|
||||
const allYLines = uniqueSorted(filteredH.flatMap(s => [s.y1, s.y2])).sort((a, b) => b - a);
|
||||
if (allYLines.length < 2) {
|
||||
return { grids: [], consumedIds: [] };
|
||||
}
|
||||
const filteredSegments = [...filteredH, ...filteredV];
|
||||
const yGroups = splitYLinesIntoGroups(allYLines, filteredV);
|
||||
const grids: TableGrid[] = [];
|
||||
const gridConsumedIds: string[][] = [];
|
||||
// Flat set for the alreadyConsumed check in H-line-only tables
|
||||
const allConsumedIds: string[] = [];
|
||||
for (const yLines of yGroups) {
|
||||
if (yLines.length < 2) continue;
|
||||
const yMin = yLines[yLines.length - 1];
|
||||
const yMax = yLines[0];
|
||||
const groupVerticals = filteredV.filter(s => {
|
||||
const segMin = Math.min(s.y1, s.y2);
|
||||
const segMax = Math.max(s.y1, s.y2);
|
||||
return segMin < yMax - 1.5 && segMax > yMin + 1.5;
|
||||
});
|
||||
const groupXLines = uniqueSorted(groupVerticals.flatMap(s => [s.x1, s.x2]));
|
||||
if (groupXLines.length < 2) {
|
||||
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
|
||||
const groupHoriz = filteredH.filter(s => s.y1 >= yMin - 1.5 && s.y1 <= yMax + 1.5);
|
||||
if (groupHoriz.length === 0) continue;
|
||||
const hxMin = Math.min(...groupHoriz.map(s => s.x1));
|
||||
const hxMax = Math.max(...groupHoriz.map(s => s.x2));
|
||||
const result = buildHLineOnlyTable(pageNumber, yLines, hxMin, hxMax, textBoxes, new Set(allConsumedIds));
|
||||
if (result) {
|
||||
grids.push(result.grid);
|
||||
gridConsumedIds.push(result.consumedIds);
|
||||
allConsumedIds.push(...result.consumedIds);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (yMax - yMin < MIN_TABLE_HEIGHT) continue;
|
||||
const result = buildTableGrid(pageNumber, yLines, groupXLines, filteredSegments, textBoxes);
|
||||
grids.push(result.grid);
|
||||
gridConsumedIds.push(result.consumedIds);
|
||||
allConsumedIds.push(...result.consumedIds);
|
||||
}
|
||||
// Filter out grids that look like vector diagrams, not data tables.
|
||||
// Their consumed text box IDs are released so the text becomes free text.
|
||||
const filteredGrids: TableGrid[] = [];
|
||||
const filteredConsumedIds: string[] = [];
|
||||
for (let i = 0; i < grids.length; i++) {
|
||||
if (isDiagram(grids[i])) continue;
|
||||
filteredGrids.push(grids[i]);
|
||||
filteredConsumedIds.push(...gridConsumedIds[i]);
|
||||
}
|
||||
return { grids: filteredGrids, consumedIds: filteredConsumedIds };
|
||||
}
|
||||
@@ -0,0 +1,106 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Running header/footer detection and removal.
|
||||
*
|
||||
* Many PDFs have repeated text at the top or bottom of every page:
|
||||
* document titles, chapter names, page numbers, copyright notices.
|
||||
* These pollute the markdown output as false headings or noise.
|
||||
*
|
||||
* Algorithm:
|
||||
* 1. For each page, bucket text boxes by Y position (top/bottom zones)
|
||||
* 2. Collect the text content at each zone across all pages
|
||||
* 3. Text appearing on >20% of pages OR 8+ consecutive pages is a
|
||||
* running header/footer
|
||||
* 4. Remove matching text boxes before further processing
|
||||
*/
|
||||
import type { PageContent } from "./types";
|
||||
|
||||
/** Minimum number of pages to enable header/footer detection. */
|
||||
const MIN_PAGES = 5;
|
||||
/** Minimum Y position for top zone (from bottom of page in PDF coords). */
|
||||
const TOP_ZONE_MIN_Y = 700;
|
||||
/** Maximum Y position for bottom zone. */
|
||||
const BOTTOM_ZONE_MAX_Y = 80;
|
||||
/**
|
||||
* Minimum consecutive pages a text must appear on to be considered a
|
||||
* running header/footer. Catches both document-wide headers (appearing
|
||||
* on every page) and chapter-specific headers (appearing on 4+ consecutive
|
||||
* pages within a chapter).
|
||||
*/
|
||||
const MIN_CONSECUTIVE_PAGES = 8;
|
||||
|
||||
/**
|
||||
* Detect and remove running headers and footers from all pages.
|
||||
* Mutates the pages array in place, removing header/footer text boxes.
|
||||
*
|
||||
* Uses two strategies:
|
||||
* 1. Global frequency: text appearing on > 20% of all pages
|
||||
* 2. Consecutive runs: text appearing on 8+ consecutive pages
|
||||
*/
|
||||
export function stripHeadersFooters(pages: PageContent[]): void {
|
||||
if (pages.length < MIN_PAGES) return;
|
||||
// Step 1: Build per-page zone text sets
|
||||
const pageZoneTexts: Set<string>[] = [];
|
||||
for (const page of pages) {
|
||||
const zoneTexts = new Set<string>();
|
||||
for (const tb of page.textBoxes) {
|
||||
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (midY >= TOP_ZONE_MIN_Y || midY <= BOTTOM_ZONE_MAX_Y) {
|
||||
const key = tb.text.trim().replace(/\s+/g, " ");
|
||||
if (key.length > 0) zoneTexts.add(key);
|
||||
}
|
||||
}
|
||||
pageZoneTexts.push(zoneTexts);
|
||||
}
|
||||
// Step 2: Count global frequency AND longest consecutive run for each text
|
||||
const globalCount = new Map<string, number>();
|
||||
const maxConsecutive = new Map<string, number>();
|
||||
// Collect all unique zone texts
|
||||
const allTexts = new Set<string>();
|
||||
for (const zts of pageZoneTexts) {
|
||||
for (const t of zts) allTexts.add(t);
|
||||
}
|
||||
for (const text of allTexts) {
|
||||
let total = 0;
|
||||
let consecutive = 0;
|
||||
let maxRun = 0;
|
||||
for (const zts of pageZoneTexts) {
|
||||
if (zts.has(text)) {
|
||||
total++;
|
||||
consecutive++;
|
||||
if (consecutive > maxRun) maxRun = consecutive;
|
||||
} else {
|
||||
consecutive = 0;
|
||||
}
|
||||
}
|
||||
globalCount.set(text, total);
|
||||
maxConsecutive.set(text, maxRun);
|
||||
}
|
||||
// Step 3: Identify running headers/footers
|
||||
const globalThreshold = Math.max(3, Math.floor(pages.length * 0.2));
|
||||
const repeatedTexts = new Set<string>();
|
||||
for (const text of allTexts) {
|
||||
const gc = globalCount.get(text) ?? 0;
|
||||
const mc = maxConsecutive.get(text) ?? 0;
|
||||
// Global: appears on 20%+ of pages
|
||||
if (gc >= globalThreshold) {
|
||||
repeatedTexts.add(text);
|
||||
continue;
|
||||
}
|
||||
// Consecutive: appears on 8+ consecutive pages (chapter-level headers)
|
||||
if (mc >= MIN_CONSECUTIVE_PAGES) {
|
||||
repeatedTexts.add(text);
|
||||
}
|
||||
}
|
||||
if (repeatedTexts.size === 0) return;
|
||||
// Step 4: Remove matching text boxes from each page
|
||||
for (const page of pages) {
|
||||
page.textBoxes = page.textBoxes.filter(tb => {
|
||||
const midY = (tb.bounds.top + tb.bounds.bottom) / 2;
|
||||
if (midY < TOP_ZONE_MIN_Y && midY > BOTTOM_ZONE_MAX_Y) return true;
|
||||
const normalized = tb.text.trim().replace(/\s+/g, " ");
|
||||
return !repeatedTexts.has(normalized);
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* PDF to Markdown converter.
|
||||
*
|
||||
* Uses mupdf (native WASM) for fast PDF parsing and a custom pipeline for
|
||||
* table detection via vector line extraction + raycasting.
|
||||
*
|
||||
* Pipeline:
|
||||
* 1. Extract text boxes + vector segments + image regions per page (mupdf)
|
||||
* 2. Detect column layout (single vs multi-column)
|
||||
* 3. Per column: detect table grids from segments (grid detection + raycasting)
|
||||
* 4. Render diagrams as PNG files (if output directory provided)
|
||||
* 5. Render tables as markdown tables, free text as paragraphs/headings
|
||||
*/
|
||||
import * as path from "node:path";
|
||||
import type { ConversionResult, Converter, StreamInfo } from "../../types";
|
||||
import { detectColumns } from "./columns";
|
||||
import { extractPages, renderImageRegion } from "./extract";
|
||||
import { resolveTableGrids } from "./grid";
|
||||
import { stripHeadersFooters } from "./headers";
|
||||
import { renderPageContent } from "./render";
|
||||
import type { Segment, TextBox } from "./types";
|
||||
|
||||
const EXTENSIONS = [".pdf"];
|
||||
const MIMETYPES = ["application/pdf", "application/x-pdf"];
|
||||
|
||||
type ImageBlock = { topY: number; markdown: string };
|
||||
|
||||
/**
|
||||
* Process a set of text boxes (one column or full page): run table detection,
|
||||
* separate free text, and render to markdown.
|
||||
*/
|
||||
function processColumn(
|
||||
pageNumber: number,
|
||||
textBoxes: TextBox[],
|
||||
segments: Segment[],
|
||||
imageBlocks: ImageBlock[],
|
||||
): string {
|
||||
const { grids, consumedIds } = resolveTableGrids(pageNumber, textBoxes, segments);
|
||||
const consumedSet = new Set(consumedIds);
|
||||
const freeTextBoxes = textBoxes.filter(tb => !consumedSet.has(tb.id));
|
||||
return renderPageContent(freeTextBoxes, grids, imageBlocks, textBoxes);
|
||||
}
|
||||
|
||||
export class PdfConverter implements Converter {
|
||||
name = "pdf";
|
||||
|
||||
accepts(streamInfo: StreamInfo): boolean {
|
||||
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) {
|
||||
return true;
|
||||
}
|
||||
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) {
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const pdfBytes = new Uint8Array(input);
|
||||
const pages = await extractPages(pdfBytes);
|
||||
// Remove running headers/footers before processing.
|
||||
stripHeadersFooters(pages);
|
||||
const imageDir = streamInfo.imageDir;
|
||||
|
||||
const pageMarkdowns: string[] = [];
|
||||
for (const page of pages) {
|
||||
// Build image blocks for this page.
|
||||
const imageBlocks: ImageBlock[] = [];
|
||||
if (imageDir && page.images.length > 0) {
|
||||
for (const img of page.images) {
|
||||
const filename = `${img.id}.png`;
|
||||
const filepath = path.join(imageDir, filename);
|
||||
try {
|
||||
const png = renderImageRegion(pdfBytes, img);
|
||||
await Bun.write(filepath, png);
|
||||
imageBlocks.push({ topY: img.topY, markdown: `` });
|
||||
} catch {
|
||||
// Image rendering failed — skip.
|
||||
}
|
||||
}
|
||||
} else if (page.images.length > 0) {
|
||||
for (const img of page.images) {
|
||||
imageBlocks.push({
|
||||
topY: img.topY,
|
||||
markdown: `<!-- image: ${img.id} (page ${img.pageNumber}, ${img.bbox.w}x${img.bbox.h}pt) -->`,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Detect column layout.
|
||||
// If the page has vertical segments (tables), suppress column detection
|
||||
// when one detected column is very narrow — that's a table's first column,
|
||||
// not a page layout column.
|
||||
const layout = detectColumns(page.textBoxes);
|
||||
if (layout.columnCount > 1 && page.segments.some(s => Math.abs(s.x1 - s.x2) <= 0.8)) {
|
||||
const pageXMin = Math.min(...page.textBoxes.map(tb => tb.bounds.left));
|
||||
const pageXMax = Math.max(...page.textBoxes.map(tb => tb.bounds.right));
|
||||
const pageWidth = pageXMax - pageXMin;
|
||||
const minColFraction = 0.3;
|
||||
const tooNarrow = layout.columns.some(col => {
|
||||
const colXMin = Math.min(...col.map(tb => tb.bounds.left));
|
||||
const colXMax = Math.max(...col.map(tb => tb.bounds.right));
|
||||
return (colXMax - colXMin) / pageWidth < minColFraction;
|
||||
});
|
||||
if (tooNarrow) {
|
||||
layout.columnCount = 1;
|
||||
layout.columns = [page.textBoxes];
|
||||
layout.boundaries = [];
|
||||
}
|
||||
}
|
||||
|
||||
if (layout.columnCount === 1) {
|
||||
// Single column — process normally.
|
||||
const md = processColumn(page.pageNumber, page.textBoxes, page.segments, imageBlocks);
|
||||
if (md.length > 0) pageMarkdowns.push(md);
|
||||
} else {
|
||||
// Multi-column — process each column independently, then join.
|
||||
const columnMarkdowns: string[] = [];
|
||||
for (const colBoxes of layout.columns) {
|
||||
// Filter segments to those within this column's X range.
|
||||
const colXMin = Math.min(...colBoxes.map(tb => tb.bounds.left));
|
||||
const colXMax = Math.max(...colBoxes.map(tb => tb.bounds.right));
|
||||
const margin = 10;
|
||||
const colSegments = page.segments.filter(seg => {
|
||||
const segXMin = Math.min(seg.x1, seg.x2);
|
||||
const segXMax = Math.max(seg.x1, seg.x2);
|
||||
return segXMax >= colXMin - margin && segXMin <= colXMax + margin;
|
||||
});
|
||||
// Images go with the first column only (no X info to split by).
|
||||
const md = processColumn(
|
||||
page.pageNumber,
|
||||
colBoxes,
|
||||
colSegments,
|
||||
columnMarkdowns.length === 0 ? imageBlocks : [],
|
||||
);
|
||||
if (md.length > 0) columnMarkdowns.push(md);
|
||||
}
|
||||
const joined = columnMarkdowns.join("\n\n");
|
||||
if (joined.length > 0) pageMarkdowns.push(joined);
|
||||
}
|
||||
}
|
||||
|
||||
return { markdown: pageMarkdowns.join("\n\n") };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,501 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/**
|
||||
* Markdown rendering for PDF pages.
|
||||
*
|
||||
* Converts table grids and free text boxes into markdown, handling:
|
||||
* - Table grid → markdown table (`| col | col |`)
|
||||
* - Free text → paragraphs with heading detection (by font size)
|
||||
* - Content ordering (top-to-bottom via Y coordinate)
|
||||
* - Paragraph wrap merging (lines broken across PDF line boundaries)
|
||||
* - Page number removal
|
||||
*
|
||||
* Ported from @oharato/pdf2md-ts, stripped of CJK/TDnet-specific logic.
|
||||
*/
|
||||
import type { ContentBlock, TableGrid, TextBox } from "./types";
|
||||
|
||||
/** A free-text line grouped from horizontally adjacent text boxes. */
|
||||
interface RenderLine {
|
||||
text: string;
|
||||
topY: number;
|
||||
fontSize: number;
|
||||
isBold: boolean;
|
||||
isTabular: boolean;
|
||||
}
|
||||
|
||||
/** A content block carrying the Y of its last wrapped line during merging. */
|
||||
type WrapBlock = ContentBlock & { lastTopY: number };
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Utility
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Convert full-width ASCII characters (A→A, !→! etc.) to normal ASCII. */
|
||||
function normalizeFullWidthAscii(text: string): string {
|
||||
return text.replace(/[!-~]/g, ch => String.fromCharCode(ch.charCodeAt(0) - 0xfee0));
|
||||
}
|
||||
|
||||
function escapePipes(text: string): string {
|
||||
return normalizeFullWidthAscii(text).replaceAll("|", "\\|").replaceAll("\n", "<br>");
|
||||
}
|
||||
|
||||
/** Parse a markdown pipe-delimited row into cell strings. */
|
||||
function parsePipeRow(line: string): string[] {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed.startsWith("|") || !trimmed.endsWith("|")) return [];
|
||||
return trimmed
|
||||
.slice(1, -1)
|
||||
.split("|")
|
||||
.map(cell => cell.trim());
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Table rendering
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Render a TableGrid as a markdown table.
|
||||
*/
|
||||
export function renderTableToMarkdown(table: TableGrid): string {
|
||||
if (table.rows === 0 || table.cols === 0) return "";
|
||||
const matrix = Array.from({ length: table.rows }, () => Array.from({ length: table.cols }, () => ""));
|
||||
for (const cell of table.cells) {
|
||||
if (cell.row < table.rows && cell.col < table.cols) {
|
||||
matrix[cell.row][cell.col] = escapePipes(cell.text.trim());
|
||||
}
|
||||
}
|
||||
const normalized = normalizeShiftedSparseColumns(matrix);
|
||||
const promoted = promoteSubHeaderPrefixes(normalized);
|
||||
const header = `| ${promoted[0].join(" | ")} |`;
|
||||
const divider = `| ${Array.from({ length: promoted[0].length }, () => "---").join(" | ")} |`;
|
||||
const body = promoted
|
||||
.slice(1)
|
||||
.map(row => `| ${row.join(" | ")} |`)
|
||||
.join("\n");
|
||||
return [header, divider, body].filter(l => l.length > 0).join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
* Fix tables with ≥5 columns where sparse single-value columns are
|
||||
* misaligned. Shifts those values to the adjacent dense column and
|
||||
* removes the now-empty sparse columns.
|
||||
*/
|
||||
function normalizeShiftedSparseColumns(matrix: string[][]): string[][] {
|
||||
if (matrix.length === 0 || matrix[0].length < 5) return matrix;
|
||||
const _rows = matrix.length;
|
||||
const cols = matrix[0].length;
|
||||
const counts = Array.from({ length: cols }, (_, c) =>
|
||||
matrix.reduce((n, row) => n + (row[c].trim().length > 0 ? 1 : 0), 0),
|
||||
);
|
||||
const denseCols = new Set(
|
||||
counts
|
||||
.map((count, col) => ({ count, col }))
|
||||
.filter(({ col, count }) => col === 0 || count >= 2)
|
||||
.map(({ col }) => col),
|
||||
);
|
||||
const sparseCols = counts
|
||||
.map((count, col) => ({ count, col }))
|
||||
.filter(({ col, count }) => col > 0 && col < cols - 1 && count === 1)
|
||||
.map(({ col }) => col);
|
||||
if (sparseCols.length < 2 || denseCols.size < 4) return matrix;
|
||||
const moves: Array<{ from: number; to: number; row: number }> = [];
|
||||
for (const from of sparseCols) {
|
||||
const row = matrix.findIndex(r => r[from].trim().length > 0);
|
||||
const to = from + 1;
|
||||
if (row < 0) return matrix;
|
||||
if (!denseCols.has(to)) return matrix;
|
||||
if (matrix[row][to].trim().length > 0) return matrix;
|
||||
moves.push({ from, to, row });
|
||||
}
|
||||
const copy = matrix.map(row => [...row]);
|
||||
for (const { from, to, row } of moves) {
|
||||
copy[row][to] = copy[row][to].trim().length > 0 ? `${copy[row][to]} ${copy[row][from]}` : copy[row][from];
|
||||
copy[row][from] = "";
|
||||
}
|
||||
const keepCols = Array.from({ length: cols }, (_, c) => c).filter(c => copy.some(row => row[c].trim().length > 0));
|
||||
if (keepCols.length === cols) return copy;
|
||||
return copy.map(row => keepCols.map(c => row[c]));
|
||||
}
|
||||
|
||||
/**
|
||||
* When a data row has ≥2 parenthesized qualifiers in non-first columns
|
||||
* (and the first column is empty), promote them into the header row.
|
||||
*/
|
||||
function promoteSubHeaderPrefixes(matrix: string[][]): string[][] {
|
||||
if (matrix.length < 2) return matrix;
|
||||
const PAREN_RE = /^\([^)]{1,40}\)$/;
|
||||
const result = matrix.map(row => [...row]);
|
||||
const cols = matrix[0].length;
|
||||
const rowsToRemove = new Set<number>();
|
||||
for (let r = 1; r < result.length; r++) {
|
||||
if (rowsToRemove.has(r)) continue;
|
||||
const promotable: Array<{ col: number; prefix: string; isFullCell: boolean }> = [];
|
||||
for (let col = 1; col < cols; col++) {
|
||||
const cell = (result[r][col] ?? "").trim();
|
||||
if (!cell) continue;
|
||||
const parts = cell.split("<br>");
|
||||
if (parts.length === 1 && PAREN_RE.test(cell)) {
|
||||
promotable.push({ col, prefix: cell, isFullCell: true });
|
||||
} else if (parts.length >= 2 && PAREN_RE.test(parts[0].trim())) {
|
||||
promotable.push({
|
||||
col,
|
||||
prefix: parts[0].trim(),
|
||||
isFullCell: false,
|
||||
});
|
||||
}
|
||||
}
|
||||
if (promotable.length < 2) continue;
|
||||
if (promotable.some(p => p.isFullCell) && result[r][0].trim().length > 0) continue;
|
||||
for (const { col, prefix, isFullCell } of promotable) {
|
||||
result[0][col] = result[0][col].trim() ? `${result[0][col]} ${prefix}` : prefix;
|
||||
if (isFullCell) {
|
||||
result[r][col] = "";
|
||||
} else {
|
||||
const parts = result[r][col].split("<br>");
|
||||
result[r][col] = parts.slice(1).join("<br>");
|
||||
}
|
||||
}
|
||||
if (result[r].every(cell => cell.trim().length === 0)) {
|
||||
rowsToRemove.add(r);
|
||||
}
|
||||
}
|
||||
return result.filter((_, r) => !rowsToRemove.has(r));
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Free text rendering
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Y tolerance for grouping text boxes onto the same visual line. */
|
||||
const TEXT_LINE_Y_TOLERANCE = 3;
|
||||
/** Minimum X gap between adjacent boxes to mark line as tabular. */
|
||||
const TABULAR_X_GAP = 30;
|
||||
/**
|
||||
* Minimum font size (pts) to consider when computing the modal body font.
|
||||
* Tiny labels from diagrams, footnote markers, and superscripts are excluded
|
||||
* so they don't skew the modal toward small sizes.
|
||||
*/
|
||||
const MIN_BODY_FONT_SIZE = 7;
|
||||
|
||||
/**
|
||||
* Compute the most frequent font size among text boxes, ignoring very small
|
||||
* text that likely comes from diagrams, footnotes, or superscripts.
|
||||
*/
|
||||
function modalFontSize(textBoxes: TextBox[]): number {
|
||||
const counts = new Map<number, number>();
|
||||
for (const tb of textBoxes) {
|
||||
const size = Math.round((tb.fontSize ?? 0) * 10) / 10;
|
||||
if (size < MIN_BODY_FONT_SIZE) continue;
|
||||
counts.set(size, (counts.get(size) ?? 0) + 1);
|
||||
}
|
||||
let modal = 0;
|
||||
let maxCount = 0;
|
||||
for (const [size, count] of counts) {
|
||||
if (count > maxCount) {
|
||||
maxCount = count;
|
||||
modal = size;
|
||||
}
|
||||
}
|
||||
return modal;
|
||||
}
|
||||
|
||||
/** Group free text boxes into horizontal lines, sorted top-to-bottom. */
|
||||
function groupFreeTextIntoLines(textBoxes: TextBox[]): RenderLine[] {
|
||||
if (textBoxes.length === 0) return [];
|
||||
const sorted = [...textBoxes].sort((a, b) => {
|
||||
const ya = (a.bounds.top + a.bounds.bottom) / 2;
|
||||
const yb = (b.bounds.top + b.bounds.bottom) / 2;
|
||||
const dy = yb - ya;
|
||||
if (Math.abs(dy) > TEXT_LINE_Y_TOLERANCE) return dy;
|
||||
return a.bounds.left - b.bounds.left;
|
||||
});
|
||||
const lines: RenderLine[] = [];
|
||||
let curParts = [sorted[0].text];
|
||||
let curBoxes = [sorted[0]];
|
||||
let curY = (sorted[0].bounds.top + sorted[0].bounds.bottom) / 2;
|
||||
let curTopY = curY;
|
||||
let curFontSize = sorted[0].fontSize;
|
||||
let curIsBold = sorted[0].isBold;
|
||||
const finishLine = () => {
|
||||
let isTabular = false;
|
||||
for (let j = 1; j < curBoxes.length; j++) {
|
||||
if (curBoxes[j].bounds.left - curBoxes[j - 1].bounds.right > TABULAR_X_GAP) {
|
||||
isTabular = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
lines.push({
|
||||
text: curParts.join(" "),
|
||||
topY: curTopY,
|
||||
fontSize: curFontSize,
|
||||
isBold: curIsBold,
|
||||
isTabular,
|
||||
});
|
||||
};
|
||||
for (let i = 1; i < sorted.length; i++) {
|
||||
const box = sorted[i];
|
||||
const cy = (box.bounds.top + box.bounds.bottom) / 2;
|
||||
if (Math.abs(cy - curY) <= TEXT_LINE_Y_TOLERANCE) {
|
||||
curParts.push(box.text);
|
||||
curBoxes.push(box);
|
||||
curFontSize = Math.max(curFontSize, box.fontSize);
|
||||
curIsBold = curIsBold || box.isBold;
|
||||
} else {
|
||||
finishLine();
|
||||
curParts = [box.text];
|
||||
curBoxes = [box];
|
||||
curY = cy;
|
||||
curTopY = cy;
|
||||
curFontSize = box.fontSize;
|
||||
curIsBold = box.isBold;
|
||||
}
|
||||
}
|
||||
finishLine();
|
||||
return lines;
|
||||
}
|
||||
|
||||
/** Determine markdown heading prefix based on font size relative to body. */
|
||||
function headingPrefix(fontSize: number, bodyFontSize: number, isBold: boolean): string {
|
||||
if (bodyFontSize <= 0) return "";
|
||||
const ratio = fontSize / bodyFontSize;
|
||||
// Large headings (>2x body size)
|
||||
if (ratio >= 2.0) return "# ";
|
||||
// Medium headings (~1.5x body size)
|
||||
if (ratio >= 1.4) return "## ";
|
||||
// Small headings (bold and slightly larger)
|
||||
if (ratio >= 1.1 && isBold) return "### ";
|
||||
return "";
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Block merging
|
||||
// ---------------------------------------------------------------------------
|
||||
/** Merge consecutive blocks with the same heading prefix (wrapped headings). */
|
||||
function mergeConsecutiveHeadings(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
|
||||
if (blocks.length === 0) return [];
|
||||
const HEADING_RE = /^(#{1,6} )/;
|
||||
const maxGap = Math.max(bodyFS * 3, 30);
|
||||
const merged: ContentBlock[] = [];
|
||||
let cur: ContentBlock = { ...blocks[0] };
|
||||
for (let i = 1; i < blocks.length; i++) {
|
||||
const next = blocks[i];
|
||||
const curMatch = cur.content.match(HEADING_RE);
|
||||
const nextMatch = next.content.match(HEADING_RE);
|
||||
const gap = cur.topY - next.topY;
|
||||
if (curMatch && nextMatch && curMatch[1] === nextMatch[1] && gap <= maxGap) {
|
||||
cur = {
|
||||
topY: cur.topY,
|
||||
content: `${cur.content} ${next.content.slice(nextMatch[1].length)}`,
|
||||
isTabular: cur.isTabular || next.isTabular,
|
||||
};
|
||||
} else {
|
||||
merged.push(cur);
|
||||
cur = { ...next };
|
||||
}
|
||||
}
|
||||
merged.push(cur);
|
||||
return merged;
|
||||
}
|
||||
|
||||
/**
|
||||
* Merge consecutive plain-text blocks that are wrapped lines of the same paragraph.
|
||||
*/
|
||||
function mergeParagraphWraps(blocks: ContentBlock[], bodyFS: number): ContentBlock[] {
|
||||
if (blocks.length === 0 || bodyFS <= 0) return blocks;
|
||||
const HEADING_RE = /^#{1,6} /;
|
||||
const SENTENCE_END_RE = /[.!?…)\]]\s*$/;
|
||||
const maxGap = bodyFS * 2.0;
|
||||
const MIN_WRAP_LENGTH = 25;
|
||||
const merged: ContentBlock[] = [];
|
||||
let cur: WrapBlock = { ...blocks[0], lastTopY: blocks[0].topY };
|
||||
for (let i = 1; i < blocks.length; i++) {
|
||||
const next = blocks[i];
|
||||
const curIsBody = !HEADING_RE.test(cur.content) && !cur.content.startsWith("|");
|
||||
const nextIsBody = !HEADING_RE.test(next.content) && !next.content.startsWith("|");
|
||||
const gap = cur.lastTopY - next.topY;
|
||||
const isWrap =
|
||||
curIsBody &&
|
||||
nextIsBody &&
|
||||
!cur.isTabular &&
|
||||
!next.isTabular &&
|
||||
gap > 0 &&
|
||||
gap <= maxGap &&
|
||||
cur.content.length > MIN_WRAP_LENGTH &&
|
||||
!SENTENCE_END_RE.test(cur.content);
|
||||
if (isWrap) {
|
||||
cur = {
|
||||
topY: cur.topY,
|
||||
lastTopY: next.topY,
|
||||
content: `${cur.content.trimEnd()} ${next.content.trimStart()}`,
|
||||
isTabular: false,
|
||||
};
|
||||
} else {
|
||||
merged.push({ topY: cur.topY, content: cur.content });
|
||||
cur = { ...next, lastTopY: next.topY };
|
||||
}
|
||||
}
|
||||
merged.push({ topY: cur.topY, content: cur.content });
|
||||
return merged;
|
||||
}
|
||||
|
||||
/** Remove page number blocks near the bottom of the page. */
|
||||
function removePageNumbers(blocks: ContentBlock[]): ContentBlock[] {
|
||||
const PAGE_NUM_RE = /^(?:#{1,6}\s*)?\d+\s*$/;
|
||||
const BOTTOM_Y = 120;
|
||||
return blocks.filter((block, idx) => {
|
||||
const isBottom = idx >= blocks.length - 3;
|
||||
const isLowY = block.topY <= BOTTOM_Y;
|
||||
const isPageNum = PAGE_NUM_RE.test(block.content.trim());
|
||||
return !(isBottom && isLowY && isPageNum);
|
||||
});
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Detached first-column table reconstruction
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Fix tables where the first column was emitted as free text blocks
|
||||
* around a markdown table containing only the right-side columns.
|
||||
*
|
||||
* Detects: a plain-text header line with (N+1) tokens above an N-column
|
||||
* markdown table, plus short label lines whose count matches the table's
|
||||
* logical row count. Reconstructs into a proper (N+1)-column table.
|
||||
*/
|
||||
function normalizeDetachedFirstColumnTables(blocks: ContentBlock[]): ContentBlock[] {
|
||||
const HEADING_RE = /^#{1,6}\s/;
|
||||
const isTableBlock = (text: string) => text.trimStart().startsWith("|");
|
||||
const isPlainBlock = (text: string) => !HEADING_RE.test(text) && !isTableBlock(text);
|
||||
const isShortLabel = (text: string) => {
|
||||
const t = text.trim();
|
||||
return t.length > 0 && t.length <= 40;
|
||||
};
|
||||
const splitTokens = (text: string) =>
|
||||
text
|
||||
.trim()
|
||||
.split(/[ \t]+/)
|
||||
.filter(Boolean);
|
||||
const replacements = new Map<number, string>();
|
||||
const remove = new Set<number>();
|
||||
for (let tableIdx = 0; tableIdx < blocks.length; tableIdx++) {
|
||||
if (remove.has(tableIdx)) continue;
|
||||
const tableBlock = blocks[tableIdx];
|
||||
if (!isTableBlock(tableBlock.content)) continue;
|
||||
const tableLines = tableBlock.content
|
||||
.split("\n")
|
||||
.map(line => line.trim())
|
||||
.filter(line => line.startsWith("|"));
|
||||
const dataRows = tableLines
|
||||
.filter(line => !/^\|\s*[-: ]+\|/.test(line))
|
||||
.map(parsePipeRow)
|
||||
.filter(row => row.length > 0);
|
||||
if (dataRows.length === 0) continue;
|
||||
const cols = dataRows[0].length;
|
||||
if (cols < 2 || dataRows.some(row => row.length !== cols)) continue;
|
||||
// Expand by <br> count to get logical row count
|
||||
const logicalRows: string[][] = [];
|
||||
for (const row of dataRows) {
|
||||
const splitCells = row.map(cell => cell.split("<br>").map(p => p.trim()));
|
||||
const rowSpan = Math.max(...splitCells.map(parts => parts.length));
|
||||
for (let k = 0; k < rowSpan; k++) {
|
||||
logicalRows.push(splitCells.map(parts => parts[k] ?? ""));
|
||||
}
|
||||
}
|
||||
if (logicalRows.length < 2) continue;
|
||||
// Find header with (cols + 1) non-numeric tokens
|
||||
let headerIdx = -1;
|
||||
let headerTokens: string[] = [];
|
||||
for (let i = Math.max(0, tableIdx - 4); i <= tableIdx - 1; i++) {
|
||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
||||
if (!isPlainBlock(text)) continue;
|
||||
const tokens = splitTokens(text);
|
||||
if (tokens.length === cols + 1 && tokens.every(tok => !/[0-9]/.test(tok))) {
|
||||
headerIdx = i;
|
||||
headerTokens = tokens;
|
||||
}
|
||||
}
|
||||
if (headerIdx < 0) continue;
|
||||
// Collect short label lines above/below table
|
||||
const aboveLabels: Array<{ idx: number; text: string }> = [];
|
||||
for (let i = tableIdx - 1; i > headerIdx; i--) {
|
||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
||||
if (!isPlainBlock(text) || !isShortLabel(text)) break;
|
||||
aboveLabels.push({ idx: i, text });
|
||||
}
|
||||
aboveLabels.reverse();
|
||||
const belowLabels: Array<{ idx: number; text: string }> = [];
|
||||
for (let i = tableIdx + 1; i < blocks.length; i++) {
|
||||
const text = normalizeFullWidthAscii(blocks[i].content).trim();
|
||||
if (!isPlainBlock(text) || !isShortLabel(text)) break;
|
||||
belowLabels.push({ idx: i, text });
|
||||
}
|
||||
const labels = [...aboveLabels, ...belowLabels];
|
||||
if (labels.length !== logicalRows.length) continue;
|
||||
// Reconstruct the full table
|
||||
const normalizedLines: string[] = [];
|
||||
normalizedLines.push(`| ${headerTokens.join(" | ")} |`);
|
||||
normalizedLines.push(`| ${Array.from({ length: cols + 1 }, () => "---").join(" | ")} |`);
|
||||
for (let r = 0; r < logicalRows.length; r++) {
|
||||
normalizedLines.push(`| ${labels[r].text} | ${logicalRows[r].join(" | ")} |`);
|
||||
}
|
||||
replacements.set(tableIdx, normalizedLines.join("\n"));
|
||||
remove.add(headerIdx);
|
||||
for (const label of labels) remove.add(label.idx);
|
||||
}
|
||||
if (replacements.size === 0 && remove.size === 0) return blocks;
|
||||
const out: ContentBlock[] = [];
|
||||
for (let i = 0; i < blocks.length; i++) {
|
||||
if (remove.has(i)) continue;
|
||||
const replaced = replacements.get(i);
|
||||
if (replaced) {
|
||||
out.push({ topY: blocks[i].topY, content: replaced });
|
||||
} else {
|
||||
out.push(blocks[i]);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public API
|
||||
// ---------------------------------------------------------------------------
|
||||
/**
|
||||
* Render one page's content: free text and tables interleaved top-to-bottom.
|
||||
*/
|
||||
export function renderPageContent(
|
||||
freeTextBoxes: TextBox[],
|
||||
tables: TableGrid[],
|
||||
imageBlocks: Array<{ topY: number; markdown: string }> = [],
|
||||
allTextBoxes?: TextBox[],
|
||||
): string {
|
||||
const blocks: ContentBlock[] = [];
|
||||
// Use ALL text boxes (before table/diagram filtering) for modal font size,
|
||||
// so that diagram labels released as free text don't skew the body size.
|
||||
const bodyFS = modalFontSize(allTextBoxes ?? freeTextBoxes);
|
||||
// Free text lines
|
||||
for (const line of groupFreeTextIntoLines(freeTextBoxes)) {
|
||||
const prefix = headingPrefix(line.fontSize, bodyFS, line.isBold);
|
||||
blocks.push({
|
||||
topY: line.topY,
|
||||
content: prefix + line.text,
|
||||
isTabular: prefix === "" && line.isTabular,
|
||||
});
|
||||
}
|
||||
// Tables
|
||||
for (const table of tables) {
|
||||
const md = renderTableToMarkdown(table);
|
||||
if (md.length > 0) {
|
||||
blocks.push({ topY: table.topY, content: md });
|
||||
}
|
||||
}
|
||||
// Images
|
||||
for (const img of imageBlocks) {
|
||||
blocks.push({ topY: img.topY, content: img.markdown });
|
||||
}
|
||||
// Sort top-to-bottom (higher Y = higher on page = comes first)
|
||||
blocks.sort((a, b) => b.topY - a.topY);
|
||||
const cleaned = removePageNumbers(blocks);
|
||||
const headingsMerged = mergeConsecutiveHeadings(cleaned, bodyFS);
|
||||
const merged = mergeParagraphWraps(headingsMerged, bodyFS);
|
||||
const normalized = normalizeDetachedFirstColumnTables(merged);
|
||||
return normalized
|
||||
.map(b => b.content)
|
||||
.join("\n\n")
|
||||
.trim();
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
// Adapted from markit-ai (MIT). See ../../NOTICE.
|
||||
|
||||
/** Bounding box in PDF coordinate space (origin = bottom-left). */
|
||||
export type Bounds = {
|
||||
left: number;
|
||||
right: number;
|
||||
/** Higher value = higher on the page. */
|
||||
top: number;
|
||||
bottom: number;
|
||||
};
|
||||
|
||||
/** A text fragment with position and font metadata. */
|
||||
export type TextBox = {
|
||||
id: string;
|
||||
text: string;
|
||||
bounds: Bounds;
|
||||
pageNumber: number;
|
||||
/** Dominant font size in points. */
|
||||
fontSize: number;
|
||||
/** True if rendered bold (font name or rendering mode). */
|
||||
isBold: boolean;
|
||||
};
|
||||
|
||||
/** A horizontal or vertical line segment extracted from vector graphics. */
|
||||
export type Segment = {
|
||||
id: string;
|
||||
x1: number;
|
||||
y1: number;
|
||||
x2: number;
|
||||
y2: number;
|
||||
};
|
||||
|
||||
/** A single cell in a resolved table grid. */
|
||||
export type TableCell = {
|
||||
row: number;
|
||||
col: number;
|
||||
text: string;
|
||||
rowSpan: number;
|
||||
colSpan: number;
|
||||
};
|
||||
|
||||
/** A resolved table grid ready for markdown rendering. */
|
||||
export type TableGrid = {
|
||||
pageNumber: number;
|
||||
rows: number;
|
||||
cols: number;
|
||||
cells: TableCell[];
|
||||
warnings: string[];
|
||||
/** Top Y coordinate (PDF space: larger = higher on page). */
|
||||
topY: number;
|
||||
/** True for tables detected without vector borders. */
|
||||
isBorderless: boolean;
|
||||
};
|
||||
|
||||
/** An image/diagram region detected on a page. */
|
||||
export type ImageRegion = {
|
||||
id: string;
|
||||
pageNumber: number;
|
||||
/** Bounding box in mupdf coordinates (top-left origin). */
|
||||
bbox: {
|
||||
x: number;
|
||||
y: number;
|
||||
w: number;
|
||||
h: number;
|
||||
};
|
||||
/** Y position in PDF coordinates (bottom-left) for ordering. */
|
||||
topY: number;
|
||||
};
|
||||
|
||||
/** Result of extracting content from a single PDF page. */
|
||||
export type PageContent = {
|
||||
pageNumber: number;
|
||||
textBoxes: TextBox[];
|
||||
segments: Segment[];
|
||||
images: ImageRegion[];
|
||||
};
|
||||
|
||||
/** A block of rendered content (text paragraph or table). */
|
||||
export type ContentBlock = {
|
||||
topY: number;
|
||||
content: string;
|
||||
/** True if this line has wide gaps between text boxes (column headers). */
|
||||
isTabular?: boolean;
|
||||
};
|
||||
@@ -0,0 +1,325 @@
|
||||
// Adapted from markit-ai (MIT). See ../NOTICE.
|
||||
import * as path from "node:path";
|
||||
import { XMLParser } from "fast-xml-parser";
|
||||
import { unzip, unzipText } from "../../utils/zip";
|
||||
import type { ConversionResult, Converter, StreamInfo } from "../types";
|
||||
|
||||
const EXTENSIONS = [".pptx"];
|
||||
const MIMETYPES = ["application/vnd.openxmlformats-officedocument.presentationml.presentation"];
|
||||
|
||||
/** A text value: bare string/number, or a `{ "#text" }` node when the element carries attributes. */
|
||||
type XmlText = string | number | { "#text"?: string };
|
||||
|
||||
interface TextRun {
|
||||
"a:t"?: XmlText;
|
||||
}
|
||||
interface Paragraph {
|
||||
"a:r"?: TextRun | TextRun[];
|
||||
}
|
||||
interface TextBody {
|
||||
"a:p"?: Paragraph | Paragraph[];
|
||||
}
|
||||
interface CNvPr {
|
||||
"@_name": string;
|
||||
}
|
||||
interface Placeholder {
|
||||
"@_type": string;
|
||||
}
|
||||
interface NvPr {
|
||||
"p:ph"?: Placeholder;
|
||||
}
|
||||
interface NvSpPr {
|
||||
"p:cNvPr"?: CNvPr;
|
||||
"p:nvPr"?: NvPr;
|
||||
}
|
||||
interface NvPicPr {
|
||||
"p:cNvPr"?: CNvPr;
|
||||
}
|
||||
interface Shape {
|
||||
"p:txBody"?: TextBody;
|
||||
"p:nvSpPr"?: NvSpPr;
|
||||
}
|
||||
interface Blip {
|
||||
"@_r:embed": string;
|
||||
}
|
||||
interface BlipFill {
|
||||
"a:blip"?: Blip;
|
||||
}
|
||||
interface Picture {
|
||||
"p:blipFill"?: BlipFill;
|
||||
"p:nvSpPr"?: NvSpPr;
|
||||
"p:nvPicPr"?: NvPicPr;
|
||||
}
|
||||
interface TableCell {
|
||||
"a:txBody"?: TextBody;
|
||||
}
|
||||
interface TableRow {
|
||||
"a:tc"?: TableCell | TableCell[];
|
||||
}
|
||||
interface Table {
|
||||
"a:tr"?: TableRow | TableRow[];
|
||||
}
|
||||
interface GraphicData {
|
||||
"a:tbl"?: Table;
|
||||
}
|
||||
interface Graphic {
|
||||
"a:graphicData"?: GraphicData;
|
||||
}
|
||||
interface GraphicFrame {
|
||||
"a:graphic"?: Graphic;
|
||||
}
|
||||
interface SpTree {
|
||||
"p:sp"?: Shape | Shape[];
|
||||
"p:pic"?: Picture | Picture[];
|
||||
"p:graphicFrame"?: GraphicFrame | GraphicFrame[];
|
||||
}
|
||||
interface CSld {
|
||||
"p:spTree"?: SpTree;
|
||||
}
|
||||
interface SlideDoc {
|
||||
"p:sld"?: { "p:cSld"?: CSld };
|
||||
}
|
||||
interface NotesDoc {
|
||||
"p:notes"?: { "p:cSld"?: CSld };
|
||||
}
|
||||
interface SldId {
|
||||
"@_r:id": string;
|
||||
}
|
||||
interface PresentationDoc {
|
||||
"p:presentation"?: { "p:sldIdLst"?: { "p:sldId"?: SldId | SldId[] } };
|
||||
}
|
||||
interface Relationship {
|
||||
"@_Id": string;
|
||||
"@_Target": string;
|
||||
}
|
||||
interface RelationshipsDoc {
|
||||
Relationships?: { Relationship?: Relationship | Relationship[] };
|
||||
}
|
||||
|
||||
export class PptxConverter implements Converter {
|
||||
name = "pptx";
|
||||
|
||||
accepts(streamInfo: StreamInfo): boolean {
|
||||
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true;
|
||||
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const entries = unzip(input);
|
||||
const parser = new XMLParser({
|
||||
ignoreAttributes: false,
|
||||
attributeNamePrefix: "@_",
|
||||
textNodeName: "#text",
|
||||
processEntities: { maxTotalExpansions: 1_000_000 },
|
||||
});
|
||||
// Get slide order from presentation.xml
|
||||
const presXml = unzipText(entries, "ppt/presentation.xml");
|
||||
if (!presXml) throw new Error("Invalid PPTX: missing presentation.xml");
|
||||
const pres = parser.parse(presXml) as PresentationDoc;
|
||||
const sldIdList = pres["p:presentation"]?.["p:sldIdLst"]?.["p:sldId"];
|
||||
const sldIds = Array.isArray(sldIdList) ? sldIdList : sldIdList ? [sldIdList] : [];
|
||||
// Get relationship mappings
|
||||
const relsXml = unzipText(entries, "ppt/_rels/presentation.xml.rels");
|
||||
const rels = relsXml ? (parser.parse(relsXml) as RelationshipsDoc) : null;
|
||||
const relList = rels?.Relationships?.Relationship;
|
||||
const relArray = Array.isArray(relList) ? relList : relList ? [relList] : [];
|
||||
const relMap = new Map<string, string>();
|
||||
for (const r of relArray) {
|
||||
relMap.set(r["@_Id"], r["@_Target"]);
|
||||
}
|
||||
// Map slide IDs to file paths in order
|
||||
const slidePaths: string[] = [];
|
||||
for (const sld of sldIds) {
|
||||
const rId = sld["@_r:id"];
|
||||
const target = relMap.get(rId);
|
||||
if (target) slidePaths.push(`ppt/${target}`);
|
||||
}
|
||||
// If we couldn't resolve from rels, fall back to finding slide files
|
||||
if (slidePaths.length === 0) {
|
||||
const slideFiles = Object.keys(entries)
|
||||
.filter(f => /^ppt\/slides\/slide\d+\.xml$/.test(f))
|
||||
.sort((a, b) => {
|
||||
const na = parseInt(a.match(/slide(\d+)/)?.[1] || "0", 10);
|
||||
const nb = parseInt(b.match(/slide(\d+)/)?.[1] || "0", 10);
|
||||
return na - nb;
|
||||
});
|
||||
slidePaths.push(...slideFiles);
|
||||
}
|
||||
const imageDir = streamInfo.imageDir;
|
||||
const sections: string[] = [];
|
||||
let imageCount = 0;
|
||||
for (let i = 0; i < slidePaths.length; i++) {
|
||||
const slideXml = unzipText(entries, slidePaths[i]);
|
||||
if (!slideXml) continue;
|
||||
const slide = parser.parse(slideXml) as SlideDoc;
|
||||
const spTree = slide["p:sld"]?.["p:cSld"]?.["p:spTree"];
|
||||
if (!spTree) continue;
|
||||
// Parse slide-level rels for image references
|
||||
const slideRelsPath = `${slidePaths[i].replace("slides/slide", "slides/_rels/slide")}.rels`;
|
||||
const slideRelsXml = unzipText(entries, slideRelsPath);
|
||||
const slideRelMap = new Map<string, string>();
|
||||
if (slideRelsXml) {
|
||||
const slideRels = parser.parse(slideRelsXml) as RelationshipsDoc;
|
||||
const relItems = toList(slideRels?.Relationships?.Relationship);
|
||||
for (const r of relItems) {
|
||||
slideRelMap.set(r["@_Id"], r["@_Target"]);
|
||||
}
|
||||
}
|
||||
const slideLines = [`<!-- Slide ${i + 1} -->`];
|
||||
const shapes = spTree["p:sp"];
|
||||
const shapeList = Array.isArray(shapes) ? shapes : shapes ? [shapes] : [];
|
||||
let isTitle = true;
|
||||
for (const shape of shapeList) {
|
||||
const text = this.extractText(shape);
|
||||
if (!text) continue;
|
||||
if (isTitle) {
|
||||
slideLines.push(`# ${text}`);
|
||||
isTitle = false;
|
||||
} else {
|
||||
slideLines.push(text);
|
||||
}
|
||||
}
|
||||
// Extract embedded images
|
||||
const pics = toList(spTree["p:pic"]);
|
||||
for (const pic of pics) {
|
||||
const blipFill = pic["p:blipFill"];
|
||||
const rEmbed = blipFill?.["a:blip"]?.["@_r:embed"];
|
||||
if (!rEmbed) continue;
|
||||
const target = slideRelMap.get(rEmbed);
|
||||
if (!target) continue;
|
||||
// Resolve relative target against slide directory
|
||||
const imagePath = target.startsWith("/") ? target.slice(1) : `ppt/slides/${target}`;
|
||||
// Normalize path (e.g. ppt/slides/../media/image1.png → ppt/media/image1.png)
|
||||
const normalizedPath = imagePath
|
||||
.split("/")
|
||||
.reduce<string[]>((parts, seg) => {
|
||||
if (seg === "..") parts.pop();
|
||||
else parts.push(seg);
|
||||
return parts;
|
||||
}, [])
|
||||
.join("/");
|
||||
const buf = entries[normalizedPath];
|
||||
if (!buf) continue;
|
||||
imageCount++;
|
||||
const name =
|
||||
pic["p:nvSpPr"]?.["p:cNvPr"]?.["@_name"] ||
|
||||
pic["p:nvPicPr"]?.["p:cNvPr"]?.["@_name"] ||
|
||||
`image_${imageCount}`;
|
||||
if (imageDir) {
|
||||
try {
|
||||
const ext = normalizedPath.split(".").pop() || "png";
|
||||
const filename = `slide${i + 1}_${imageCount}.${ext}`;
|
||||
const filepath = path.join(imageDir, filename);
|
||||
await Bun.write(filepath, buf);
|
||||
slideLines.push(``);
|
||||
} catch {
|
||||
slideLines.push(`<!-- image: ${name} (slide ${i + 1}) -->`);
|
||||
}
|
||||
} else {
|
||||
slideLines.push(`<!-- image: ${name} (slide ${i + 1}) -->`);
|
||||
}
|
||||
}
|
||||
// Tables
|
||||
const graphicFrames = spTree["p:graphicFrame"];
|
||||
const gfList = Array.isArray(graphicFrames) ? graphicFrames : graphicFrames ? [graphicFrames] : [];
|
||||
for (const gf of gfList) {
|
||||
const table = this.extractTable(gf);
|
||||
if (table) slideLines.push(table);
|
||||
}
|
||||
// Slide notes
|
||||
const noteFile = slidePaths[i].replace("slides/slide", "notesSlides/notesSlide");
|
||||
const noteXml = unzipText(entries, noteFile);
|
||||
if (noteXml) {
|
||||
const note = parser.parse(noteXml) as NotesDoc;
|
||||
const noteSpTree = note["p:notes"]?.["p:cSld"]?.["p:spTree"];
|
||||
if (noteSpTree) {
|
||||
const noteShapes = noteSpTree["p:sp"];
|
||||
const noteList = Array.isArray(noteShapes) ? noteShapes : noteShapes ? [noteShapes] : [];
|
||||
const noteTexts: string[] = [];
|
||||
for (const ns of noteList) {
|
||||
// Skip slide image placeholder
|
||||
const phType = ns["p:nvSpPr"]?.["p:nvPr"]?.["p:ph"]?.["@_type"];
|
||||
if (phType === "sldImg") continue;
|
||||
const t = this.extractText(ns);
|
||||
if (t) noteTexts.push(t);
|
||||
}
|
||||
if (noteTexts.length > 0) {
|
||||
slideLines.push("\n### Notes:");
|
||||
slideLines.push(noteTexts.join("\n"));
|
||||
}
|
||||
}
|
||||
}
|
||||
sections.push(slideLines.join("\n"));
|
||||
}
|
||||
return { markdown: sections.join("\n\n").trim() };
|
||||
}
|
||||
|
||||
extractText(shape: Shape): string {
|
||||
const txBody = shape["p:txBody"];
|
||||
if (!txBody) return "";
|
||||
const paragraphs = txBody["a:p"];
|
||||
const pList = Array.isArray(paragraphs) ? paragraphs : paragraphs ? [paragraphs] : [];
|
||||
const lines: string[] = [];
|
||||
for (const p of pList) {
|
||||
const runs = p["a:r"];
|
||||
const rList = Array.isArray(runs) ? runs : runs ? [runs] : [];
|
||||
const parts: string[] = [];
|
||||
for (const r of rList) {
|
||||
const t = r["a:t"];
|
||||
if (t != null) parts.push(typeof t === "object" ? t["#text"] || "" : String(t));
|
||||
}
|
||||
if (parts.length > 0) lines.push(parts.join(""));
|
||||
}
|
||||
return lines.join("\n").trim();
|
||||
}
|
||||
|
||||
extractTable(gf: GraphicFrame): string | null {
|
||||
const tbl = gf?.["a:graphic"]?.["a:graphicData"]?.["a:tbl"];
|
||||
if (!tbl) return null;
|
||||
const rows = tbl["a:tr"];
|
||||
const rowList = Array.isArray(rows) ? rows : rows ? [rows] : [];
|
||||
if (rowList.length === 0) return null;
|
||||
const mdRows: string[][] = [];
|
||||
for (const row of rowList) {
|
||||
const cells = row["a:tc"];
|
||||
const cellList = Array.isArray(cells) ? cells : cells ? [cells] : [];
|
||||
const cellTexts: string[] = [];
|
||||
for (const cell of cellList) {
|
||||
const txBody = cell["a:txBody"];
|
||||
if (!txBody) {
|
||||
cellTexts.push("");
|
||||
continue;
|
||||
}
|
||||
const paragraphs = txBody["a:p"];
|
||||
const pList = Array.isArray(paragraphs) ? paragraphs : paragraphs ? [paragraphs] : [];
|
||||
const parts: string[] = [];
|
||||
for (const p of pList) {
|
||||
const runs = p["a:r"];
|
||||
const rList = Array.isArray(runs) ? runs : runs ? [runs] : [];
|
||||
for (const r of rList) {
|
||||
const t = r["a:t"];
|
||||
if (t != null) parts.push(typeof t === "object" ? t["#text"] || "" : String(t));
|
||||
}
|
||||
}
|
||||
cellTexts.push(parts.join(" "));
|
||||
}
|
||||
mdRows.push(cellTexts);
|
||||
}
|
||||
if (mdRows.length === 0) return null;
|
||||
const [header, ...body] = mdRows;
|
||||
const lines: string[] = [];
|
||||
lines.push(`| ${header.join(" | ")} |`);
|
||||
lines.push(`| ${header.map(() => "---").join(" | ")} |`);
|
||||
for (const row of body) {
|
||||
while (row.length < header.length) row.push("");
|
||||
lines.push(`| ${row.join(" | ")} |`);
|
||||
}
|
||||
return lines.join("\n");
|
||||
}
|
||||
}
|
||||
|
||||
function toList<T>(val: T | T[] | undefined): T[] {
|
||||
if (!val) return [];
|
||||
return Array.isArray(val) ? val : [val];
|
||||
}
|
||||
@@ -0,0 +1,173 @@
|
||||
// Adapted from markit-ai (MIT). See ../NOTICE.
|
||||
import { XMLParser } from "fast-xml-parser";
|
||||
import { unzip, unzipText } from "../../utils/zip";
|
||||
import type { ConversionResult, Converter, StreamInfo } from "../types";
|
||||
|
||||
const EXTENSIONS = [".xlsx"];
|
||||
const MIMETYPES = ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"];
|
||||
|
||||
/** A text value: bare string/number, or a `{ "#text" }` node when the element carries attributes. */
|
||||
type XmlText = string | number | { "#text"?: string };
|
||||
|
||||
interface RichTextRun {
|
||||
t?: XmlText;
|
||||
}
|
||||
interface StringItem {
|
||||
t?: XmlText;
|
||||
r?: RichTextRun | RichTextRun[];
|
||||
}
|
||||
interface Cell {
|
||||
"@_t"?: string;
|
||||
v?: string | number;
|
||||
is?: StringItem;
|
||||
}
|
||||
interface Row {
|
||||
c?: Cell | Cell[];
|
||||
}
|
||||
interface WorksheetDoc {
|
||||
worksheet?: { sheetData?: { row?: Row | Row[] } };
|
||||
}
|
||||
interface Sheet {
|
||||
"@_name": string;
|
||||
"@_r:id": string;
|
||||
}
|
||||
interface WorkbookDoc {
|
||||
workbook?: { sheets?: { sheet?: Sheet | Sheet[] } };
|
||||
}
|
||||
interface SharedStringsDoc {
|
||||
sst?: { si?: StringItem | StringItem[] };
|
||||
}
|
||||
interface Relationship {
|
||||
"@_Id": string;
|
||||
"@_Target": string;
|
||||
}
|
||||
interface RelationshipsDoc {
|
||||
Relationships?: { Relationship?: Relationship | Relationship[] };
|
||||
}
|
||||
|
||||
export class XlsxConverter implements Converter {
|
||||
name = "xlsx";
|
||||
|
||||
accepts(streamInfo: StreamInfo): boolean {
|
||||
if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true;
|
||||
if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
async convert(input: Buffer, _streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const entries = unzip(input);
|
||||
const parser = new XMLParser({
|
||||
ignoreAttributes: false,
|
||||
attributeNamePrefix: "@_",
|
||||
textNodeName: "#text",
|
||||
processEntities: { maxTotalExpansions: 1_000_000 },
|
||||
});
|
||||
// Parse shared strings
|
||||
const ssXml = unzipText(entries, "xl/sharedStrings.xml");
|
||||
const ss = ssXml ? (parser.parse(ssXml) as SharedStringsDoc) : null;
|
||||
const siList = ss?.sst?.si;
|
||||
const shared = toArray(siList);
|
||||
// Parse workbook for sheet names
|
||||
const wbXml = unzipText(entries, "xl/workbook.xml");
|
||||
if (!wbXml) throw new Error("Invalid XLSX: missing workbook.xml");
|
||||
const wb = parser.parse(wbXml) as WorkbookDoc;
|
||||
const sheets = toArray(wb.workbook?.sheets?.sheet);
|
||||
// Parse workbook rels to map rIds to sheet files
|
||||
const relsXml = unzipText(entries, "xl/_rels/workbook.xml.rels");
|
||||
const rels = relsXml ? (parser.parse(relsXml) as RelationshipsDoc) : null;
|
||||
const relList = toArray(rels?.Relationships?.Relationship);
|
||||
const relMap = new Map<string, string>();
|
||||
for (const r of relList) {
|
||||
relMap.set(r["@_Id"], r["@_Target"]);
|
||||
}
|
||||
const sections: string[] = [];
|
||||
for (const sheet of sheets) {
|
||||
const sheetName = sheet["@_name"];
|
||||
const rId = sheet["@_r:id"];
|
||||
const target = relMap.get(rId);
|
||||
if (!target) continue;
|
||||
const sheetPath = target.startsWith("/") ? target.slice(1) : `xl/${target}`;
|
||||
const sheetXml = unzipText(entries, sheetPath);
|
||||
if (!sheetXml) continue;
|
||||
const parsed = parser.parse(sheetXml) as WorksheetDoc;
|
||||
const rows = toArray(parsed.worksheet?.sheetData?.row);
|
||||
if (rows.length === 0) continue;
|
||||
// Extract all rows as string arrays
|
||||
const tableRows: string[][] = [];
|
||||
for (const row of rows) {
|
||||
const cells = toArray(row.c);
|
||||
const values: string[] = [];
|
||||
for (const cell of cells) {
|
||||
values.push(this.getCellValue(cell, shared));
|
||||
}
|
||||
tableRows.push(values);
|
||||
}
|
||||
if (tableRows.length === 0) continue;
|
||||
// Normalize column count
|
||||
const maxCols = Math.max(...tableRows.map(r => r.length));
|
||||
for (const row of tableRows) {
|
||||
while (row.length < maxCols) row.push("");
|
||||
}
|
||||
sections.push(`## ${sheetName}`);
|
||||
const [header, ...body] = tableRows;
|
||||
const lines: string[] = [];
|
||||
lines.push(`| ${header.join(" | ")} |`);
|
||||
lines.push(`| ${header.map(() => "---").join(" | ")} |`);
|
||||
for (const row of body) {
|
||||
lines.push(`| ${row.join(" | ")} |`);
|
||||
}
|
||||
sections.push(lines.join("\n"));
|
||||
}
|
||||
return { markdown: sections.join("\n\n") };
|
||||
}
|
||||
|
||||
getCellValue(cell: Cell, shared: StringItem[]): string {
|
||||
// Shared string
|
||||
if (cell["@_t"] === "s") {
|
||||
return this.getSharedString(shared, Number(cell.v));
|
||||
}
|
||||
// Inline string
|
||||
if (cell["@_t"] === "inlineStr") {
|
||||
const is = cell.is;
|
||||
if (!is) return "";
|
||||
if (is.t != null) return textValue(is.t);
|
||||
if (is.r)
|
||||
return toArray(is.r)
|
||||
.map(r => textValue(r.t))
|
||||
.join("");
|
||||
return "";
|
||||
}
|
||||
// Boolean
|
||||
if (cell["@_t"] === "b") {
|
||||
return cell.v === 1 || cell.v === "1" ? "TRUE" : "FALSE";
|
||||
}
|
||||
// Number or formula result
|
||||
if (cell.v != null) return String(cell.v);
|
||||
return "";
|
||||
}
|
||||
|
||||
getSharedString(shared: StringItem[], idx: number): string {
|
||||
const si = shared[idx];
|
||||
if (!si) return "";
|
||||
// Simple text
|
||||
if (si.t != null) return textValue(si.t);
|
||||
// Rich text runs
|
||||
if (si.r) {
|
||||
return toArray(si.r)
|
||||
.map(r => textValue(r.t))
|
||||
.join("");
|
||||
}
|
||||
return "";
|
||||
}
|
||||
}
|
||||
|
||||
function textValue(t: XmlText | undefined): string {
|
||||
if (t == null) return "";
|
||||
if (typeof t === "object") return t["#text"] || "";
|
||||
return String(t);
|
||||
}
|
||||
|
||||
function toArray<T>(val: T | T[] | undefined): T[] {
|
||||
if (!val) return [];
|
||||
return Array.isArray(val) ? val : [val];
|
||||
}
|
||||
@@ -0,0 +1,2 @@
|
||||
export * from "./registry";
|
||||
export * from "./types";
|
||||
@@ -0,0 +1,59 @@
|
||||
// Adapted from markit-ai (MIT). See ./NOTICE.
|
||||
import * as path from "node:path";
|
||||
import { DocxConverter } from "./converters/docx";
|
||||
import { EpubConverter } from "./converters/epub";
|
||||
import { PdfConverter } from "./converters/pdf";
|
||||
import { PptxConverter } from "./converters/pptx";
|
||||
import { XlsxConverter } from "./converters/xlsx";
|
||||
import type { ConversionResult, Converter, MarkitOptions, StreamInfo } from "./types";
|
||||
|
||||
/**
|
||||
* In-house document → markdown engine (replaces the `markit-ai` package).
|
||||
*
|
||||
* Only the document converters omp routes are registered (pdf, docx, pptx,
|
||||
* xlsx, epub). The first converter whose `accepts()` returns true and whose
|
||||
* `convert()` succeeds wins.
|
||||
*/
|
||||
export class Markit {
|
||||
readonly #converters: readonly Converter[];
|
||||
readonly #options: MarkitOptions;
|
||||
|
||||
constructor(options: MarkitOptions = {}) {
|
||||
this.#options = options;
|
||||
this.#converters = [
|
||||
new PdfConverter(),
|
||||
new DocxConverter(),
|
||||
new PptxConverter(),
|
||||
new XlsxConverter(),
|
||||
new EpubConverter(),
|
||||
];
|
||||
}
|
||||
|
||||
async convertFile(filePath: string, extra?: { imageDir?: string }): Promise<ConversionResult> {
|
||||
const buffer = Buffer.from(await Bun.file(filePath).arrayBuffer());
|
||||
const streamInfo: StreamInfo = {
|
||||
localPath: filePath,
|
||||
extension: path.extname(filePath).toLowerCase(),
|
||||
filename: path.basename(filePath),
|
||||
...extra,
|
||||
};
|
||||
return this.convert(buffer, streamInfo);
|
||||
}
|
||||
|
||||
async convert(input: Buffer, streamInfo: StreamInfo): Promise<ConversionResult> {
|
||||
const errors: { converter: string; error: Error }[] = [];
|
||||
for (const converter of this.#converters) {
|
||||
if (!converter.accepts(streamInfo)) continue;
|
||||
try {
|
||||
return await converter.convert(input, streamInfo, this.#options);
|
||||
} catch (err) {
|
||||
errors.push({ converter: converter.name, error: err instanceof Error ? err : new Error(String(err)) });
|
||||
}
|
||||
}
|
||||
if (errors.length > 0) {
|
||||
const details = errors.map(e => ` ${e.converter}: ${e.error.message}`).join("\n");
|
||||
throw new Error(`Conversion failed:\n${details}`);
|
||||
}
|
||||
throw new Error(`Unsupported format: ${streamInfo.extension || streamInfo.mimetype || "unknown"}`);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
// Adapted from markit-ai (MIT). See ./NOTICE.
|
||||
|
||||
export interface StreamInfo {
|
||||
mimetype?: string;
|
||||
extension?: string;
|
||||
charset?: string;
|
||||
filename?: string;
|
||||
localPath?: string;
|
||||
url?: string;
|
||||
/** Directory to write extracted images/diagrams. */
|
||||
imageDir?: string;
|
||||
}
|
||||
|
||||
export interface ConversionResult {
|
||||
markdown: string;
|
||||
title?: string;
|
||||
}
|
||||
|
||||
export interface MarkitOptions {
|
||||
/** Describe an image, return markdown. Receives raw bytes and mimetype. */
|
||||
describe?: (image: Buffer, mimetype: string) => Promise<string>;
|
||||
/** Transcribe audio, return text. Receives raw bytes and mimetype. */
|
||||
transcribe?: (audio: Buffer, mimetype: string) => Promise<string>;
|
||||
/** Extra instructions appended to the image description prompt. */
|
||||
prompt?: string;
|
||||
}
|
||||
|
||||
export interface Converter {
|
||||
/** Human-readable name for error messages. */
|
||||
name: string;
|
||||
/** Quick check: can this converter handle the given stream? */
|
||||
accepts(streamInfo: StreamInfo): boolean;
|
||||
/** Convert the source to markdown. */
|
||||
convert(input: Buffer, streamInfo: StreamInfo, options?: MarkitOptions): Promise<ConversionResult>;
|
||||
}
|
||||
@@ -1,7 +1,7 @@
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { inflateSync, strFromU8 } from "fflate";
|
||||
import { bytesToText, inflateRaw } from "../utils/zip";
|
||||
|
||||
import { formatBytes } from "./render-utils";
|
||||
import { ToolError } from "./tool-errors";
|
||||
@@ -417,7 +417,7 @@ function parseZipCentralDirectory(
|
||||
throw new ToolError("Invalid ZIP archive: truncated central directory entry");
|
||||
}
|
||||
|
||||
const rawPath = strFromU8(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0);
|
||||
const rawPath = bytesToText(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0);
|
||||
const normalizedPath = normalizeArchiveEntryPath(rawPath);
|
||||
if (normalizedPath) {
|
||||
const values = readZip64EntryValues(
|
||||
@@ -490,7 +490,7 @@ async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number):
|
||||
}
|
||||
|
||||
try {
|
||||
return inflateSync(compressedBytes, { out: new Uint8Array(uncompressedSize) });
|
||||
return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize));
|
||||
} catch (error) {
|
||||
throw new ToolError(error instanceof Error ? error.message : String(error));
|
||||
}
|
||||
|
||||
@@ -51,34 +51,9 @@ const CONVERTIBLE_MIMES = new Set([
|
||||
"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
"application/rtf",
|
||||
"application/epub+zip",
|
||||
"image/png",
|
||||
"image/jpeg",
|
||||
"image/gif",
|
||||
"image/webp",
|
||||
"audio/mpeg",
|
||||
"audio/wav",
|
||||
"audio/ogg",
|
||||
]);
|
||||
|
||||
const CONVERTIBLE_EXTENSIONS = new Set([
|
||||
".pdf",
|
||||
".doc",
|
||||
".docx",
|
||||
".ppt",
|
||||
".pptx",
|
||||
".xls",
|
||||
".xlsx",
|
||||
".rtf",
|
||||
".epub",
|
||||
".png",
|
||||
".jpg",
|
||||
".jpeg",
|
||||
".gif",
|
||||
".webp",
|
||||
".mp3",
|
||||
".wav",
|
||||
".ogg",
|
||||
]);
|
||||
const CONVERTIBLE_EXTENSIONS = new Set([".pdf", ".doc", ".docx", ".ppt", ".pptx", ".xls", ".xlsx", ".rtf", ".epub"]);
|
||||
|
||||
const NOTEBOOK_MIMES = new Set(["application/x-ipynb+json"]);
|
||||
const NOTEBOOK_EXTENSIONS = new Set([".ipynb"]);
|
||||
@@ -1144,45 +1119,25 @@ async function renderUrl(
|
||||
notes.push(
|
||||
`Image MIME type ${imageMimeType} is unsupported for inline model serialization; returning text metadata only`,
|
||||
);
|
||||
const shouldTryConvertibleFallback = isConvertible(mime, extHint);
|
||||
if (shouldTryConvertibleFallback) {
|
||||
notes.push("Attempting binary conversion fallback for unsupported image MIME type");
|
||||
} else {
|
||||
notes.push("Falling back to textual rendering from initial response");
|
||||
}
|
||||
skipConvertibleBinaryRetry = !shouldTryConvertibleFallback;
|
||||
notes.push("Falling back to textual rendering from initial response");
|
||||
skipConvertibleBinaryRetry = true;
|
||||
} else {
|
||||
const binary = await fetchBinary(finalUrl, timeout, signal);
|
||||
if (binary.ok) {
|
||||
notes.push("Fetched image binary");
|
||||
const conversionExtension = getExtensionHint(finalUrl, binary.contentDisposition) || extHint;
|
||||
let convertedText: string | null = null;
|
||||
const converted = await convertWithMarkit(binary.buffer, conversionExtension, timeout, signal);
|
||||
if (converted.ok) {
|
||||
if (converted.content.trim().length > 50) {
|
||||
notes.push("Converted with markit");
|
||||
convertedText = converted.content;
|
||||
} else {
|
||||
notes.push("markit conversion produced no usable output");
|
||||
}
|
||||
} else if (converted.error) {
|
||||
notes.push(`markit conversion failed: ${converted.error}`);
|
||||
} else {
|
||||
notes.push("markit conversion failed");
|
||||
}
|
||||
|
||||
if (binary.buffer.byteLength > MAX_INLINE_IMAGE_SOURCE_BYTES) {
|
||||
notes.push(
|
||||
`Image exceeds inline source limit (${binary.buffer.byteLength} bytes > ${MAX_INLINE_IMAGE_SOURCE_BYTES} bytes)`,
|
||||
);
|
||||
const output = finalizeOutput(
|
||||
convertedText ?? `Fetched image content (${imageMimeType}), but it is too large to inline render.`,
|
||||
`Fetched image content (${imageMimeType}), but it is too large to inline render.`,
|
||||
);
|
||||
return {
|
||||
url,
|
||||
finalUrl,
|
||||
contentType: imageMimeType,
|
||||
method: convertedText ? "markit" : "image-too-large",
|
||||
method: "image-too-large",
|
||||
content: output.content,
|
||||
fetchedAt,
|
||||
truncated: output.truncated,
|
||||
@@ -1199,15 +1154,13 @@ async function renderUrl(
|
||||
if (!isDecodedImage) {
|
||||
notes.push(`Fetched payload could not be decoded as ${imageMimeType}; returning text metadata only`);
|
||||
const output = finalizeOutput(
|
||||
convertedText ??
|
||||
rawContent ??
|
||||
`Fetched payload was labeled ${imageMimeType}, but bytes were not a valid image.`,
|
||||
rawContent ?? `Fetched payload was labeled ${imageMimeType}, but bytes were not a valid image.`,
|
||||
);
|
||||
return {
|
||||
url,
|
||||
finalUrl,
|
||||
contentType: imageMimeType,
|
||||
method: convertedText ? "markit" : "image-invalid",
|
||||
method: "image-invalid",
|
||||
content: output.content,
|
||||
fetchedAt,
|
||||
truncated: output.truncated,
|
||||
@@ -1219,13 +1172,13 @@ async function renderUrl(
|
||||
`Image exceeds inline output limit after resize (${resized.buffer.length} bytes > ${MAX_INLINE_IMAGE_OUTPUT_BYTES} bytes)`,
|
||||
);
|
||||
const output = finalizeOutput(
|
||||
convertedText ?? `Fetched image content (${imageMimeType}), but it is too large to inline render.`,
|
||||
`Fetched image content (${imageMimeType}), but it is too large to inline render.`,
|
||||
);
|
||||
return {
|
||||
url,
|
||||
finalUrl,
|
||||
contentType: imageMimeType,
|
||||
method: convertedText ? "markit" : "image-too-large",
|
||||
method: "image-too-large",
|
||||
content: output.content,
|
||||
fetchedAt,
|
||||
truncated: output.truncated,
|
||||
@@ -1234,7 +1187,7 @@ async function renderUrl(
|
||||
}
|
||||
|
||||
const dimensionNote = formatDimensionNote(resized);
|
||||
let imageSummary = convertedText ?? `Fetched image content (${resized.mimeType}).`;
|
||||
let imageSummary = `Fetched image content (${resized.mimeType}).`;
|
||||
if (dimensionNote) {
|
||||
imageSummary += `\n${dimensionNote}`;
|
||||
}
|
||||
|
||||
@@ -65,12 +65,6 @@ import { toolResult } from "./tool-result";
|
||||
const LOOSE_HASHLINE_HEADER_RE = /^\s*\[[^#\r\n]+#[^ \t\r\n]*\]\s*$/;
|
||||
const EXECUTABLE_NOTICE = "[Notice: Made executable via chmod +x]";
|
||||
|
||||
let fflateModulePromise: Promise<typeof import("fflate")> | undefined;
|
||||
async function loadFflate(): Promise<typeof import("fflate")> {
|
||||
if (!fflateModulePromise) fflateModulePromise = import("fflate");
|
||||
return fflateModulePromise;
|
||||
}
|
||||
|
||||
const writeSchema = type({
|
||||
path: type("string").describe("file path"),
|
||||
content: type("string").describe("file content"),
|
||||
@@ -387,8 +381,8 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
|
||||
if (resolvedArchivePath.exists) {
|
||||
try {
|
||||
const bytes = await Bun.file(resolvedArchivePath.absolutePath).bytes();
|
||||
const { unzipSync } = await loadFflate();
|
||||
const existing = unzipSync(new Uint8Array(bytes));
|
||||
const { unzip } = await import("../utils/zip");
|
||||
const existing = unzip(new Uint8Array(bytes));
|
||||
for (const [entryPath, data] of Object.entries(existing)) {
|
||||
zipEntries[entryPath.replace(/\\/g, "/")] = data;
|
||||
}
|
||||
@@ -400,8 +394,8 @@ export class WriteTool implements AgentTool<typeof writeSchema, WriteToolDetails
|
||||
zipEntries[resolvedArchivePath.archiveSubPath] = new TextEncoder().encode(content);
|
||||
|
||||
try {
|
||||
const { zipSync } = await loadFflate();
|
||||
const zipBuffer = zipSync(zipEntries);
|
||||
const { zip } = await import("../utils/zip");
|
||||
const zipBuffer = zip(zipEntries);
|
||||
await Bun.write(tmpPath, zipBuffer);
|
||||
await fs.rename(tmpPath, finalPath);
|
||||
} catch (error) {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { logger, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import type { Markit, StreamInfo } from "markit-ai";
|
||||
import type { Markit, StreamInfo } from "../markit";
|
||||
import { ToolAbortError } from "../tools/tool-errors";
|
||||
|
||||
export interface MarkitConversionResult {
|
||||
@@ -23,26 +23,27 @@ interface MuPdfWasmModuleConfig {
|
||||
printErr?: (...values: unknown[]) => void;
|
||||
}
|
||||
|
||||
declare global {
|
||||
var $libmupdf_wasm_Module: MuPdfWasmModuleConfig | undefined;
|
||||
}
|
||||
|
||||
function logMuPdfWasmOutput(stream: "stdout" | "stderr", values: unknown[]): void {
|
||||
const message = values.length === 1 && typeof values[0] === "string" ? values[0] : values.map(String).join(" ");
|
||||
logger.debug("mupdf wasm output", { stream, message });
|
||||
}
|
||||
|
||||
// `$libmupdf_wasm_Module` is declared globally (as `any`) by the mupdf package.
|
||||
// Install print hooks before the WASM module initializes so its stdout/stderr
|
||||
// route to the file logger instead of corrupting the TUI.
|
||||
function installMuPdfWasmLogger(): void {
|
||||
const moduleConfig = globalThis.$libmupdf_wasm_Module ?? {};
|
||||
moduleConfig.print = (...values) => logMuPdfWasmOutput("stdout", values);
|
||||
moduleConfig.printErr = (...values) => logMuPdfWasmOutput("stderr", values);
|
||||
const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {};
|
||||
moduleConfig.print = (...values: unknown[]) => logMuPdfWasmOutput("stdout", values);
|
||||
moduleConfig.printErr = (...values: unknown[]) => logMuPdfWasmOutput("stderr", values);
|
||||
globalThis.$libmupdf_wasm_Module = moduleConfig;
|
||||
}
|
||||
|
||||
installMuPdfWasmLogger();
|
||||
|
||||
let markit: () => Markit | Promise<Markit> = async () => {
|
||||
const promise = import("markit-ai").then(({ Markit }) => {
|
||||
// Lazy: keep the document engine (mammoth/fflate/mupdf) off the startup
|
||||
// import graph — it loads only when a document is first converted.
|
||||
const promise = import("../markit").then(({ Markit }) => {
|
||||
const instance = new Markit();
|
||||
markit = () => instance;
|
||||
return instance;
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
import TurndownService from "turndown";
|
||||
import { gfm } from "turndown-plugin-gfm";
|
||||
|
||||
type TurndownListParent = {
|
||||
nodeName: string;
|
||||
getAttribute(name: string): string | null;
|
||||
children: ArrayLike<unknown>;
|
||||
};
|
||||
|
||||
/**
|
||||
* Build a Turndown instance configured for GFM with the fixes omp relies on:
|
||||
* `~~strikethrough~~`, unescaped heading periods, and single-space list markers.
|
||||
*
|
||||
* Shared by the web scrapers (HTML → markdown) and the markit document engine
|
||||
* (`src/markit`). The rule set must stay identical across both call sites.
|
||||
*/
|
||||
export function createTurndown(): TurndownService {
|
||||
const turndown = new TurndownService({
|
||||
headingStyle: "atx",
|
||||
codeBlockStyle: "fenced",
|
||||
bulletListMarker: "-",
|
||||
});
|
||||
turndown.use(gfm);
|
||||
// GFM spec uses ~~ (double tilde), not ~ (single)
|
||||
turndown.addRule("strikethrough", {
|
||||
filter: ["del", "s", "strike"],
|
||||
replacement(content) {
|
||||
return `~~${content}~~`;
|
||||
},
|
||||
});
|
||||
// Unescape the backslash turndown inserts before periods in headings ("1." -> "1\.")
|
||||
turndown.addRule("heading", {
|
||||
filter: ["h1", "h2", "h3", "h4", "h5", "h6"],
|
||||
replacement(content, node) {
|
||||
const level = Number(node.nodeName.charAt(1));
|
||||
const prefix = "#".repeat(level);
|
||||
const cleaned = content.replace(/\\([.])/g, "$1").trim();
|
||||
return `\n\n${prefix} ${cleaned}\n\n`;
|
||||
},
|
||||
});
|
||||
// Single space after the marker (turndown hardcodes three)
|
||||
turndown.addRule("listItem", {
|
||||
filter: "li",
|
||||
replacement(content, node, options) {
|
||||
const body = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n ");
|
||||
const parent = node.parentNode as unknown as TurndownListParent | null;
|
||||
let prefix = `${options.bulletListMarker} `;
|
||||
if (parent?.nodeName === "OL") {
|
||||
const start = parent.getAttribute("start");
|
||||
const index = Array.prototype.indexOf.call(parent.children, node);
|
||||
prefix = `${(start ? Number(start) : 1) + index}. `;
|
||||
}
|
||||
return prefix + body + (node.nextSibling ? "\n" : "");
|
||||
},
|
||||
});
|
||||
return turndown;
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize HTML tables so turndown-plugin-gfm can render them:
|
||||
* - strip `<p>` tags inside `<td>`/`<th>` cells (joining paragraphs with a space)
|
||||
* - wrap the first row in `<thead>` when missing
|
||||
*/
|
||||
export function normalizeTablesHtml(html: string): string {
|
||||
let result = html.replace(
|
||||
/<(td|th)([^>]*)>([\s\S]*?)<\/(td|th)>/gi,
|
||||
(_match, tag: string, attrs: string, inner: string, closeTag: string) => {
|
||||
const stripped = inner
|
||||
.replace(/^\s*<p>/i, "")
|
||||
.replace(/<\/p>\s*$/i, "")
|
||||
.replace(/<\/p>\s*<p>/gi, " ");
|
||||
return `<${tag}${attrs}>${stripped}</${closeTag}>`;
|
||||
},
|
||||
);
|
||||
result = result.replace(
|
||||
/<table([^>]*)>\s*(?:<tbody>\s*)?(<tr[\s\S]*?<\/tr>)([\s\S]*?)<\/(?:tbody>\s*<\/)?table>/gi,
|
||||
(_match, attrs: string, firstRow: string, rest: string) => {
|
||||
const theadRow = firstRow.replace(/<td/gi, "<th").replace(/<\/td>/gi, "</th>");
|
||||
return `<table${attrs}><thead>${theadRow}</thead><tbody>${rest}</tbody></table>`;
|
||||
},
|
||||
);
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
// The single ZIP/DEFLATE boundary for the codebase. This is the ONLY module
|
||||
// that imports `fflate`; the markit document converters, the write tool, and
|
||||
// the archive reader all go through here so there is exactly one ZIP
|
||||
// implementation to reason about. Do not import `fflate` (or another archive
|
||||
// library) anywhere else.
|
||||
import type { Unzipped } from "fflate";
|
||||
import { inflateSync, strFromU8 } from "fflate";
|
||||
|
||||
export type { Unzipped } from "fflate";
|
||||
export { unzipSync as unzip, zipSync as zip } from "fflate";
|
||||
|
||||
/** Read a single ZIP entry as UTF-8 text, or `undefined` when the entry is absent. */
|
||||
export function unzipText(entries: Unzipped, entryPath: string): string | undefined {
|
||||
const data = entries[entryPath];
|
||||
return data ? strFromU8(data) : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Inflate a raw DEFLATE stream (a single deflate-compressed ZIP member). Pass a
|
||||
* preallocated `into` buffer when the uncompressed size is known up front.
|
||||
*/
|
||||
export function inflateRaw(bytes: Uint8Array, into?: Uint8Array): Uint8Array {
|
||||
return into ? inflateSync(bytes, { out: into }) : inflateSync(bytes);
|
||||
}
|
||||
|
||||
/** Decode raw bytes as text — UTF-8 by default, latin1 when `latin1` is set. */
|
||||
export function bytesToText(bytes: Uint8Array, latin1?: boolean): string {
|
||||
return strFromU8(bytes, latin1);
|
||||
}
|
||||
@@ -243,58 +243,15 @@ export async function loadPage(url: string, options: LoadPageOptions = {}): Prom
|
||||
/** Module-level Turndown instance — built lazily on first use. */
|
||||
let turndownPromise: Promise<TurndownService> | undefined;
|
||||
|
||||
type TurndownListParent = {
|
||||
nodeName: string;
|
||||
getAttribute(name: string): string | null;
|
||||
children: ArrayLike<unknown>;
|
||||
};
|
||||
|
||||
function getTurndown(): Promise<TurndownService> {
|
||||
turndownPromise ||= initTurndown();
|
||||
return turndownPromise;
|
||||
}
|
||||
|
||||
async function initTurndown(): Promise<TurndownService> {
|
||||
const [{ default: TurndownService }, { gfm }] = await Promise.all([
|
||||
import("turndown"),
|
||||
import("turndown-plugin-gfm"),
|
||||
]);
|
||||
const turndown = new TurndownService({
|
||||
headingStyle: "atx",
|
||||
codeBlockStyle: "fenced",
|
||||
bulletListMarker: "-",
|
||||
});
|
||||
turndown.use(gfm);
|
||||
turndown.addRule("strikethrough", {
|
||||
filter: ["del", "s", "strike"],
|
||||
replacement(content) {
|
||||
return `~~${content}~~`;
|
||||
},
|
||||
});
|
||||
turndown.addRule("heading", {
|
||||
filter: ["h1", "h2", "h3", "h4", "h5", "h6"],
|
||||
replacement(content, node) {
|
||||
const level = Number(node.nodeName.charAt(1));
|
||||
const prefix = "#".repeat(level);
|
||||
const cleaned = content.replace(/\\([.])/g, "$1").trim();
|
||||
return `\n\n${prefix} ${cleaned}\n\n`;
|
||||
},
|
||||
});
|
||||
turndown.addRule("listItem", {
|
||||
filter: "li",
|
||||
replacement(content, node, options) {
|
||||
content = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n ");
|
||||
const parent = node.parentNode as unknown as TurndownListParent | null;
|
||||
let prefix = `${options.bulletListMarker} `;
|
||||
if (parent?.nodeName === "OL") {
|
||||
const start = parent.getAttribute("start");
|
||||
const index = Array.prototype.indexOf.call(parent.children, node);
|
||||
prefix = `${(start ? Number(start) : 1) + index}. `;
|
||||
}
|
||||
return prefix + content + (node.nextSibling ? "\n" : "");
|
||||
},
|
||||
});
|
||||
return turndown;
|
||||
// Lazy import keeps turndown/turndown-plugin-gfm off the startup graph.
|
||||
const { createTurndown } = await import("../../utils/turndown");
|
||||
return createTurndown();
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,160 @@
|
||||
/**
|
||||
* Runtime coverage for the in-house markit document engine (src/markit), which
|
||||
* replaced the `markit-ai` package. Each format is generated in-memory via the
|
||||
* shared zip util (src/utils/zip) — no external fixtures — and converted through
|
||||
* the public wrapper (src/utils/markit), locking: docx text, xlsx tables, pptx
|
||||
* slides, epub metadata+spine, shared HTML-table normalization, image
|
||||
* extraction, the nested/relative zip path resolution (the JSZip→fflate
|
||||
* regression surface), and the unsupported-format error contract.
|
||||
*/
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { convertBufferWithMarkit, convertFileWithMarkit } from "@oh-my-pi/pi-coding-agent/utils/markit";
|
||||
import { zip } from "@oh-my-pi/pi-coding-agent/utils/zip";
|
||||
|
||||
const enc = (s: string): Uint8Array => new TextEncoder().encode(s);
|
||||
const WML = "http://schemas.openxmlformats.org/wordprocessingml/2006/main";
|
||||
|
||||
function makeDocx(bodyXml: string): Uint8Array {
|
||||
return zip({
|
||||
"[Content_Types].xml": enc(
|
||||
`<?xml version="1.0"?><Types xmlns="http://schemas.openxmlformats.org/package/2006/content-types"><Default Extension="rels" ContentType="application/vnd.openxmlformats-package.relationships+xml"/><Override PartName="/word/document.xml" ContentType="application/vnd.openxmlformats-officedocument.wordprocessingml.document.main+xml"/></Types>`,
|
||||
),
|
||||
"_rels/.rels": enc(
|
||||
`<?xml version="1.0"?><Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships"><Relationship Id="rId1" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/officeDocument" Target="word/document.xml"/></Relationships>`,
|
||||
),
|
||||
"word/document.xml": enc(
|
||||
`<?xml version="1.0"?><w:document xmlns:w="${WML}"><w:body>${bodyXml}</w:body></w:document>`,
|
||||
),
|
||||
});
|
||||
}
|
||||
|
||||
describe("markit converters", () => {
|
||||
it("converts docx paragraphs to markdown", async () => {
|
||||
const docx = makeDocx(
|
||||
`<w:p><w:r><w:t>First paragraph.</w:t></w:r></w:p><w:p><w:r><w:t>Second paragraph.</w:t></w:r></w:p>`,
|
||||
);
|
||||
const result = await convertBufferWithMarkit(docx, ".docx");
|
||||
expect(result.ok).toBe(true);
|
||||
expect(result.content).toBe("First paragraph.\n\nSecond paragraph.");
|
||||
});
|
||||
|
||||
it("converts xlsx sheets to markdown tables", async () => {
|
||||
const xlsx = zip({
|
||||
"xl/workbook.xml": enc(
|
||||
`<?xml version="1.0"?><workbook xmlns:r="r"><sheets><sheet name="People" sheetId="1" r:id="rId1"/></sheets></workbook>`,
|
||||
),
|
||||
"xl/_rels/workbook.xml.rels": enc(
|
||||
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="worksheets/sheet1.xml"/></Relationships>`,
|
||||
),
|
||||
"xl/worksheets/sheet1.xml": enc(
|
||||
`<?xml version="1.0"?><worksheet><sheetData><row><c t="inlineStr"><is><t>Name</t></is></c><c t="inlineStr"><is><t>Age</t></is></c></row><row><c t="inlineStr"><is><t>Alice</t></is></c><c><v>30</v></c></row></sheetData></worksheet>`,
|
||||
),
|
||||
});
|
||||
const result = await convertBufferWithMarkit(xlsx, ".xlsx");
|
||||
expect(result.ok).toBe(true);
|
||||
expect(result.content).toContain("## People");
|
||||
expect(result.content).toContain("| Name | Age |");
|
||||
expect(result.content).toContain("| --- | --- |");
|
||||
expect(result.content).toContain("| Alice | 30 |");
|
||||
});
|
||||
|
||||
it("reads an xlsx worksheet through an absolute (/-prefixed) rel target", async () => {
|
||||
const xlsx = zip({
|
||||
"xl/workbook.xml": enc(
|
||||
`<?xml version="1.0"?><workbook xmlns:r="r"><sheets><sheet name="S" sheetId="1" r:id="rId1"/></sheets></workbook>`,
|
||||
),
|
||||
"xl/_rels/workbook.xml.rels": enc(
|
||||
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="/xl/worksheets/sheet1.xml"/></Relationships>`,
|
||||
),
|
||||
"xl/worksheets/sheet1.xml": enc(
|
||||
`<?xml version="1.0"?><worksheet><sheetData><row><c t="inlineStr"><is><t>Header</t></is></c></row><row><c><v>42</v></c></row></sheetData></worksheet>`,
|
||||
),
|
||||
});
|
||||
const result = await convertBufferWithMarkit(xlsx, ".xlsx");
|
||||
expect(result.ok).toBe(true);
|
||||
expect(result.content).toContain("| Header |");
|
||||
expect(result.content).toContain("| 42 |");
|
||||
});
|
||||
|
||||
it("converts pptx slides with a title heading and body text", async () => {
|
||||
const pptx = zip({
|
||||
"ppt/presentation.xml": enc(
|
||||
`<?xml version="1.0"?><p:presentation xmlns:p="p" xmlns:r="r"><p:sldIdLst><p:sldId id="256" r:id="rId1"/></p:sldIdLst></p:presentation>`,
|
||||
),
|
||||
"ppt/_rels/presentation.xml.rels": enc(
|
||||
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="slides/slide1.xml"/></Relationships>`,
|
||||
),
|
||||
"ppt/slides/slide1.xml": enc(
|
||||
`<?xml version="1.0"?><p:sld xmlns:p="p" xmlns:a="a"><p:cSld><p:spTree><p:sp><p:txBody><a:p><a:r><a:t>The Title</a:t></a:r></a:p></p:txBody></p:sp><p:sp><p:txBody><a:p><a:r><a:t>Body line</a:t></a:r></a:p></p:txBody></p:sp></p:spTree></p:cSld></p:sld>`,
|
||||
),
|
||||
});
|
||||
const result = await convertBufferWithMarkit(pptx, ".pptx");
|
||||
expect(result.ok).toBe(true);
|
||||
expect(result.content).toContain("# The Title");
|
||||
expect(result.content).toContain("Body line");
|
||||
});
|
||||
|
||||
it("extracts a pptx image through a ../media relative rel target into imageDir", async () => {
|
||||
const dir = await fs.mkdtemp(path.join(os.tmpdir(), "markit-pptx-"));
|
||||
try {
|
||||
const pptx = zip({
|
||||
"ppt/presentation.xml": enc(
|
||||
`<?xml version="1.0"?><p:presentation xmlns:p="p" xmlns:r="r"><p:sldIdLst><p:sldId id="256" r:id="rId1"/></p:sldIdLst></p:presentation>`,
|
||||
),
|
||||
"ppt/_rels/presentation.xml.rels": enc(
|
||||
`<?xml version="1.0"?><Relationships><Relationship Id="rId1" Target="slides/slide1.xml"/></Relationships>`,
|
||||
),
|
||||
"ppt/slides/slide1.xml": enc(
|
||||
`<?xml version="1.0"?><p:sld xmlns:p="p" xmlns:a="a"><p:cSld><p:spTree><p:pic><p:nvPicPr><p:cNvPr name="Pic1"/></p:nvPicPr><p:blipFill><a:blip r:embed="rId2"/></p:blipFill></p:pic></p:spTree></p:cSld></p:sld>`,
|
||||
),
|
||||
"ppt/slides/_rels/slide1.xml.rels": enc(
|
||||
`<?xml version="1.0"?><Relationships><Relationship Id="rId2" Target="../media/image1.png"/></Relationships>`,
|
||||
),
|
||||
"ppt/media/image1.png": new Uint8Array([137, 80, 78, 71, 13, 10, 26, 10]),
|
||||
});
|
||||
const pptxPath = path.join(dir, "deck.pptx");
|
||||
const imageDir = path.join(dir, "imgs");
|
||||
await Bun.write(pptxPath, pptx);
|
||||
const result = await convertFileWithMarkit(pptxPath, undefined, { imageDir });
|
||||
expect(result.ok).toBe(true);
|
||||
const written = await fs.readdir(imageDir);
|
||||
expect(written).toHaveLength(1);
|
||||
expect(result.content).toContain(`](${path.join(imageDir, written[0]!)})`);
|
||||
} finally {
|
||||
await fs.rm(dir, { recursive: true, force: true });
|
||||
}
|
||||
});
|
||||
|
||||
it("converts epub spine, normalizes HTML tables, and resolves a non-root OPF basePath", async () => {
|
||||
const epub = zip({
|
||||
"META-INF/container.xml": enc(
|
||||
`<?xml version="1.0"?><container><rootfiles><rootfile full-path="OEBPS/content.opf"/></rootfiles></container>`,
|
||||
),
|
||||
"OEBPS/content.opf": enc(
|
||||
`<?xml version="1.0"?><package><metadata xmlns:dc="dc"><dc:title>Nested Book</dc:title><dc:creator>Ada</dc:creator></metadata><manifest><item id="c1" href="text/ch1.xhtml"/></manifest><spine><itemref idref="c1"/></spine></package>`,
|
||||
),
|
||||
"OEBPS/text/ch1.xhtml": enc(
|
||||
`<html><body><h2>Chapter One</h2><p>Body text.</p><table><tr><td>A</td><td>B</td></tr><tr><td>1</td><td>2</td></tr></table></body></html>`,
|
||||
),
|
||||
});
|
||||
const result = await convertBufferWithMarkit(epub, ".epub");
|
||||
expect(result.ok).toBe(true);
|
||||
expect(result.content).toContain("**Title:** Nested Book");
|
||||
expect(result.content).toContain("**Authors:** Ada");
|
||||
expect(result.content).toContain("## Chapter One");
|
||||
expect(result.content).toContain("Body text.");
|
||||
// normalizeTablesHtml promotes the first row to a header so GFM renders a table.
|
||||
expect(result.content).toContain("| A | B |");
|
||||
expect(result.content).toContain("| --- | --- |");
|
||||
});
|
||||
|
||||
it("reports an unsupported format instead of emitting garbage", async () => {
|
||||
const rtf = enc("{\\rtf1\\ansi binary-ish}");
|
||||
const result = await convertBufferWithMarkit(rtf, ".rtf");
|
||||
expect(result.ok).toBe(false);
|
||||
expect(result.error).toContain("Unsupported format");
|
||||
});
|
||||
});
|
||||
@@ -18,8 +18,8 @@ import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
|
||||
import { DEFAULT_FILE_LIMIT, MULTI_FILE_PER_FILE_MATCHES, SearchTool } from "@oh-my-pi/pi-coding-agent/tools/search";
|
||||
import * as toolTimeouts from "@oh-my-pi/pi-coding-agent/tools/tool-timeouts";
|
||||
import { WriteTool } from "@oh-my-pi/pi-coding-agent/tools/write";
|
||||
import { unzip } from "@oh-my-pi/pi-coding-agent/utils/zip";
|
||||
import { $which, Snowflake } from "@oh-my-pi/pi-utils";
|
||||
import { unzipSync } from "fflate";
|
||||
|
||||
// Helper to extract text from content blocks
|
||||
function getTextOutput(result: any): string {
|
||||
@@ -931,7 +931,7 @@ describe("Coding Agent Tools", () => {
|
||||
`Successfully wrote ${content.length} bytes to ${path.basename(archivePath)}:pkg/README.md`,
|
||||
);
|
||||
|
||||
const unzipped = unzipSync(new Uint8Array(fs.readFileSync(archivePath)));
|
||||
const unzipped = unzip(new Uint8Array(fs.readFileSync(archivePath)));
|
||||
expect(new TextDecoder().decode(unzipped["pkg/README.md"])).toBe(content);
|
||||
expect(new TextDecoder().decode(unzipped["pkg/src/index.ts"])).toBe("export const archiveValue = 1;\n");
|
||||
});
|
||||
|
||||
@@ -7,10 +7,10 @@ import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
|
||||
import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read";
|
||||
import { zip } from "@oh-my-pi/pi-coding-agent/utils/zip";
|
||||
import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types";
|
||||
import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils";
|
||||
import { Snowflake } from "@oh-my-pi/pi-utils";
|
||||
import { zipSync } from "fflate";
|
||||
|
||||
function makeSession(testDir: string): ToolSession {
|
||||
const sessionFile = path.join(testDir, "session.jsonl");
|
||||
@@ -112,7 +112,7 @@ describe("read URL binary dispatch", () => {
|
||||
});
|
||||
|
||||
it("lists a remote zip instead of dumping decoded bytes", async () => {
|
||||
const zipBytes = zipSync({
|
||||
const zipBytes = zip({
|
||||
"root.txt": Buffer.from("root file\n"),
|
||||
"nested/data.txt": Buffer.from("nested file\n"),
|
||||
});
|
||||
|
||||
@@ -167,11 +167,6 @@ describe("read tool URL handling", () => {
|
||||
ok: true,
|
||||
buffer: imageBytes,
|
||||
});
|
||||
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
|
||||
ok: false,
|
||||
content: "",
|
||||
error: "markit unavailable",
|
||||
});
|
||||
vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
|
||||
buffer: imageBytes,
|
||||
mimeType: "image/png",
|
||||
@@ -222,11 +217,7 @@ describe("read tool URL handling", () => {
|
||||
ok: true,
|
||||
buffer: new Uint8Array([137, 80, 78, 71]),
|
||||
});
|
||||
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
|
||||
ok: false,
|
||||
content: "",
|
||||
error: "markit unavailable",
|
||||
});
|
||||
const convertSpy = vi.spyOn(scraperUtils, "convertWithMarkit");
|
||||
|
||||
const result = await tool.execute("fetch-image-resized", { path: "https://example.com/image.png" });
|
||||
const imageBlock = result.content.find(
|
||||
@@ -240,52 +231,9 @@ describe("read tool URL handling", () => {
|
||||
expect(imageBlock?.data).toBe("cmVzaXplZA==");
|
||||
expect(textBlock?.type).toBe("text");
|
||||
expect(textBlock?.text).toContain("displayed at 1000x500");
|
||||
expect(convertSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("keeps markit extracted text for image responses", async () => {
|
||||
const session = createSession();
|
||||
const tool = new ReadTool(session);
|
||||
const extractedText = "Converted image text content that is definitely longer than fifty characters.";
|
||||
vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
|
||||
buffer: new Uint8Array([1, 2, 3]),
|
||||
mimeType: "image/png",
|
||||
originalWidth: 100,
|
||||
originalHeight: 100,
|
||||
width: 100,
|
||||
height: 100,
|
||||
wasResized: false,
|
||||
get data() {
|
||||
return "aW1hZ2U=";
|
||||
},
|
||||
});
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "image/png",
|
||||
finalUrl: "https://example.com/image.png",
|
||||
content: "",
|
||||
});
|
||||
vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({
|
||||
ok: true,
|
||||
buffer: new Uint8Array([137, 80, 78, 71]),
|
||||
});
|
||||
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
|
||||
ok: true,
|
||||
content: extractedText,
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-image-with-ocr", { path: "https://example.com/image.png" });
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
const imageBlock = result.content.find(
|
||||
(content): content is { type: "image"; data: string; mimeType: string } => content.type === "image",
|
||||
);
|
||||
|
||||
expect(result.details?.method).toBe("image");
|
||||
expect(textBlock?.type).toBe("text");
|
||||
expect(textBlock?.text).toContain(extractedText);
|
||||
expect(imageBlock?.mimeType).toBe("image/png");
|
||||
expect(imageBlock?.data).toBe("aW1hZ2U=");
|
||||
});
|
||||
it("falls back to text-only output for unsupported image MIME types", async () => {
|
||||
const session = createSession();
|
||||
const tool = new ReadTool(session);
|
||||
@@ -309,39 +257,6 @@ describe("read tool URL handling", () => {
|
||||
expect(textBlock?.text).toContain("<svg></svg>");
|
||||
});
|
||||
|
||||
it("uses binary conversion fallback for unsupported image MIME when extension is convertible", async () => {
|
||||
const session = createSession();
|
||||
const tool = new ReadTool(session);
|
||||
const convertedText = "Converted image text from markit fallback with sufficient length to pass threshold.";
|
||||
const fetchBinarySpy = vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({
|
||||
ok: true,
|
||||
buffer: new Uint8Array([255, 216, 255, 224]),
|
||||
});
|
||||
const convertSpy = vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
|
||||
ok: true,
|
||||
content: convertedText,
|
||||
});
|
||||
vi.spyOn(scrapers, "loadPage").mockResolvedValue({
|
||||
ok: true,
|
||||
status: 200,
|
||||
contentType: "image/jpg",
|
||||
finalUrl: "https://example.com/image.jpg",
|
||||
content: "\u0000\u0001garbage",
|
||||
});
|
||||
|
||||
const result = await tool.execute("fetch-image-jpg-fallback", { path: "https://example.com/image.jpg" });
|
||||
const imageBlock = result.content.find(content => content.type === "image");
|
||||
const textBlock = result.content.find(content => content.type === "text");
|
||||
|
||||
expect(result.details?.method).toBe("markit");
|
||||
expect(fetchBinarySpy).toHaveBeenCalledTimes(1);
|
||||
expect(convertSpy).toHaveBeenCalledTimes(1);
|
||||
expect(result.details?.notes).toContain("Attempting binary conversion fallback for unsupported image MIME type");
|
||||
expect(imageBlock).toBeUndefined();
|
||||
expect(textBlock?.type).toBe("text");
|
||||
expect(textBlock?.text).toContain(convertedText);
|
||||
});
|
||||
|
||||
it("does not treat text/html at .png paths as inline images", async () => {
|
||||
const session = createSession();
|
||||
const tool = new ReadTool(session);
|
||||
@@ -404,11 +319,6 @@ describe("read tool URL handling", () => {
|
||||
ok: true,
|
||||
buffer: new Uint8Array([60, 104, 116, 109, 108]),
|
||||
});
|
||||
vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({
|
||||
ok: false,
|
||||
content: "",
|
||||
error: "conversion failed",
|
||||
});
|
||||
vi.spyOn(imageResize, "resizeImage").mockResolvedValue({
|
||||
buffer: new Uint8Array([60, 104, 116, 109, 108]),
|
||||
mimeType: "image/png",
|
||||
|
||||
Reference in New Issue
Block a user