From 38d1d3a2f51aaa86bc0d5ca0b2aef4b3fa929a23 Mon Sep 17 00:00:00 2001 From: can1357 Date: Thu, 18 Jun 2026 18:59:35 +0200 Subject: [PATCH] feat(coding-agent): vendor relevant parts of markit --- .gitignore | 1 - biome.json | 1 - bun.lock | 53 +- package.json | 6 +- packages/coding-agent/CHANGELOG.md | 9 + packages/coding-agent/package.json | 6 +- packages/coding-agent/src/markit/NOTICE | 32 + .../src/markit/converters/docx.ts | 56 ++ .../src/markit/converters/epub.ts | 136 +++ .../src/markit/converters/mammoth.d.ts | 24 + .../src/markit/converters/pdf/columns.ts | 103 +++ .../src/markit/converters/pdf/extract.ts | 557 +++++++++++++ .../src/markit/converters/pdf/grid.ts | 780 ++++++++++++++++++ .../src/markit/converters/pdf/headers.ts | 106 +++ .../src/markit/converters/pdf/index.ts | 146 ++++ .../src/markit/converters/pdf/render.ts | 501 +++++++++++ .../src/markit/converters/pdf/types.ts | 84 ++ .../src/markit/converters/pptx.ts | 325 ++++++++ .../src/markit/converters/xlsx.ts | 173 ++++ packages/coding-agent/src/markit/index.ts | 2 + packages/coding-agent/src/markit/registry.ts | 59 ++ packages/coding-agent/src/markit/types.ts | 35 + .../coding-agent/src/tools/archive-reader.ts | 6 +- packages/coding-agent/src/tools/fetch.ts | 67 +- packages/coding-agent/src/tools/write.ts | 14 +- packages/coding-agent/src/utils/markit.ts | 19 +- packages/coding-agent/src/utils/turndown.ts | 83 ++ packages/coding-agent/src/utils/zip.ts | 29 + .../coding-agent/src/web/scrapers/types.ts | 49 +- .../test/markit-converters.test.ts | 160 ++++ packages/coding-agent/test/tools.test.ts | 4 +- .../test/tools/fetch-binary-dispatch.test.ts | 4 +- .../test/tools/fetch-kagi-toggle.test.ts | 94 +-- 33 files changed, 3450 insertions(+), 274 deletions(-) create mode 100644 packages/coding-agent/src/markit/NOTICE create mode 100644 packages/coding-agent/src/markit/converters/docx.ts create mode 100644 packages/coding-agent/src/markit/converters/epub.ts create mode 100644 packages/coding-agent/src/markit/converters/mammoth.d.ts create mode 100644 packages/coding-agent/src/markit/converters/pdf/columns.ts create mode 100644 packages/coding-agent/src/markit/converters/pdf/extract.ts create mode 100644 packages/coding-agent/src/markit/converters/pdf/grid.ts create mode 100644 packages/coding-agent/src/markit/converters/pdf/headers.ts create mode 100644 packages/coding-agent/src/markit/converters/pdf/index.ts create mode 100644 packages/coding-agent/src/markit/converters/pdf/render.ts create mode 100644 packages/coding-agent/src/markit/converters/pdf/types.ts create mode 100644 packages/coding-agent/src/markit/converters/pptx.ts create mode 100644 packages/coding-agent/src/markit/converters/xlsx.ts create mode 100644 packages/coding-agent/src/markit/index.ts create mode 100644 packages/coding-agent/src/markit/registry.ts create mode 100644 packages/coding-agent/src/markit/types.ts create mode 100644 packages/coding-agent/src/utils/turndown.ts create mode 100644 packages/coding-agent/src/utils/zip.ts create mode 100644 packages/coding-agent/test/markit-converters.test.ts diff --git a/.gitignore b/.gitignore index de59ee70f..5d17cfa1d 100644 --- a/.gitignore +++ b/.gitignore @@ -55,7 +55,6 @@ out.html pi-*.html # Generated files -packages/coding-agent/src/internal-urls/docs-index.generated.ts packages/coding-agent/src/export/html/tool-views.generated.js packages/natives/npm/ /runs/ diff --git a/biome.json b/biome.json index bd20dcfb1..b6bc9f1a5 100644 --- a/biome.json +++ b/biome.json @@ -60,7 +60,6 @@ "!**/vendor/**/*", "!**/node_modules/**/*", "!**/test-sessions.ts", - "!**/docs-index.generated.ts", "!**/agent_pb.ts", "!.worktrees/**/*", "!.wt/**/*" diff --git a/bun.lock b/bun.lock index ec6746d7a..98965b1b8 100644 --- a/bun.lock +++ b/bun.lock @@ -98,11 +98,13 @@ "arktype": "catalog:", "chalk": "catalog:", "diff": "catalog:", + "fast-xml-parser": "catalog:", "fflate": "catalog:", "handlebars": "catalog:", "linkedom": "catalog:", "lru-cache": "catalog:", - "markit-ai": "catalog:", + "mammoth": "catalog:", + "mupdf": "catalog:", "puppeteer-core": "catalog:", "turndown": "catalog:", "turndown-plugin-gfm": "catalog:", @@ -277,7 +279,6 @@ "version": "16.0.7", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", - "beautiful-mermaid": "catalog:", "handlebars": "catalog:", "winston": "catalog:", "winston-daily-rotate-file": "catalog:", @@ -311,7 +312,6 @@ }, "patchedDependencies": { "@ark/schema@0.56.0": "patches/@ark%2Fschema@0.56.0.patch", - "beautiful-mermaid@1.1.3": "patches/beautiful-mermaid@1.1.3.patch", }, "catalog": { "@agentclientprotocol/sdk": "0.25.0", @@ -355,11 +355,11 @@ "@typescript/native-preview": "7.0.0-dev.20260609.1", "@xterm/headless": "^6.0.0", "arktype": "^2.2.0", - "beautiful-mermaid": "^1.1.3", "chalk": "^5.6.2", "chart.js": "^4.5.1", "date-fns": "^4.4.0", "diff": "^9.0.0", + "fast-xml-parser": "^5.9.0", "fastembed": "2.1.0", "fflate": "0.8.3", "ghostty-web": "^0.4.0", @@ -368,8 +368,9 @@ "lint-staged": "^17.0.7", "lru-cache": "11.5.1", "lucide-react": "^1.17.0", + "mammoth": "^1.12.0", "marked": "^18.0.5", - "markit-ai": "0.5.3", + "mupdf": "^1.27.0", "onnxruntime-node": "1.26.0", "partial-json": "^0.1.7", "postcss": "^8.5.15", @@ -460,8 +461,6 @@ "@biomejs/cli-win32-x64": ["@biomejs/cli-win32-x64@2.5.0", "", { "os": "win32", "cpu": "x64" }, "sha512-VT/lF+GId+67j8aDfLkxdxNoVApsPSTbyAtB3jJq0IWTrY77WXfbPfpngxq0bA6JCEv/7k8C9qWjDRKRznDlyw=="], - "@borewit/text-codec": ["@borewit/text-codec@0.2.2", "", {}, "sha512-DDaRehssg1aNrH4+2hnj1B7vnUGEjU6OIlyRdkMd0aUdIUvKXrJfXsy8LVtXAy7DRvYVluWbMspsRhz2lcW0mQ=="], - "@bufbuild/protobuf": ["@bufbuild/protobuf@2.12.0", "", {}, "sha512-B/XlCaFIP8LOwzo+bz5uFzATYokcwCKQcghqnlfwSmM5eX/qTkvDBnDPs+gXtX/RyjxJ4DRikECcPJbyALA8FA=="], "@colors/colors": ["@colors/colors@1.6.0", "", {}, "sha512-Ir+AOibqzrIsL6ajt3Rz3LskB7OiMVHqltZmspbW/TJuTVuyOMirVqAkjfY6JISiLHgyNqicAC8AyHHGzNd/dA=="], @@ -860,10 +859,6 @@ "@tailwindcss/vite": ["@tailwindcss/vite@4.3.1", "", { "dependencies": { "@tailwindcss/node": "4.3.1", "@tailwindcss/oxide": "4.3.1", "tailwindcss": "4.3.1" }, "peerDependencies": { "vite": "^5.2.0 || ^6 || ^7 || ^8" } }, "sha512-hItDHuIIlEV61R+faXu66s1K36aTurO/Qw0e45Vskz57gXl9pWOT6eg3zmcEui6CZXddbN7zd41bwmvag4JGwQ=="], - "@tokenizer/inflate": ["@tokenizer/inflate@0.4.1", "", { "dependencies": { "debug": "^4.4.3", "token-types": "^6.1.1" } }, "sha512-2mAv+8pkG6GIZiF1kNg1jAjh27IDxEPKwdGul3snfztFerfPGI1LjDezZp3i7BElXompqEtPmoPx6c2wgtWsOA=="], - - "@tokenizer/token": ["@tokenizer/token@0.3.0", "", {}, "sha512-OvjF+z51L3ov0OyAU0duzsYuvO01PH7x4t6DJx+guahgTnBHkhJdG7soQeTSFLWN3efnHyibZ4Z8l2EuWwJN3A=="], - "@ts-morph/common": ["@ts-morph/common@0.29.0", "", { "dependencies": { "minimatch": "^10.0.1", "path-browserify": "^1.0.1", "tinyglobby": "^0.2.14" } }, "sha512-35oUmphHbJvQ/+UTwFNme/t2p3FoKiGJ5auTjjpNTop2dyREspirjMy82PLSC1pnDJ8ah1GU98hwpVt64YXQsg=="], "@tybys/wasm-util": ["@tybys/wasm-util@0.10.2", "", { "dependencies": { "tslib": "^2.4.0" } }, "sha512-RoBvJ2X0wuKlWFIjrwffGw1IqZHKQqzIchKaadZZfnNpsAYp2mM0h36JtPCjNDAHGgYez/15uMBpfGwchhiMgg=="], @@ -936,8 +931,6 @@ "baseline-browser-mapping": ["baseline-browser-mapping@2.10.37", "", { "bin": { "baseline-browser-mapping": "dist/cli.cjs" } }, "sha512-girxaJ7WZssDOFhzCGZTDKoTa1gk6A1TbflaYTpykLJ4UU9Fz9kx1aREM8JCuoVHbL8X8T/mJg7w2oYSq72Oig=="], - "beautiful-mermaid": ["beautiful-mermaid@1.1.3", "", { "dependencies": { "elkjs": "^0.11.0", "entities": "^7.0.1" } }, "sha512-TItrtrAyHp1vwFfFVYauWGrquouk/6SS21Aq3RsxindSYZODcN4xYrPZD6BiZRU+o5mKJzDPz9MUSMvELdylyg=="], - "before-after-hook": ["before-after-hook@4.0.0", "", {}, "sha512-q6tR3RPqIB1pMiTRMFcZwuG5T8vwp+vUvEG0vuI6B+Rikh5BfPp2fQ82c925FOs+b0lcFQ8CFrL+KbilfZFhOQ=="], "bluebird": ["bluebird@3.4.7", "", {}, "sha512-iD3898SR7sWVRHbiQv+sHUtHnMvC1o3nW5rAcqnq3uOn07DSAppZYUkIGslDz6gXC7HfunPe7YVBgoEJASPcHA=="], @@ -988,8 +981,6 @@ "colorette": ["colorette@2.0.20", "", {}, "sha512-IfEDxwoWIjkeXL1eXcDiow4UbKjhLdq6/EuSVR9GMN7KVH3r9gQ83e73hsz1Nd1T3ijd5xv1wcWRYO+D6kCI2w=="], - "commander": ["commander@14.0.3", "", {}, "sha512-H+y0Jo/T1RZ9qPP4Eh1pkcQcLRglraJaSLoyOtHxu6AapkjWVCy2Sit1QQ4x3Dng8qDlSsZEet7g5Pq06MvTgw=="], - "content-type": ["content-type@2.0.0", "", {}, "sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ=="], "convert-source-map": ["convert-source-map@2.0.0", "", {}, "sha512-Kvp459HrV2FEJ1CAsi1Ku+MY3kasH19TFykTz2xWmMeq6bk2NU3XXvfJ+Q61m0xktWwt+1HSYf3JZsTms3aRJg=="], @@ -1034,8 +1025,6 @@ "electron-to-chromium": ["electron-to-chromium@1.5.372", "", {}, "sha512-M3yhbAlilnwqC8D21t28UCDGHyitShTmmLRU/H+b74P6Ski16Nb9HONYEaVpMj/pwC7BEo5B95FpjODLCWbtfA=="], - "elkjs": ["elkjs@0.11.1", "", {}, "sha512-zxxR9k+rx5ktMwT/FwyLdPCrq7xN6e4VGGHH8hA01vVYKjTFik7nHOxBnAYtrgYUB1RpAiLvA1/U2YraWxyKKg=="], - "emnapi": ["emnapi@1.11.1", "", { "peerDependencies": { "node-addon-api": ">= 6.1.0" }, "optionalPeers": ["node-addon-api"] }, "sha512-kSRjhIcxjMFsBqk7ORvoc9aA5SBKDmecrtF5RMcmOTao0kD/zamaxsuTxMI8C1//wGUuvE7a+19pCE7AEhGVnA=="], "emoji-regex": ["emoji-regex@8.0.0", "", {}, "sha512-MSjYzcWNOA0ewAHpz0MxpYFvwg6yjy1NG3xteoqz644VCo/RPgnr1/GGt+ic3iJTzQ8Eu3TdM14SawnVUmGE6A=="], @@ -1062,8 +1051,6 @@ "eventemitter3": ["eventemitter3@5.0.4", "", {}, "sha512-mlsTRyGaPBjPedk6Bvw+aqbsXDtoAyAzm5MO7JgU+yVRyMQ5O8bD4Kcci7BS85f93veegeCPkL8R4GLClnjLFw=="], - "exifr": ["exifr@7.1.3", "", {}, "sha512-g/aje2noHivrRSLbAUtBPWFbxKdKhgj/xr1vATDdUXPOFYJlQ62Ft0oy+72V6XLIpDJfHs6gXLbBLAolqOXYRw=="], - "fast-string-truncated-width": ["fast-string-truncated-width@3.0.3", "", {}, "sha512-0jjjIEL6+0jag3l2XWWizO64/aZVtpiGE3t0Zgqxv0DPuxiMjvB3M24fCyhZUO4KomJQPj3LTSUnDP3GpdwC0g=="], "fast-string-width": ["fast-string-width@3.0.2", "", { "dependencies": { "fast-string-truncated-width": "^3.0.2" } }, "sha512-gX8LrtNEI5hq8DVUfRQMbr5lpaS4nMIWV+7XEbXk2b8kiQIizgnlr12B4dA3ZEx3308ze0O4Q1R+cHts8kyUJg=="], @@ -1084,8 +1071,6 @@ "file-stream-rotator": ["file-stream-rotator@0.6.1", "", { "dependencies": { "moment": "^2.29.1" } }, "sha512-u+dBid4PvZw17PmDeRcNOtCP9CCK/9lRN2w+r1xIS7yOL9JFrIBKTvrYsxT4P0pGtThYTn++QS5ChHaUov3+zQ=="], - "file-type": ["file-type@21.3.4", "", { "dependencies": { "@tokenizer/inflate": "^0.4.1", "strtok3": "^10.3.4", "token-types": "^6.1.1", "uint8array-extras": "^1.4.0" } }, "sha512-Ievi/yy8DS3ygGvT47PjSfdFoX+2isQueoYP1cntFW1JLYAuS4GD7NUPGg4zv2iZfV52uDyk5w5Z0TdpRS6Q1g=="], - "flatbuffers": ["flatbuffers@25.9.23", "", {}, "sha512-MI1qs7Lo4Syw0EOzUl0xjs2lsoeqFku44KpngfIduHBYvzm8h2+7K8YMQh1JtVVVrUvhLpNwqVi4DERegUJhPQ=="], "fn.name": ["fn.name@1.1.0", "", {}, "sha512-GRnmB5gPyJpAhTQdSZTSp9uaPSvl09KoYcMQtsB9rQoOmzs9dH6ffeccH+Z+cv6P68Hu5bC6JjRh4Ah/mHSNRw=="], @@ -1126,8 +1111,6 @@ "iconv-lite": ["iconv-lite@0.7.2", "", { "dependencies": { "safer-buffer": ">= 2.1.2 < 3.0.0" } }, "sha512-im9DjEDQ55s9fL4EYzOAv0yMqmMBSZp6G0VvFyTMPKWxiSBHUj9NW/qqLmXUwXrrM7AvqSlTCfvqRb0cM8yYqw=="], - "ieee754": ["ieee754@1.2.1", "", {}, "sha512-dcyqhDvX1C46lXZcVqCpK+FtMRQVdIMN6/Df5js2zouUsqG7I6sFxitIC+7KYK29KdXOLHdu9zL4sFnoVQnqaA=="], - "immediate": ["immediate@3.0.6", "", {}, "sha512-XXOFtyqDjNDAQxVfYxuF7g9Il/IbWmmlQg2MYKOH8ExIT1qg6xc4zyS3HaEEATgs1btfzxq15ciUiY7gjSXRGQ=="], "inherits": ["inherits@2.0.4", "", {}, "sha512-k/vGaX4/Yla3WzyMCvTQOXYeIHvqOKtnqBduzTHpzpQZzAskKMhZ2K+EnBiSM9zGSoIFeMpXKxa4dYeZIQqewQ=="], @@ -1210,12 +1193,8 @@ "marked": ["marked@18.0.5", "", { "bin": { "marked": "bin/marked.js" } }, "sha512-S6GcvALHg6K4ohtu4E7x0a1AqhAjp6cV8KhLSyN9qVapnzJkusVBxZRcIU9AeYsbe6P1hKDusSbEOzGyyuce6w=="], - "markit-ai": ["markit-ai@0.5.3", "", { "dependencies": { "chalk": "^5.6.2", "commander": "^14.0.3", "exifr": "^7.1.3", "fast-xml-parser": "^5.5.9", "jszip": "^3.10.1", "mammoth": "^1.9.0", "mupdf": "^1.27.0", "music-metadata": "^11.12.3", "rss-parser": "^3.13.0", "turndown": "^7.2.0", "turndown-plugin-gfm": "^1.0.2" }, "bin": { "markit": "dist/main.js" } }, "sha512-h4nhn6a/SNXEdc3kLVtL37TspxjUNCNL0OM7LRWxd389ZByI/B7bjNNgxFdVAT0O+H7ZekSwLdVe/lws1l2AZQ=="], - "matcher": ["matcher@4.0.0", "", { "dependencies": { "escape-string-regexp": "^4.0.0" } }, "sha512-S6x5wmcDmsDRRU/c2dkccDwQPXoFczc5+HpQ2lON8pnvHlnvHAHj5WlLVvw6n6vNyHuVugYrFohYxbS+pvFpKQ=="], - "media-typer": ["media-typer@2.0.0", "", {}, "sha512-kOy3OxT2HH39N70UnKgu4NWDZjLOz8W/mfyvniHjRH/DrL3f2pOfvWQ4p60offbbtDAnXWp0v9LfMIqMec269Q=="], - "merge-anything": ["merge-anything@5.1.7", "", { "dependencies": { "is-what": "^4.1.8" } }, "sha512-eRtbOb1N5iyH0tkQDAoQ4Ipsp/5qSR79Dzrz8hEPxRX10RWWR/iQXdoKmBSRCThY1Fh5EhISDtpSc93fpxUniQ=="], "mimic-function": ["mimic-function@5.0.1", "", {}, "sha512-VP79XUPxV2CigYP3jWwAUFSku2aKqBH7uTAapFWCBqutsbmDo96KY5o8uh6U+/YSIn5OxJnXp73beVkpqMIGhA=="], @@ -1240,8 +1219,6 @@ "mupdf": ["mupdf@1.27.0", "", {}, "sha512-vEPUYwZeu5NgiFLz4e20R7Vp2pNY7szirGEvTxHyQQpQs6ab4DeGdonwT6sH1JZG5EhyHSrojZrZn2/0ee6qZQ=="], - "music-metadata": ["music-metadata@11.13.0", "", { "dependencies": { "@borewit/text-codec": "^0.2.2", "@tokenizer/token": "^0.3.0", "content-type": "^2.0.0", "debug": "^4.4.3", "file-type": "^21.3.4", "media-typer": "^2.0.0", "strtok3": "^10.3.5", "token-types": "^6.1.2", "uint8array-extras": "^1.5.0", "win-guid": "^0.2.1" } }, "sha512-uXRaov9dfjSpQufXIU7sMxVZnh+FilCQv2mXn+K5EJ/decP3dTWrgvPYa5r6MtRbieNSCE708Da4J0u1UGfQIw=="], - "mute-stream": ["mute-stream@3.0.0", "", {}, "sha512-dkEJPVvun4FryqBmZ5KhDo0K9iDXAwn08tMLDinNdRBNPcYEDiWYysLcc6k3mjTMlbP9KyylvRpd4wFtwrT9rw=="], "nanoid": ["nanoid@3.3.12", "", { "bin": { "nanoid": "bin/nanoid.cjs" } }, "sha512-ZB9RH/39qpq5Vu6Y+NmUaFhQR6pp+M2Xt76XBnEwDaGcVAqhlvxrl3B2bKS5D3NH3QR76v3aSrKaF/Kiy7lEtQ=="], @@ -1322,16 +1299,12 @@ "rolldown": ["rolldown@1.0.3", "", { "dependencies": { "@oxc-project/types": "=0.133.0", "@rolldown/pluginutils": "^1.0.0" }, "optionalDependencies": { "@rolldown/binding-android-arm64": "1.0.3", "@rolldown/binding-darwin-arm64": "1.0.3", "@rolldown/binding-darwin-x64": "1.0.3", "@rolldown/binding-freebsd-x64": "1.0.3", "@rolldown/binding-linux-arm-gnueabihf": "1.0.3", "@rolldown/binding-linux-arm64-gnu": "1.0.3", "@rolldown/binding-linux-arm64-musl": "1.0.3", "@rolldown/binding-linux-ppc64-gnu": "1.0.3", "@rolldown/binding-linux-s390x-gnu": "1.0.3", "@rolldown/binding-linux-x64-gnu": "1.0.3", "@rolldown/binding-linux-x64-musl": "1.0.3", "@rolldown/binding-openharmony-arm64": "1.0.3", "@rolldown/binding-wasm32-wasi": "1.0.3", "@rolldown/binding-win32-arm64-msvc": "1.0.3", "@rolldown/binding-win32-x64-msvc": "1.0.3" }, "bin": { "rolldown": "./bin/cli.mjs" } }, "sha512-i00lAJ2ks1BYr7rjNjKC7BcqAS7nVfiT3QX1SI5aY+AFHblCmaUf9OE9dbdzDvW6dJxbi2ZCZiy9v3CcwOiX3g=="], - "rss-parser": ["rss-parser@3.13.0", "", { "dependencies": { "entities": "^2.0.3", "xml2js": "^0.5.0" } }, "sha512-7jWUBV5yGN3rqMMj7CZufl/291QAhvrrGpDNE4k/02ZchL0npisiYYqULF71jCEKoIiHvK/Q2e6IkDwPziT7+w=="], - "safe-buffer": ["safe-buffer@5.1.2", "", {}, "sha512-Gd2UZBJDkXlY7GbJxfsE8/nvKkUEU1G38c1siN6QP6a9PT9MmHB8GnpscSmMJSoF8LOIrt8ud/wPtojys4G6+g=="], "safe-stable-stringify": ["safe-stable-stringify@2.5.0", "", {}, "sha512-b3rppTKm9T+PsVCBEOUR46GWI7fdOs00VKZ1+9c1EWDaDMvjQc6tUwuFyIprgGgTcWoVHSKrU8H31ZHA2e0RHA=="], "safer-buffer": ["safer-buffer@2.1.2", "", {}, "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg=="], - "sax": ["sax@1.6.0", "", {}, "sha512-6R3J5M4AcbtLUdZmRv2SygeVaM7IhrLXu9BmnOGmmACak8fiUtOsYNWUS4uK7upbmHIBbLBeFeI//477BKLBzA=="], - "scheduler": ["scheduler@0.27.0", "", {}, "sha512-eNv+WrVbKu1f3vbYJT/xtiF5syA5HPIMtf9IgY/nKg0sWqzAUEvqY/xm7OcZc/qafLx/iO9FgOmeSAp4v5ti/Q=="], "semver": ["semver@7.8.4", "", { "bin": { "semver": "bin/semver.js" } }, "sha512-rUCObTnP32Q08R2uuIrt7r9PlEonuTmtuXYcW6s5kjdlj3xbnwe+21yXptAUYcMAABLkYYTtnmzb3w3EDZfueA=="], @@ -1390,8 +1363,6 @@ "strnum": ["strnum@2.4.0", "", { "dependencies": { "anynum": "^1.0.0" } }, "sha512-sHrVyWWdq28RbhjuJdZsA1SnGRJV6NiXbk6AXBxDOsgAcA+lmpUZCYjOdLBxkXMwis6RRe7dlZt4VlIWFVzkmg=="], - "strtok3": ["strtok3@10.3.5", "", { "dependencies": { "@tokenizer/token": "^0.3.0" } }, "sha512-ki4hZQfh5rX0QDLLkOCj+h+CVNkqmp/CMf8v8kZpkNVK6jGQooMytqzLZYUVYIZcFZ6yDB70EfD8POcFXiF5oA=="], - "tailwindcss": ["tailwindcss@4.3.1", "", {}, "sha512-hk+TB1m+K8CYNrP6rjQaq/Y+4Zylwpa87mLYBKCunwnnQ9p+fHb7kmSfGqyEJoxF/O6CDyABWVFEafNSYKll+Q=="], "tapable": ["tapable@2.3.3", "", {}, "sha512-uxc/zpqFg6x7C8vOE7lh6Lbda8eEL9zmVm/PLeTPBRhh1xCgdWaQ+J1CUieGpIfm2HdtsUpRv+HshiasBMcc6A=="], @@ -1404,8 +1375,6 @@ "tinyglobby": ["tinyglobby@0.2.17", "", { "dependencies": { "fdir": "^6.5.0", "picomatch": "^4.0.4" } }, "sha512-wXR/dYpcqKmfWpEdZjiKJOwCNFndD0DMnrW/cYjVGttEkBfVgcLFHoNrlj47mjOVic9yyNu65alsgF4NQyTa2g=="], - "token-types": ["token-types@6.1.2", "", { "dependencies": { "@borewit/text-codec": "^0.2.1", "@tokenizer/token": "^0.3.0", "ieee754": "^1.2.1" } }, "sha512-dRXchy+C0IgK8WPC6xvCHFRIWYUbqqdEIKPaKo/AcTUNzwLTK6AH7RjdLWsEZcAN/TBdtfUw3PYEgPr5VPr6ww=="], - "triple-beam": ["triple-beam@1.4.1", "", {}, "sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg=="], "ts-morph": ["ts-morph@28.0.0", "", { "dependencies": { "@ts-morph/common": "~0.29.0", "code-block-writer": "^13.0.3" } }, "sha512-Wp3tnZ2bzwxyTZMtgWVzXDfm7lB1Drz+y9DmmYH/L702PQhPyVrp3pkou3yIz4qjS14GY9kcpmLiOOMvl8oG1g=="], @@ -1428,8 +1397,6 @@ "uhyphen": ["uhyphen@0.2.0", "", {}, "sha512-qz3o9CHXmJJPGBdqzab7qAYuW8kQGKNEuoHFYrBwV6hWIMcpAmxDLXojcHfFr9US1Pe6zUswEIJIbLI610fuqA=="], - "uint8array-extras": ["uint8array-extras@1.5.0", "", {}, "sha512-rvKSBiC5zqCCiDZ9kAOszZcDvdAHwwIKJG33Ykj43OKcWsnmcBRL09YTU4nOeHZ8Y2a7l1MgTd08SBe9A8Qj6A=="], - "underscore": ["underscore@1.13.8", "", {}, "sha512-DXtD3ZtEQzc7M8m4cXotyHR+FAS18C64asBYY5vqZexfYryNNnDc02W4hKg3rdQuqOYas1jkseX0+nZXjTXnvQ=="], "undici-types": ["undici-types@7.24.6", "", {}, "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg=="], @@ -1448,8 +1415,6 @@ "webdriver-bidi-protocol": ["webdriver-bidi-protocol@0.4.2", "", {}, "sha512-VSV+fzfChirL3e7jay2yUC7B4HQCGtEWEg/MSSQbK+qWbqeGlRLlXTzPpYr3XGUvbpDHumWZBJxgesg4N7dbtA=="], - "win-guid": ["win-guid@0.2.1", "", {}, "sha512-gEIQU4mkgl2OPeoNrWflcJFJ3Ae2BPd4eCsHHA/XikslkIVms/nHhvnvzIZV7VLmBvtFlDOzLt9rrZT+n6D67A=="], - "winston": ["winston@3.19.0", "", { "dependencies": { "@colors/colors": "^1.6.0", "@dabh/diagnostics": "^2.0.8", "async": "^3.2.3", "is-stream": "^2.0.0", "logform": "^2.7.0", "one-time": "^1.0.0", "readable-stream": "^3.4.0", "safe-stable-stringify": "^2.3.1", "stack-trace": "0.0.x", "triple-beam": "^1.3.0", "winston-transport": "^4.9.0" } }, "sha512-LZNJgPzfKR+/J3cHkxcpHKpKKvGfDZVPS4hfJCc4cCG0CgYzvlD6yE/S3CIL/Yt91ak327YCpiF/0MyeZHEHKA=="], "winston-daily-rotate-file": ["winston-daily-rotate-file@5.0.0", "", { "dependencies": { "file-stream-rotator": "^0.6.1", "object-hash": "^3.0.0", "triple-beam": "^1.4.1", "winston-transport": "^4.7.0" }, "peerDependencies": { "winston": "^3" } }, "sha512-JDjiXXkM5qvwY06733vf09I2wnMXpZEhxEVOSPenZMii+g7pcDcTBt2MRugnoi8BwVSuCT2jfRXBUy+n1Zz/Yw=="], @@ -1464,8 +1429,6 @@ "xml-naming": ["xml-naming@0.1.0", "", {}, "sha512-k8KO9hrMyNk6tUWqUfkTEZbezRRpONVOzUTnc97VnCvyj6Tf9lyUR9EDAIeiVLv56jsMcoXEwjW8Kv5yPY52lw=="], - "xml2js": ["xml2js@0.5.0", "", { "dependencies": { "sax": ">=0.6.0", "xmlbuilder": "~11.0.0" } }, "sha512-drPFnkQJik/O+uPKpqSgr22mpuFHqKdbS835iAQrUC73L2F5WkboIRd63ai/2Yg6I1jzifPFKH2NTK+cfglkIA=="], - "xmlbuilder": ["xmlbuilder@10.1.1", "", {}, "sha512-OyzrcFLL/nb6fMGHbiRDuPup9ljBycsdCypwuyg5AAHvyWzGfChJpCXMG88AGTIMFhGZ9RccFN1e6lhg3hkwKg=="], "y18n": ["y18n@5.0.8", "", {}, "sha512-0pfFzegeDWJHJIAmTLRP2DwHjdF5s7jo9tuztdQxAhINCdvS+3nGINqPd00AphqJR/0LhANUS6/+7SCb98YOfA=="], @@ -1560,8 +1523,6 @@ "roarr/sprintf-js": ["sprintf-js@1.1.3", "", {}, "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA=="], - "rss-parser/entities": ["entities@2.2.0", "", {}, "sha512-p92if5Nz619I0w+akJrLZH0MX0Pb5DX39XOwQTtXSdQQOaYH03S1uIQp4mhOZtAXrxq4ViO67YTiLBo2638o9A=="], - "slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], "string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="], @@ -1570,8 +1531,6 @@ "wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="], - "xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="], - "@babel/helper-compilation-targets/lru-cache/yallist": ["yallist@3.1.1", "", {}, "sha512-a4UGQaWPH59mOXUYnAG2ewncQS4i4F43Tv3JoAM+s2VDAmS9NsK8GpDMLrCHPksFT7h3K6TOoUNn2pb7RoXx4g=="], "@huggingface/transformers/onnxruntime-node/global-agent": ["global-agent@3.0.0", "", { "dependencies": { "boolean": "^3.0.1", "es6-error": "^4.1.1", "matcher": "^3.0.0", "roarr": "^2.15.3", "semver": "^7.3.2", "serialize-error": "^7.0.1" } }, "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q=="], diff --git a/package.json b/package.json index a27a32b4b..f3109d35e 100644 --- a/package.json +++ b/package.json @@ -60,14 +60,16 @@ "diff": "^9.0.0", "fflate": "0.8.3", "fastembed": "2.1.0", + "fast-xml-parser": "^5.9.0", "ghostty-web": "^0.4.0", "handlebars": "^4.7.9", "linkedom": "^0.18.12", "lint-staged": "^17.0.7", "lru-cache": "11.5.1", "lucide-react": "^1.17.0", + "mammoth": "^1.12.0", "marked": "^18.0.5", - "markit-ai": "0.5.3", + "mupdf": "^1.27.0", "onnxruntime-node": "1.26.0", "partial-json": "^0.1.7", "postcss": "^8.5.15", @@ -169,7 +171,7 @@ "lint:py": "ruff check python && ruff format --check python", "fix:py": "ruff check --fix python && ruff format python", "prepublishOnly": "bun run check", - "prepare": "bun run generate-docs-index && bun run build-tool-views", + "prepare": "bun run build-tool-views", "publish": "bun run prepublishOnly && npm publish -ws --access public", "publish:dry": "bun run prepublishOnly && npm publish -ws --access public --dry-run", "release": "bun scripts/release.ts", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index b0af27b67..66fd5ffdf 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,14 +1,23 @@ # Changelog ## [Unreleased] + ### Changed +- Updated internal image processing to no longer include metadata text for fetched images - Optimized `omp://` documentation indexing by compressing doc bodies into a lazily-inflated blob - Changed Mermaid fenced-block ASCII rendering to use the first-party vendored renderer in `@oh-my-pi/pi-utils` (`src/vendor/mermaid-ascii`), dropping the `beautiful-mermaid` npm package, its transitive `elkjs` (~3.13MB), and the `beautiful-mermaid` `bun patch`; CJK/emoji width handling and the layout-direction override are preserved. - Changed `omp://` documentation embedding to a gzipped base64 index (`docs-index.generated.txt`, populated at build time and reset afterward) inflated on first read, instead of a ~1.6MB raw TypeScript map. The compiled binary / npm bundle drops ~0.9MB; the dev tree and source checkouts read `docs/` from disk. +- Replaced the `markit-ai` package with a vendored in-house document engine (`src/markit`) for the document formats it converts: PDF (via `mupdf`), DOCX (via `mammoth`), and PPTX/XLSX/EPUB (via `fast-xml-parser`). Conversion output is preserved, including the PDF column/table-detection pipeline and HTML-table normalization; `markit-ai`'s unused converters and their dependency tail are gone. Legacy `.doc`/`.ppt`/`.xls`/`.rtf` remain unsupported (a conversion error), as before. +- Centralized ZIP handling behind a single `src/utils/zip.ts` (`fflate`): the new document converters, the `write` tool's in-place archive editing, and the `read` tool's ranged archive reader now share one ZIP implementation instead of mixing `jszip` and `fflate`. + +### Removed + +- Removed the `markit-ai`, `exifr`, `music-metadata`, and direct `jszip` dependencies. As a side effect, fetching an image or audio URL no longer appends EXIF/audio metadata text; image inlining and resizing are unchanged, and document conversion (PDF/DOCX/PPTX/XLSX/EPUB) is unaffected. ### Fixed +- Fixed auto-retry after transient model-stream socket closes to replay text/thinking-only partial assistant output, including turns where incomplete tool-call arguments were dropped; completed tool calls still block retry to avoid duplicating tool execution. - Fixed `omp update` reporting `EPERM: operation not permitted, unlink '.bak'` on Windows when self-replacing a standalone binary, even though the new binary had already been installed. The backed-up old executable is still the running process image and cannot be unlinked until the process exits, so the post-verify backup cleanup is now best-effort, backups use a unique per-attempt name, and stale backups are swept on the next update ([#845](https://github.com/can1357/oh-my-pi/issues/845)). - Fixed plan-mode `Refine plan` so the internal approval abort is hidden and the editor is ready for a follow-up prompt instead of showing `Operation aborted` ([#2971](https://github.com/can1357/oh-my-pi/issues/2971)). - Fixed TUI prompts beginning with shell-style variables such as `$HOME` being misrouted to Python eval; Python shortcuts now require `$ ` or `$$ `. ([#2944](https://github.com/can1357/oh-my-pi/issues/2944)) diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index 926667355..71c1787e9 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -35,7 +35,7 @@ "check": "biome check . && bun run check:types", "check:types": "tsgo -p tsconfig.json --noEmit", "lint": "biome lint .", - "test": "bun test --parallel=4", + "test": "bun test --parallel=4 test src", "fix": "biome check --write --unsafe . && bun run format-prompts", "fmt": "biome format --write . && bun run format-prompts", "format-prompts": "bun scripts/format-prompts.ts", @@ -71,11 +71,13 @@ "arktype": "catalog:", "chalk": "catalog:", "diff": "catalog:", + "fast-xml-parser": "catalog:", "fflate": "catalog:", "handlebars": "catalog:", "linkedom": "catalog:", "lru-cache": "catalog:", - "markit-ai": "catalog:", + "mammoth": "catalog:", + "mupdf": "catalog:", "puppeteer-core": "catalog:", "turndown": "catalog:", "turndown-plugin-gfm": "catalog:", diff --git a/packages/coding-agent/src/markit/NOTICE b/packages/coding-agent/src/markit/NOTICE new file mode 100644 index 000000000..09c69d6aa --- /dev/null +++ b/packages/coding-agent/src/markit/NOTICE @@ -0,0 +1,32 @@ +This directory contains an in-house document-to-markdown engine adapted from +markit-ai (https://github.com/Michaelliv/markit), used under the MIT License. + + Copyright (c) 2026 Michael Liv + +Only the converters for the document formats omp supports are ported (pdf, +docx, pptx, xlsx, epub); the CLI, plugin/provider, and unused converters +(html, image, audio, plain-text, rss, github, wikipedia, csv, json, yaml, +ipynb, iwork, zip, xml) were dropped. Legacy binary `.doc`/`.ppt`/`.xls` and +`.rtf` are routed by the read/fetch tools but have no converter — they surface +a conversion error, exactly as upstream markit did. Logic is ported faithfully +so conversion output matches the upstream package. + +MIT License + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/packages/coding-agent/src/markit/converters/docx.ts b/packages/coding-agent/src/markit/converters/docx.ts new file mode 100644 index 000000000..5dfe643fb --- /dev/null +++ b/packages/coding-agent/src/markit/converters/docx.ts @@ -0,0 +1,56 @@ +// Adapted from markit-ai (MIT). See ../NOTICE. +import * as path from "node:path"; +import mammoth from "mammoth"; +import { createTurndown, normalizeTablesHtml } from "../../utils/turndown"; +import type { ConversionResult, Converter, StreamInfo } from "../types"; + +const EXTENSIONS = [".docx"]; +const MIMETYPES = ["application/vnd.openxmlformats-officedocument.wordprocessingml.document"]; + +export class DocxConverter implements Converter { + name = "docx"; + + accepts(streamInfo: StreamInfo): boolean { + if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) { + return true; + } + if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) { + return true; + } + return false; + } + + async convert(input: Buffer, streamInfo: StreamInfo): Promise { + const imageDir = streamInfo.imageDir; + let imageCount = 0; + const convertImage = imageDir + ? mammoth.images.imgElement(image => { + imageCount++; + const ext = (image.contentType?.split("/")[1] || "png").replace("jpeg", "jpg"); + const filename = `image_${imageCount}.${ext}`; + const filepath = path.join(imageDir, filename); + return image.read("base64").then(async base64 => { + await Bun.write(filepath, Buffer.from(base64, "base64")); + return { src: filepath, alt: `image_${imageCount}` }; + }); + }) + : mammoth.images.imgElement(image => { + imageCount++; + const contentType = image.contentType || "image/png"; + return image.read("base64").then(base64 => { + return { + src: `data:${contentType};base64,${base64.slice(0, 0)}`, + alt: `image_${imageCount}`, + }; + }); + }); + const { value: html } = await mammoth.convertToHtml({ buffer: input }, { convertImage }); + const turndown = createTurndown(); + let markdown = turndown.turndown(normalizeTablesHtml(html)); + // Replace data URI images with comment placeholders when no imageDir + if (!imageDir) { + markdown = markdown.replace(/!\[([^\]]*)\]\(data:[^)]*\)/g, ""); + } + return { markdown: markdown.trim() }; + } +} diff --git a/packages/coding-agent/src/markit/converters/epub.ts b/packages/coding-agent/src/markit/converters/epub.ts new file mode 100644 index 000000000..dda23c0e9 --- /dev/null +++ b/packages/coding-agent/src/markit/converters/epub.ts @@ -0,0 +1,136 @@ +// Adapted from markit-ai (MIT). See ../NOTICE. +import { XMLParser } from "fast-xml-parser"; +import { createTurndown, normalizeTablesHtml } from "../../utils/turndown"; +import { unzip, unzipText } from "../../utils/zip"; +import type { ConversionResult, Converter, StreamInfo } from "../types"; + +const EXTENSIONS = [".epub"]; +const MIMETYPES = ["application/epub", "application/epub+zip", "application/x-epub+zip"]; + +/** A metadata value: a bare string, or a node carrying `#text` and/or array children. */ +type MetaValue = string | MetaNode; +interface MetaNode { + "#text"?: string; + [index: number]: MetaValue; +} +interface Metadata { + "dc:title"?: MetaValue; + "dc:creator"?: MetaValue; + "dc:language"?: MetaValue; + "dc:publisher"?: MetaValue; + "dc:date"?: MetaValue; + "dc:description"?: MetaValue; +} +interface ManifestItem { + "@_id": string; + "@_href": string; +} +interface SpineItem { + "@_idref": string; +} +interface OpfDoc { + package?: { + metadata?: Metadata; + manifest?: { item?: ManifestItem | ManifestItem[] }; + spine?: { itemref?: SpineItem | SpineItem[] }; + }; +} +interface Rootfile { + "@_full-path": string; +} +interface ContainerDoc { + container?: { rootfiles?: { rootfile?: Rootfile | Rootfile[] } }; +} + +export class EpubConverter implements Converter { + name = "epub"; + + accepts(streamInfo: StreamInfo): boolean { + if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true; + if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true; + return false; + } + + async convert(input: Buffer, _streamInfo: StreamInfo): Promise { + const entries = unzip(input); + const parser = new XMLParser({ + ignoreAttributes: false, + attributeNamePrefix: "@_", + textNodeName: "#text", + processEntities: { maxTotalExpansions: 1_000_000 }, + }); + // Find content.opf path from container.xml + const containerXml = unzipText(entries, "META-INF/container.xml"); + if (!containerXml) throw new Error("Invalid EPUB: missing container.xml"); + const container = parser.parse(containerXml) as ContainerDoc; + const rootfile = container.container?.rootfiles?.rootfile; + const opfPath = Array.isArray(rootfile) ? rootfile[0]["@_full-path"] : rootfile?.["@_full-path"]; + if (!opfPath) throw new Error("Invalid EPUB: missing rootfile path"); + // Parse content.opf + const opfXml = unzipText(entries, opfPath); + if (!opfXml) throw new Error("Invalid EPUB: missing content.opf"); + const opf = parser.parse(opfXml) as OpfDoc; + // Extract metadata + const meta: Metadata = opf.package?.metadata ?? {}; + const metadata: Record = { + title: this.getText(meta["dc:title"]), + authors: this.getTextArray(meta["dc:creator"]).join(", ") || undefined, + language: this.getText(meta["dc:language"]), + publisher: this.getText(meta["dc:publisher"]), + date: this.getText(meta["dc:date"]), + description: this.getText(meta["dc:description"]), + }; + // Build manifest map (id → href) + const manifestItems = opf.package?.manifest?.item; + const itemList = Array.isArray(manifestItems) ? manifestItems : manifestItems ? [manifestItems] : []; + const manifest = new Map(); + for (const item of itemList) { + manifest.set(item["@_id"], item["@_href"]); + } + // Get spine order + const spineItems = opf.package?.spine?.itemref; + const spineList = Array.isArray(spineItems) ? spineItems : spineItems ? [spineItems] : []; + const spineOrder = spineList.map(s => s["@_idref"]); + // Resolve file paths + const basePath = opfPath.includes("/") ? opfPath.substring(0, opfPath.lastIndexOf("/")) : ""; + const turndown = createTurndown(); + const sections: string[] = []; + // Add metadata header + const metaLines: string[] = []; + for (const key in metadata) { + const value = metadata[key]; + if (value) metaLines.push(`**${key.charAt(0).toUpperCase() + key.slice(1)}:** ${value}`); + } + if (metaLines.length > 0) sections.push(metaLines.join("\n")); + // Convert spine files + for (const idref of spineOrder) { + const href = manifest.get(idref); + if (!href) continue; + const filePath = basePath ? `${basePath}/${href}` : href; + const html = unzipText(entries, filePath); + if (!html) continue; + // Strip script/style, convert to markdown + const cleaned = html.replace(//gi, "").replace(//gi, ""); + const md = turndown.turndown(normalizeTablesHtml(cleaned)).trim(); + if (md) sections.push(md); + } + return { + markdown: sections.join("\n\n").trim(), + title: metadata.title, + }; + } + + getText(node: MetaValue | undefined): string | undefined { + if (!node) return undefined; + if (typeof node === "string") return node; + if (node["#text"]) return String(node["#text"]); + if (Array.isArray(node)) return this.getText(node[0]); + return undefined; + } + + getTextArray(node: MetaValue | undefined): (string | undefined)[] { + if (!node) return []; + const list = Array.isArray(node) ? node : [node]; + return list.map(n => this.getText(n)).filter(Boolean); + } +} diff --git a/packages/coding-agent/src/markit/converters/mammoth.d.ts b/packages/coding-agent/src/markit/converters/mammoth.d.ts new file mode 100644 index 000000000..012333658 --- /dev/null +++ b/packages/coding-agent/src/markit/converters/mammoth.d.ts @@ -0,0 +1,24 @@ +// Minimal ambient types for `mammoth` (ships no types). Declares only what +// DocxConverter uses. See ../NOTICE. +declare module "mammoth" { + interface MammothImage { + contentType?: string; + read(encoding: "base64"): Promise; + } + interface ImgAttributes { + src: string; + alt?: string; + } + type ConvertImageHandler = (image: MammothImage) => Promise; + interface ConvertOptions { + convertImage?: ConvertImageHandler; + } + interface ConvertResult { + value: string; + messages: unknown[]; + } + export const images: { imgElement(fn: ConvertImageHandler): ConvertImageHandler }; + export function convertToHtml(input: { buffer: Buffer }, options?: ConvertOptions): Promise; + const _default: { convertToHtml: typeof convertToHtml; images: typeof images }; + export default _default; +} diff --git a/packages/coding-agent/src/markit/converters/pdf/columns.ts b/packages/coding-agent/src/markit/converters/pdf/columns.ts new file mode 100644 index 000000000..ab3c83bcb --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pdf/columns.ts @@ -0,0 +1,103 @@ +// Adapted from markit-ai (MIT). See ../../NOTICE. + +/** + * Multi-column layout detection and text box reordering. + * + * Many PDFs (legal documents, datasheets, academic papers) use two-column + * layouts. Without column detection, text boxes are ordered by Y position + * only, interleaving left and right column content. + * + * Algorithm: + * 1. Collect left edges of all text boxes on the page + * 2. Find the largest horizontal gap between consecutive left edges + * 3. If gap > MIN_GAP_RATIO of the text width and both sides have + * enough boxes → multi-column detected + * 4. Assign each text box to a column based on its center X + * 5. Return columns in reading order (left-to-right, top-to-bottom) + * + * This only detects the column structure. The caller is responsible for + * processing each column's text boxes independently (table detection, + * rendering, etc.). + */ +import type { TextBox } from "./types"; + +export interface ColumnLayout { + /** Number of columns detected (1 = single column, 2+ = multi-column). */ + columnCount: number; + /** Text boxes grouped by column, in reading order (left to right). */ + columns: TextBox[][]; + /** X positions of column boundaries (between columns). */ + boundaries: number[]; +} + +/** + * Minimum gap as a fraction of the total text width to consider a column + * boundary. A two-column layout typically has ~50% gap; we use a lower + * threshold to catch asymmetric columns. + */ +const MIN_GAP_RATIO = 0.15; +/** Minimum number of text boxes on each side of the gap. */ +const MIN_BOXES_PER_COLUMN = 4; +/** Minimum gap in absolute points to avoid splitting on small whitespace. */ +const MIN_GAP_PTS = 40; + +/** + * Detect column layout and return text boxes grouped by column. + * + * For single-column pages, returns all boxes in one group. + * For multi-column pages, returns boxes split by column in reading order. + */ +export function detectColumns(textBoxes: TextBox[]): ColumnLayout { + if (textBoxes.length < MIN_BOXES_PER_COLUMN * 2) { + return { columnCount: 1, columns: [textBoxes], boundaries: [] }; + } + // Collect unique left edges (rounded to avoid float noise) + const lefts = [...new Set(textBoxes.map(tb => Math.round(tb.bounds.left)))].sort((a, b) => a - b); + if (lefts.length < 2) { + return { columnCount: 1, columns: [textBoxes], boundaries: [] }; + } + const textXMin = lefts[0]; + const textXMax = Math.max(...textBoxes.map(tb => Math.round(tb.bounds.right))); + const textWidth = textXMax - textXMin; + if (textWidth <= 0) { + return { columnCount: 1, columns: [textBoxes], boundaries: [] }; + } + // Find the largest gap between consecutive left-edge positions + let maxGap = 0; + let gapLeft = 0; + let gapRight = 0; + for (let i = 1; i < lefts.length; i++) { + const gap = lefts[i] - lefts[i - 1]; + if (gap > maxGap) { + maxGap = gap; + gapLeft = lefts[i - 1]; + gapRight = lefts[i]; + } + } + const gapRatio = maxGap / textWidth; + if (gapRatio < MIN_GAP_RATIO || maxGap < MIN_GAP_PTS) { + return { columnCount: 1, columns: [textBoxes], boundaries: [] }; + } + // Split point is the midpoint of the gap + const splitX = (gapLeft + gapRight) / 2; + // Assign boxes to columns based on center X + const leftCol: TextBox[] = []; + const rightCol: TextBox[] = []; + for (const tb of textBoxes) { + const cx = (tb.bounds.left + tb.bounds.right) / 2; + if (cx < splitX) { + leftCol.push(tb); + } else { + rightCol.push(tb); + } + } + // Validate both columns have enough content + if (leftCol.length < MIN_BOXES_PER_COLUMN || rightCol.length < MIN_BOXES_PER_COLUMN) { + return { columnCount: 1, columns: [textBoxes], boundaries: [] }; + } + return { + columnCount: 2, + columns: [leftCol, rightCol], + boundaries: [splitX], + }; +} diff --git a/packages/coding-agent/src/markit/converters/pdf/extract.ts b/packages/coding-agent/src/markit/converters/pdf/extract.ts new file mode 100644 index 000000000..f27b7947c --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pdf/extract.ts @@ -0,0 +1,557 @@ +// Adapted from markit-ai (MIT). See ../../NOTICE. + +/** + * PDF content extraction using mupdf. + * + * Extracts text boxes (with position, font size, bold) and vector line + * segments (table borders) from each page. Uses mupdf's native WASM + * engine for fast parsing, and reads raw content streams for vector graphics. + * + * Coordinate system: PDF native (origin = bottom-left, Y increases upward). + */ +import * as mupdf from "mupdf"; +import type { ImageRegion, PageContent, Segment, TextBox } from "./types"; + +/** mupdf structured-text JSON bounding box (top-left origin). */ +interface StextBBox { + x: number; + y: number; + w: number; + h: number; +} + +/** Font metadata attached to a structured-text line. */ +interface StextFont { + size?: number; + weight?: string; + name?: string; +} + +/** A line within a text block in mupdf structured-text JSON. */ +interface StextLine { + text?: string; + font?: StextFont; + bbox: StextBBox; +} + +/** A block (text or image) in mupdf structured-text JSON. */ +interface StextBlock { + type: string; + bbox: StextBBox; + lines: StextLine[]; +} + +/** Parsed mupdf structured-text JSON for a page. */ +interface StructuredTextJSON { + blocks: StextBlock[]; +} + +/** A raw text fragment before merging into word/phrase boxes. */ +interface RawTextItem { + text: string; + x: number; + y: number; + width: number; + height: number; + fontSize: number; + isBold: boolean; +} + +// --------------------------------------------------------------------------- +// Text extraction +// --------------------------------------------------------------------------- +/** Y tolerance for merging text fragments on the same visual line. */ +const SAME_LINE_Y_TOLERANCE = 2; +/** Max horizontal gap (pts) to merge adjacent fragments into one text box. */ +const MAX_MERGE_GAP = 14; + +/** + * Merge horizontally adjacent raw text items on the same visual line into + * word/phrase-level text boxes. + */ +function mergeIntoWords(raws: RawTextItem[]): RawTextItem[] { + if (raws.length === 0) return []; + // Sort by Y descending (top-first in bottom-left coords), then X ascending + const sorted = [...raws].sort((a, b) => { + const dy = b.y - a.y; + return Math.abs(dy) > SAME_LINE_Y_TOLERANCE ? dy : a.x - b.x; + }); + const merged: RawTextItem[] = []; + let cur = { ...sorted[0] }; + for (let i = 1; i < sorted.length; i++) { + const next = sorted[i]; + const sameY = Math.abs(next.y - cur.y) <= SAME_LINE_Y_TOLERANCE; + const close = next.x <= cur.x + cur.width + MAX_MERGE_GAP; + if (sameY && close) { + const gap = next.x - (cur.x + cur.width); + const sep = gap > 1 ? " " : ""; + cur.text += sep + next.text; + cur.width = next.x + next.width - cur.x; + cur.height = Math.max(cur.height, next.height); + cur.fontSize = Math.max(cur.fontSize, next.fontSize); + cur.isBold = cur.isBold || next.isBold; + } else { + merged.push(cur); + cur = { ...next }; + } + } + merged.push(cur); + return merged; +} + +/** + * Extract text boxes from a mupdf page using structured text output. + * + * mupdf's structured text JSON uses top-left origin; we convert to + * bottom-left (standard PDF coordinates) using the page height. + */ +function extractTextBoxes( + page: mupdf.Page, + pageNumber: number, + pageHeight: number, + stext?: StructuredTextJSON, +): TextBox[] { + if (!stext) { + stext = JSON.parse(page.toStructuredText("preserve-whitespace").asJSON()) as StructuredTextJSON; + } + const raws: RawTextItem[] = []; + for (const block of stext.blocks) { + if (block.type !== "text") continue; + for (const line of block.lines) { + const text = line.text?.trim(); + if (!text) continue; + const fontSize = line.font?.size ?? 0; + const weight = line.font?.weight ?? "normal"; + const fontName = line.font?.name ?? ""; + const isBold = weight === "bold" || /bold/i.test(fontName) || /Black|Heavy/i.test(fontName); + // mupdf bbox: {x, y, w, h} in top-left coords + // Convert to bottom-left: pdfY = pageHeight - (bbox.y + bbox.h) + const bboxY = line.bbox.y; + const bboxH = line.bbox.h; + const pdfY = pageHeight - (bboxY + bboxH); + raws.push({ + text, + x: line.bbox.x, + y: pdfY, + width: line.bbox.w, + height: bboxH, + fontSize, + isBold, + }); + } + } + const words = mergeIntoWords(raws); + return words + .map((w, i) => ({ + id: `p${pageNumber}-t${i}`, + text: w.text.trim(), + pageNumber, + fontSize: w.fontSize, + isBold: w.isBold, + bounds: { + left: w.x, + right: w.x + w.width, + bottom: w.y, + top: w.y + w.height, + }, + })) + .filter(b => b.text.length > 0); +} + +// --------------------------------------------------------------------------- +// Vector segment extraction from raw content stream +// --------------------------------------------------------------------------- +/** Minimum aspect ratio for a filled rect to be considered a line. */ +const LINE_ASPECT_THRESHOLD = 6; +/** Minimum length (pts) for a segment to count. */ +const MIN_LENGTH = 2; +/** Maximum thickness (pts) for a border line (filters out filled areas). */ +const MAX_THICKNESS = 3; + +/** + * Convert a thin filled rectangle to a horizontal or vertical segment. + * Returns null if the rect doesn't look like a border line. + */ +function thinRectToSegment(id: string, x: number, y: number, w: number, h: number): Segment | null { + const aw = Math.abs(w); + const ah = Math.abs(h); + if (aw > ah * LINE_ASPECT_THRESHOLD && aw >= MIN_LENGTH && ah <= MAX_THICKNESS) { + // Horizontal line + const cy = y + ah / 2; + return { id, x1: x, y1: cy, x2: x + aw, y2: cy }; + } + if (ah > aw * LINE_ASPECT_THRESHOLD && ah >= MIN_LENGTH && aw <= MAX_THICKNESS) { + // Vertical line + const cx = x + aw / 2; + return { id, x1: cx, y1: y, x2: cx, y2: y + ah }; + } + return null; +} + +/** + * Emit 4 edge segments from a stroked rectangle. + */ +function pushStrokedRectEdges(segments: Segment[], id: string, x: number, y: number, w: number, h: number): void { + const aw = Math.abs(w); + const ah = Math.abs(h); + const base = id; + if (aw >= MIN_LENGTH) { + segments.push({ id: `${base}-b`, x1: x, y1: y, x2: x + aw, y2: y }); + segments.push({ + id: `${base}-t`, + x1: x, + y1: y + ah, + x2: x + aw, + y2: y + ah, + }); + } + if (ah >= MIN_LENGTH) { + segments.push({ id: `${base}-l`, x1: x, y1: y, x2: x, y2: y + ah }); + segments.push({ + id: `${base}-r`, + x1: x + aw, + y1: y, + x2: x + aw, + y2: y + ah, + }); + } +} + +const CTM_IDENTITY = [1, 0, 0, 1, 0, 0]; + +/** Concatenate two affine matrices: result = parent × child. */ +function ctmConcat(p: number[], c: number[]): number[] { + return [ + p[0] * c[0] + p[2] * c[1], + p[1] * c[0] + p[3] * c[1], + p[0] * c[2] + p[2] * c[3], + p[1] * c[2] + p[3] * c[3], + p[0] * c[4] + p[2] * c[5] + p[4], + p[1] * c[4] + p[3] * c[5] + p[5], + ]; +} + +function ctmApply(m: number[], x: number, y: number): [number, number] { + return [m[0] * x + m[2] * y + m[4], m[1] * x + m[3] * y + m[5]]; +} + +// --------------------------------------------------------------------------- +// Content stream parsing +// --------------------------------------------------------------------------- +/** + * Parse a PDF content stream and extract line segments from thin filled + * rectangles (re+f), stroked rectangles (re+S), and explicit lines (m/l+S). + * Tracks the CTM via q/Q/cm operators so coordinates are in page space. + */ +function extractSegmentsFromContentStream(raw: string, pageNumber: number): Segment[] { + const segments: Segment[] = []; + const tokens = tokenizeContentStream(raw); + let idx = 0; + let strokeWidth = 1.0; + // Graphics state stack (q/Q): saves CTM + strokeWidth + let ctm = [...CTM_IDENTITY]; + const stateStack: Array<{ ctm: number[]; strokeWidth: number }> = []; + // State for path building (in user coordinates, pre-CTM) + let curX = 0; + let curY = 0; + let pathStartX = 0; + let pathStartY = 0; + const pendingRects: Array<{ x: number; y: number; w: number; h: number }> = []; + const pendingLines: Array<{ x1: number; y1: number; x2: number; y2: number }> = []; + function flushPath(mode: "fill" | "stroke"): void { + const sid = () => `p${pageNumber}-s${segments.length}`; + if (mode === "fill") { + for (const r of pendingRects) { + // Transform the rect corners through CTM, then check if it's a thin line + const [x0, y0] = ctmApply(ctm, r.x, r.y); + const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h); + const seg = thinRectToSegment( + sid(), + Math.min(x0, x1), + Math.min(y0, y1), + Math.abs(x1 - x0), + Math.abs(y1 - y0), + ); + if (seg) segments.push(seg); + } + } else if (mode === "stroke" && strokeWidth <= MAX_THICKNESS) { + for (const r of pendingRects) { + const [x0, y0] = ctmApply(ctm, r.x, r.y); + const [x1, y1] = ctmApply(ctm, r.x + r.w, r.y + r.h); + pushStrokedRectEdges( + segments, + sid(), + Math.min(x0, x1), + Math.min(y0, y1), + Math.abs(x1 - x0), + Math.abs(y1 - y0), + ); + } + for (const l of pendingLines) { + const [lx1, ly1] = ctmApply(ctm, l.x1, l.y1); + const [lx2, ly2] = ctmApply(ctm, l.x2, l.y2); + const dx = Math.abs(lx2 - lx1); + const dy = Math.abs(ly2 - ly1); + // Only keep H/V lines + if ((dx >= MIN_LENGTH && dy < 1) || (dy >= MIN_LENGTH && dx < 1)) { + segments.push({ id: sid(), x1: lx1, y1: ly1, x2: lx2, y2: ly2 }); + } + } + } + pendingRects.length = 0; + pendingLines.length = 0; + } + while (idx < tokens.length) { + const t = tokens[idx]; + if (t === "q") { + stateStack.push({ ctm: [...ctm], strokeWidth }); + } else if (t === "Q") { + const saved = stateStack.pop(); + if (saved) { + ctm = saved.ctm; + strokeWidth = saved.strokeWidth; + } + } else if (t === "cm" && idx >= 6) { + const a = Number(tokens[idx - 6]); + const b = Number(tokens[idx - 5]); + const c = Number(tokens[idx - 4]); + const d = Number(tokens[idx - 3]); + const e = Number(tokens[idx - 2]); + const f = Number(tokens[idx - 1]); + ctm = ctmConcat(ctm, [a, b, c, d, e, f]); + } else if (t === "w" && idx >= 1) { + strokeWidth = Number(tokens[idx - 1]) || strokeWidth; + } else if (t === "re" && idx >= 4) { + const x = Number(tokens[idx - 4]); + const y = Number(tokens[idx - 3]); + const w = Number(tokens[idx - 2]); + const h = Number(tokens[idx - 1]); + if (Number.isFinite(x + y + w + h)) { + pendingRects.push({ x, y, w, h }); + } + } else if (t === "m" && idx >= 2) { + curX = Number(tokens[idx - 2]); + curY = Number(tokens[idx - 1]); + pathStartX = curX; + pathStartY = curY; + } else if (t === "l" && idx >= 2) { + const x2 = Number(tokens[idx - 2]); + const y2 = Number(tokens[idx - 1]); + pendingLines.push({ x1: curX, y1: curY, x2, y2 }); + curX = x2; + curY = y2; + } else if (t === "h") { + // closePath: line back to start + if (curX !== pathStartX || curY !== pathStartY) { + pendingLines.push({ + x1: curX, + y1: curY, + x2: pathStartX, + y2: pathStartY, + }); + } + curX = pathStartX; + curY = pathStartY; + } else if (t === "f" || t === "F" || t === "f*") { + flushPath("fill"); + } else if (t === "S" || t === "s") { + if (t === "s") { + // closeStroke: implicit closePath + if (curX !== pathStartX || curY !== pathStartY) { + pendingLines.push({ + x1: curX, + y1: curY, + x2: pathStartX, + y2: pathStartY, + }); + } + } + flushPath("stroke"); + } else if (t === "B" || t === "B*" || t === "b" || t === "b*") { + // fill + stroke combined + flushPath("fill"); + flushPath("stroke"); + } else if (t === "n") { + // end path without painting — discard + pendingRects.length = 0; + pendingLines.length = 0; + } + idx++; + } + return segments; +} + +/** + * Fast tokenizer for PDF content streams. + * Splits on whitespace, skipping comments and string literals. + */ +function tokenizeContentStream(raw: string): string[] { + const tokens: string[] = []; + const len = raw.length; + let i = 0; + while (i < len) { + const ch = raw.charCodeAt(i); + // Skip whitespace + if (ch <= 32) { + i++; + continue; + } + // Skip comments + if (ch === 37 /* % */) { + while (i < len && raw.charCodeAt(i) !== 10) i++; + continue; + } + // Skip string literals (...) + if (ch === 40 /* ( */) { + let depth = 1; + i++; + while (i < len && depth > 0) { + const c = raw.charCodeAt(i); + if (c === 92 /* \ */) { + i++; + } else if (c === 40) { + depth++; + } else if (c === 41) { + depth--; + } + i++; + } + continue; + } + // Skip hex strings <...> + if (ch === 60 /* < */ && i + 1 < len && raw.charCodeAt(i + 1) !== 60) { + i++; + while (i < len && raw.charCodeAt(i) !== 62) i++; + i++; // skip > + continue; + } + // Skip dict delimiters << >> + if (ch === 60 && i + 1 < len && raw.charCodeAt(i + 1) === 60) { + i += 2; + continue; + } + if (ch === 62 && i + 1 < len && raw.charCodeAt(i + 1) === 62) { + i += 2; + continue; + } + // Regular token: read until whitespace or delimiter + const start = i; + while (i < len) { + const c = raw.charCodeAt(i); + if (c <= 32 || c === 40 || c === 41 || c === 60 || c === 62 || c === 37) break; + i++; + } + if (i > start) { + tokens.push(raw.substring(start, i)); + } + } + return tokens; +} + +// --------------------------------------------------------------------------- +// Image region detection +// --------------------------------------------------------------------------- +/** Minimum area (pts²) for an image to be considered a diagram, not an icon. */ +const MIN_IMAGE_AREA = 5000; + +function extractImageRegions(stext: StructuredTextJSON, pageNumber: number, pageHeight: number): ImageRegion[] { + const regions: ImageRegion[] = []; + for (const block of stext.blocks) { + if (block.type !== "image") continue; + const { x, y, w, h } = block.bbox; + if (w * h < MIN_IMAGE_AREA) continue; // skip tiny icons + // Convert Y from mupdf (top-left) to PDF (bottom-left) for ordering + const pdfTopY = pageHeight - y; + regions.push({ + id: `p${pageNumber}-img${regions.length}`, + pageNumber, + bbox: { x, y, w, h }, + topY: pdfTopY, + }); + } + return regions; +} + +// --------------------------------------------------------------------------- +// Public API +// --------------------------------------------------------------------------- +/** + * Render an image region from a PDF page as a PNG buffer. + * Uses mupdf's DrawDevice to render just the cropped area at 2x resolution. + */ +export function renderImageRegion(input: Uint8Array, region: ImageRegion): Uint8Array { + const doc = mupdf.Document.openDocument(input, "application/pdf"); + const page = doc.loadPage(region.pageNumber - 1); + const pad = 10; + const bx = region.bbox.x - pad; + const by = region.bbox.y - pad; + const bw = region.bbox.w + 2 * pad; + const bh = region.bbox.h + 2 * pad; + const scale = 2; + const pw = Math.round(bw * scale); + const ph = Math.round(bh * scale); + const pix = new mupdf.Pixmap(mupdf.ColorSpace.DeviceRGB, [0, 0, pw, ph], false); + pix.clear(255); + const matrix: mupdf.Matrix = [scale, 0, 0, scale, -bx * scale, -by * scale]; + const dl = page.toDisplayList(); + const dev = new mupdf.DrawDevice(matrix, pix); + dl.run(dev, mupdf.Matrix.identity); + dev.close(); + return pix.asPNG(); +} + +/** + * Extract text boxes and vector segments from all pages of a PDF buffer. + */ +export async function extractPages(input: Uint8Array): Promise { + const doc = mupdf.Document.openDocument(input, "application/pdf"); + const pages: PageContent[] = []; + for (let i = 0; i < doc.countPages(); i++) { + const pageNumber = i + 1; + const page = doc.loadPage(i); + const bounds = page.getBounds(); + const pageHeight = bounds[3] - bounds[1]; + // Single structured text pass with both flags + const stext = JSON.parse( + page.toStructuredText("preserve-whitespace,preserve-images").asJSON(), + ) as StructuredTextJSON; + // Extract text boxes and image regions from the same parse + const textBoxes = extractTextBoxes(page, pageNumber, pageHeight, stext); + const images = extractImageRegions(stext, pageNumber, pageHeight); + // Extract vector segments from raw content stream + let segments: Segment[] = []; + try { + const pageObj = (page as mupdf.PDFPage).getObject(); + const contents = pageObj.get("Contents"); + if (contents) { + let rawBytes: Uint8Array; + if (contents.isArray()) { + // Multiple content streams — concatenate + const parts: Uint8Array[] = []; + const len = contents.length ?? 0; + for (let j = 0; j < len; j++) { + const stream = contents.get(j); + if (stream?.readStream) { + parts.push(stream.readStream().asUint8Array()); + } + } + const totalLen = parts.reduce((s, p) => s + p.length, 0); + rawBytes = new Uint8Array(totalLen); + let offset = 0; + for (const part of parts) { + rawBytes.set(part, offset); + offset += part.length; + } + } else { + rawBytes = contents.readStream().asUint8Array(); + } + const raw = new TextDecoder().decode(rawBytes); + segments = extractSegmentsFromContentStream(raw, pageNumber); + } + } catch { + // Content stream extraction failed — proceed with text only + } + pages.push({ pageNumber, textBoxes, segments, images }); + } + return pages; +} diff --git a/packages/coding-agent/src/markit/converters/pdf/grid.ts b/packages/coding-agent/src/markit/converters/pdf/grid.ts new file mode 100644 index 000000000..a856588ff --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pdf/grid.ts @@ -0,0 +1,780 @@ +// Adapted from markit-ai (MIT). See ../../NOTICE. + +/** + * Table grid detection from vector segments and text boxes. + * + * Ported from @oharato/pdf2md-ts with TypeScript types and without + * CJK-specific borderless table heuristics. The core algorithm: + * + * 1. Classify segments as horizontal or vertical lines + * 2. Group horizontal Y-lines into table groups (split by vertical gaps) + * 3. For each group: + * a. Full grid (H+V lines): build cells from grid intersections, + * place text via raycasting + * b. H-line only (no V lines): infer columns from text X positions + * 4. Prune empty rows/cols + * + * Coordinate system: PDF native (bottom-left origin, Y increases upward). + */ +import type { Segment, TableCell, TableGrid, TextBox } from "./types"; + +export interface GridResult { + grids: TableGrid[]; + consumedIds: string[]; +} + +type RayDirection = "up" | "down" | "left" | "right"; + +interface Ray { + direction: RayDirection; + segmentId: string | null; + distance: number; +} + +interface Interval { + min: number; + max: number; +} + +function castRaysForTextBox(textBox: TextBox, segments: Segment[]): Ray[] { + const cx = (textBox.bounds.left + textBox.bounds.right) / 2; + const cy = (textBox.bounds.top + textBox.bounds.bottom) / 2; + let up: Ray = { direction: "up", segmentId: null, distance: Infinity }; + let down: Ray = { direction: "down", segmentId: null, distance: Infinity }; + let left: Ray = { direction: "left", segmentId: null, distance: Infinity }; + let right: Ray = { + direction: "right", + segmentId: null, + distance: Infinity, + }; + for (const seg of segments) { + const isH = Math.abs(seg.y1 - seg.y2) < 0.5; + const isV = Math.abs(seg.x1 - seg.x2) < 0.5; + if (isH) { + const minX = Math.min(seg.x1, seg.x2); + const maxX = Math.max(seg.x1, seg.x2); + if (cx >= minX && cx <= maxX) { + const d = seg.y1 - cy; + if (d >= 0 && d < up.distance) up = { direction: "up", segmentId: seg.id, distance: d }; + const dd = cy - seg.y1; + if (dd >= 0 && dd < down.distance) down = { direction: "down", segmentId: seg.id, distance: dd }; + } + } + if (isV) { + const minY = Math.min(seg.y1, seg.y2); + const maxY = Math.max(seg.y1, seg.y2); + if (cy >= minY && cy <= maxY) { + const d = cx - seg.x1; + if (d >= 0 && d < left.distance) left = { direction: "left", segmentId: seg.id, distance: d }; + const rd = seg.x1 - cx; + if (rd >= 0 && rd < right.distance) right = { direction: "right", segmentId: seg.id, distance: rd }; + } + } + } + return [up, down, left, right]; +} + +// --------------------------------------------------------------------------- +// Utility +// --------------------------------------------------------------------------- +const AXIS_EPSILON = 0.8; +const PAGE_MARGIN = 20; + +function uniqueSorted(values: number[]): number[] { + const sorted = [...values].sort((a, b) => a - b); + const result: number[] = []; + for (const v of sorted) { + if (result.length === 0 || Math.abs(result[result.length - 1] - v) > 1) result.push(v); + } + return result; +} + +// --------------------------------------------------------------------------- +// Y-line group splitting +// --------------------------------------------------------------------------- +function chainCoversRange(intervals: Interval[], lowerY: number, upperY: number, eps: number): boolean { + const sorted = [...intervals].sort((a, b) => a.min - b.min); + let covered = lowerY; + for (const iv of sorted) { + if (iv.min > covered + eps) break; + if (iv.max > covered) covered = iv.max; + if (covered >= upperY - eps) return true; + } + return false; +} + +function countBridgingVLineCols(upperY: number, lowerY: number, verticals: Segment[]): number { + const eps = 1.5; + const byX = new Map(); + for (const seg of verticals) { + const rx = Math.round(seg.x1); + if (!byX.has(rx)) byX.set(rx, []); + byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) }); + } + let count = 0; + for (const intervals of byX.values()) { + if (chainCoversRange(intervals, lowerY, upperY, eps)) count++; + } + return count; +} + +function bridgingXSet(upperY: number, lowerY: number, verticals: Segment[]): Set { + const eps = 1.5; + const xs = new Set(); + const byX = new Map(); + for (const seg of verticals) { + const rx = Math.round(seg.x1); + if (!byX.has(rx)) byX.set(rx, []); + byX.get(rx)?.push({ min: Math.min(seg.y1, seg.y2), max: Math.max(seg.y1, seg.y2) }); + } + for (const [rx, intervals] of byX) { + if (chainCoversRange(intervals, lowerY, upperY, eps)) xs.add(rx); + } + return xs; +} + +const MIN_RICH_BRIDGING_COLS = 3; + +function splitYLinesIntoGroups(yLines: number[], verticals: Segment[]): number[][] { + if (yLines.length === 0) return []; + const eps = 1.5; + const allX = verticals.map(s => Math.round(s.x1)); + const globalXMin = allX.length > 0 ? Math.min(...allX) : 0; + const globalXMax = allX.length > 0 ? Math.max(...allX) : 0; + const groups: number[][] = []; + let currentGroup = [yLines[0]]; + let prevBridgingCols = -1; + for (let i = 1; i < yLines.length; i++) { + const upperY = yLines[i - 1]; + const lowerY = yLines[i]; + const cols = countBridgingVLineCols(upperY, lowerY, verticals); + if (cols === 0) { + groups.push(currentGroup); + currentGroup = [yLines[i]]; + prevBridgingCols = -1; + continue; + } + if (prevBridgingCols >= MIN_RICH_BRIDGING_COLS && cols < MIN_RICH_BRIDGING_COLS) { + const bxs = bridgingXSet(upperY, lowerY, verticals); + const isOuterFrameOnly = [...bxs].every( + x => Math.abs(x - globalXMin) <= eps || Math.abs(x - globalXMax) <= eps, + ); + if (!isOuterFrameOnly) { + groups.push(currentGroup); + currentGroup = [yLines[i - 1], yLines[i]]; + prevBridgingCols = cols; + continue; + } + } + currentGroup.push(yLines[i]); + prevBridgingCols = cols; + } + groups.push(currentGroup); + return groups; +} + +// --------------------------------------------------------------------------- +// Sub-row Y-cluster expansion +// --------------------------------------------------------------------------- +const Y_CLUSTER_GAP = 10; +const MIN_COLS_IN_TOP_CLUSTER = 2; + +function assignToYCluster(y: number, clusters: number[]): number { + let closest = 0; + let closestDist = Math.abs(y - clusters[0]); + for (let k = 1; k < clusters.length; k++) { + const d = Math.abs(y - clusters[k]); + if (d < closestDist) { + closestDist = d; + closest = k; + } + } + return closest; +} + +function expandSubRowsByYClusters( + originalRows: number, + cols: number, + cells: TableCell[], + cellBoxes: Map, +): number { + let addedRows = 0; + for (let origRow = 0; origRow < originalRows; origRow++) { + const currentRow = origRow + addedRows; + const rowCellInfos: Array<{ cell: TableCell; col: number; boxes: TextBox[] }> = []; + for (let col = 0; col < cols; col++) { + const cell = cells.find(c => c.row === currentRow && c.col === col); + if (!cell) continue; + const boxes = cellBoxes.get(cell); + if (boxes && boxes.length > 0) rowCellInfos.push({ cell, col, boxes }); + } + if (rowCellInfos.length === 0) continue; + const allMidYs = rowCellInfos.flatMap(({ boxes }) => boxes.map(b => (b.bounds.top + b.bounds.bottom) / 2)); + const sortedY = [...new Set(allMidYs.map(y => Math.round(y * 10) / 10))].sort((a, b) => b - a); + const clusters = [sortedY[0]]; + for (let i = 1; i < sortedY.length; i++) { + if (clusters[clusters.length - 1] - sortedY[i] > Y_CLUSTER_GAP) { + clusters.push(sortedY[i]); + } + } + if (clusters.length < 2) continue; + const colsInTopCluster = new Set(); + const totalNonEmptyCols = new Set(); + for (const { col, boxes } of rowCellInfos) { + totalNonEmptyCols.add(col); + if (boxes.some(b => assignToYCluster((b.bounds.top + b.bounds.bottom) / 2, clusters) === 0)) { + colsInTopCluster.add(col); + } + } + if (colsInTopCluster.size < MIN_COLS_IN_TOP_CLUSTER) continue; + if (colsInTopCluster.size >= totalNonEmptyCols.size) continue; + const sparseColsHaveMultipleBoxes = rowCellInfos.some( + ({ col, boxes }) => !colsInTopCluster.has(col) && boxes.length > 1, + ); + if (!sparseColsHaveMultipleBoxes) continue; + const numSubRows = clusters.length; + const numNewRows = numSubRows - 1; + for (const cell of cells) { + if (cell.row > currentRow) cell.row += numNewRows; + } + for (let subRow = 1; subRow < numSubRows; subRow++) { + for (let col = 0; col < cols; col++) { + cells.push({ + row: currentRow + subRow, + col, + text: "", + rowSpan: 1, + colSpan: 1, + }); + } + } + for (const { cell: origCell, col, boxes } of rowCellInfos) { + const subRowBoxGroups: TextBox[][] = Array.from({ length: numSubRows }, () => []); + for (const box of boxes) { + const cy = (box.bounds.top + box.bounds.bottom) / 2; + subRowBoxGroups[assignToYCluster(cy, clusters)].push(box); + } + cellBoxes.set(origCell, subRowBoxGroups[0]); + if (subRowBoxGroups[0].length === 0) cellBoxes.delete(origCell); + for (let subRow = 1; subRow < numSubRows; subRow++) { + if (subRowBoxGroups[subRow].length > 0) { + const newCell = cells.find(c => c.row === currentRow + subRow && c.col === col); + if (newCell) cellBoxes.set(newCell, subRowBoxGroups[subRow]); + } + } + } + addedRows += numNewRows; + } + return originalRows + addedRows; +} + +// --------------------------------------------------------------------------- +// Cross-column text box splitting +// --------------------------------------------------------------------------- +/** + * Find which column a horizontal position falls into. + * Returns -1 if outside the grid. + */ +function findCol(x: number, xLines: number[]): number { + for (let i = 0; i < xLines.length - 1; i++) { + if (x >= xLines[i] && x <= xLines[i + 1]) return i; + } + return -1; +} + +/** + * When a text box spans across one or more vertical column boundaries, + * split it into multiple virtual text boxes — one per column — with the + * text divided proportionally by width. + * + * We split at word boundaries closest to the proportional split point + * so we don't chop words in half. + */ +function splitCrossColumnBoxes(textBoxes: TextBox[], xLines: number[]): TextBox[] { + const result: TextBox[] = []; + const MARGIN = 5; // allow small overlap before considering it cross-column + for (const tb of textBoxes) { + const leftCol = findCol(tb.bounds.left + MARGIN, xLines); + const rightCol = findCol(tb.bounds.right - MARGIN, xLines); + // Not spanning columns, or outside grid — keep as-is + if (leftCol < 0 || rightCol < 0 || leftCol === rightCol) { + result.push(tb); + continue; + } + // Text box spans from leftCol to rightCol — split it + const totalWidth = tb.bounds.right - tb.bounds.left; + if (totalWidth <= 0) { + result.push(tb); + continue; + } + const words = tb.text.split(/\s+/); + if (words.length <= 1) { + // Single word spanning columns — just assign to whichever col has more overlap + result.push(tb); + continue; + } + // For each column boundary crossing, find the best word-boundary split + let remainingWords = [...words]; + let currentLeft = tb.bounds.left; + for (let col = leftCol; col <= rightCol && remainingWords.length > 0; col++) { + const colRight = col < xLines.length - 1 ? xLines[col + 1] : tb.bounds.right; + const segmentRight = Math.min(colRight, tb.bounds.right); + if (col === rightCol) { + // Last column — take all remaining words + result.push({ + ...tb, + id: `${tb.id}-split${col}`, + text: remainingWords.join(" "), + bounds: { + ...tb.bounds, + left: currentLeft, + right: tb.bounds.right, + }, + }); + remainingWords = []; + } else { + // Find how many words fit in this column segment proportionally + const segmentWidth = segmentRight - currentLeft; + const fractionOfTotal = segmentWidth / totalWidth; + const approxChars = Math.round(fractionOfTotal * tb.text.length); + // Walk words to find the split closest to the proportional point + let charCount = 0; + let splitIdx = 0; + for (let w = 0; w < remainingWords.length; w++) { + const nextCount = charCount + remainingWords[w].length + (w > 0 ? 1 : 0); + if (nextCount > approxChars && splitIdx > 0) break; + charCount = nextCount; + splitIdx = w + 1; + } + if (splitIdx === 0) splitIdx = 1; // take at least one word + if (splitIdx >= remainingWords.length) { + // All remaining words fit here + result.push({ + ...tb, + id: `${tb.id}-split${col}`, + text: remainingWords.join(" "), + bounds: { + ...tb.bounds, + left: currentLeft, + right: segmentRight, + }, + }); + remainingWords = []; + } else { + const partWords = remainingWords.slice(0, splitIdx); + result.push({ + ...tb, + id: `${tb.id}-split${col}`, + text: partWords.join(" "), + bounds: { + ...tb.bounds, + left: currentLeft, + right: segmentRight, + }, + }); + remainingWords = remainingWords.slice(splitIdx); + currentLeft = segmentRight; + } + } + } + } + return result; +} + +// --------------------------------------------------------------------------- +// Full grid table (H + V lines) +// --------------------------------------------------------------------------- +function buildCells(rows: number, cols: number): TableCell[] { + const cells: TableCell[] = []; + for (let row = 0; row < rows; row++) { + for (let col = 0; col < cols; col++) { + cells.push({ row, col, text: "", rowSpan: 1, colSpan: 1 }); + } + } + return cells; +} + +function buildTableGrid( + pageNumber: number, + yLines: number[], + xLines: number[], + filteredSegments: Segment[], + textBoxes: TextBox[], +): { grid: TableGrid; consumedIds: string[] } { + let rows = yLines.length - 1; + const cols = xLines.length - 1; + const cells = buildCells(rows, cols); + const consumedIds: string[] = []; + const yMin = yLines[yLines.length - 1]; + const yMax = yLines[0]; + const xMin = xLines[0]; + const xMax = xLines[xLines.length - 1]; + // Split text boxes that span multiple columns before placement + const splitBoxes = splitCrossColumnBoxes(textBoxes, xLines); + // Track which split piece IDs get placed in cells, so we can consume + // the original (unsplit) text box IDs too. + const placedSplitIds = new Set(); + // Look for header text boxes just above the grid. + // Use the ORIGINAL (unsplit) text boxes for header detection so that + // wide paragraph text isn't falsely split into column-sized header chunks. + // Reject boxes wider than 1.5 columns — those are paragraph text, not headers. + const avgColWidth = (xMax - xMin) / cols; + const maxHeaderBoxWidth = avgColWidth * 1.5; + const headerBoxes = textBoxes.filter(tb => { + const cy = (tb.bounds.top + tb.bounds.bottom) / 2; + const cx = (tb.bounds.left + tb.bounds.right) / 2; + const boxWidth = tb.bounds.right - tb.bounds.left; + return cy > yMax && cy <= yMax + 20 && cx >= xMin && cx <= xMax && boxWidth <= maxHeaderBoxWidth; + }); + if (headerBoxes.length > 0) { + rows += 1; + for (const cell of cells) cell.row += 1; + for (let col = 0; col < cols; col++) { + cells.push({ row: 0, col, text: "", rowSpan: 1, colSpan: 1 }); + } + for (const tb of headerBoxes) { + const cx = (tb.bounds.left + tb.bounds.right) / 2; + const col = xLines.findIndex((lineX, idx) => { + const next = xLines[idx + 1]; + return next !== undefined && cx >= lineX && cx <= next; + }); + if (col >= 0 && col < cols) { + const cell = cells.find(c => c.row === 0 && c.col === col); + if (cell) { + cell.text = cell.text.length === 0 ? tb.text : `${cell.text} ${tb.text}`; + consumedIds.push(tb.id); + } + } + } + } + const cellBoxes = new Map(); + for (const tb of splitBoxes) { + const cx = (tb.bounds.left + tb.bounds.right) / 2; + const cy = (tb.bounds.top + tb.bounds.bottom) / 2; + if (cy < yMin || cy > yMax || cx < xMin || cx > xMax) continue; + const rays = castRaysForTextBox(tb, filteredSegments); + const rayConfidence = rays.filter(r => r.segmentId !== null).length; + let row = yLines.findIndex((lineY, idx) => { + const next = yLines[idx + 1]; + return next !== undefined && cy <= lineY && cy >= next; + }); + if (row < 0 || row >= (headerBoxes.length > 0 ? rows - 1 : rows)) continue; + if (headerBoxes.length > 0) row += 1; + const col = xLines.findIndex((lineX, idx) => { + const next = xLines[idx + 1]; + return next !== undefined && cx >= lineX && cx <= next; + }); + if (col < 0 || col >= cols) continue; + if (rayConfidence === 0) continue; + const cell = cells.find(c => c.row === row && c.col === col); + if (!cell) continue; + if (!cellBoxes.has(cell)) cellBoxes.set(cell, []); + cellBoxes.get(cell)?.push(tb); + consumedIds.push(tb.id); + if (tb.id.includes("-split")) placedSplitIds.add(tb.id); + } + rows = expandSubRowsByYClusters(rows, cols, cells, cellBoxes); + // Merge text boxes within each cell into cell text + for (const [cell, boxes] of cellBoxes.entries()) { + boxes.sort((a, b) => b.bounds.top - a.bounds.top); + const lines: string[] = []; + let currentLine: string[] = []; + let currentY = boxes[0].bounds.top; + for (const box of boxes) { + if (Math.abs(box.bounds.top - currentY) > 5) { + lines.push(currentLine.join(" ")); + currentLine = [box.text]; + currentY = box.bounds.top; + } else { + currentLine.push(box.text); + } + } + if (currentLine.length > 0) lines.push(currentLine.join(" ")); + cell.text = lines.join("
"); + } + const grid = pruneEmptyRowsAndCols({ + pageNumber, + rows, + cols, + cells, + warnings: [], + topY: yLines[0], + isBorderless: false, + }); + // Also consume the original (unsplit) text box IDs when any of their + // split pieces were placed in a cell. + for (const splitId of placedSplitIds) { + const origId = splitId.replace(/-split\d+$/, ""); + if (!consumedIds.includes(origId)) { + consumedIds.push(origId); + } + } + return { grid, consumedIds }; +} + +// --------------------------------------------------------------------------- +// H-line-only table (inferred columns) +// --------------------------------------------------------------------------- +const COL_GAP_THRESHOLD = 20; +const HONLY_ROW_GAP = 30; +const HONLY_ROW_TOLERANCE = 8; +const MIN_TABLE_HEIGHT = 24; +const MIN_LEFT_SPREAD = 50; + +function inferXLinesFromBoxes(textBoxes: TextBox[], xMin: number, xMax: number): number[] { + const centers = textBoxes.map(tb => (tb.bounds.left + tb.bounds.right) / 2).sort((a, b) => a - b); + if (centers.length === 0) return [xMin, xMax]; + const boundaries = [xMin]; + for (let i = 1; i < centers.length; i++) { + if (centers[i] - centers[i - 1] >= COL_GAP_THRESHOLD) { + boundaries.push((centers[i - 1] + centers[i]) / 2); + } + } + boundaries.push(xMax); + return boundaries; +} + +function buildHLineOnlyTable( + pageNumber: number, + yLines: number[], + xMin: number, + xMax: number, + textBoxes: TextBox[], + alreadyConsumed: Set, +): { grid: TableGrid; consumedIds: string[] } | null { + const yMax = yLines[0]; + const yMin = yLines[yLines.length - 1]; + const candidates = textBoxes.filter(tb => !alreadyConsumed.has(tb.id)); + const BOX_LEFT_TOLERANCE = 30; + const inRange = candidates.filter(tb => { + const cy = (tb.bounds.top + tb.bounds.bottom) / 2; + return ( + tb.bounds.left >= xMin - BOX_LEFT_TOLERANCE && + tb.bounds.right <= xMax + BOX_LEFT_TOLERANCE && + cy >= yMin && + cy <= yMax + ); + }); + // Extend downward below yMin + const belowYMin = candidates + .filter(tb => { + const cx = (tb.bounds.left + tb.bounds.right) / 2; + const cy = (tb.bounds.top + tb.bounds.bottom) / 2; + return cx >= xMin && cx <= xMax && cy < yMin; + }) + .sort((a, b) => (b.bounds.top + b.bounds.bottom) / 2 - (a.bounds.top + a.bounds.bottom) / 2); + const extensionBoxes: TextBox[] = []; + let lastY = yMin; + for (const tb of belowYMin) { + const cy = (tb.bounds.top + tb.bounds.bottom) / 2; + if (lastY - cy > HONLY_ROW_GAP) break; + extensionBoxes.push(tb); + lastY = cy; + } + const allBoxes = [...inRange, ...extensionBoxes]; + if (allBoxes.length === 0) return null; + const leftEdges = allBoxes.map(tb => tb.bounds.left); + if (Math.max(...leftEdges) - Math.min(...leftEdges) < MIN_LEFT_SPREAD) return null; + const xLines = inferXLinesFromBoxes(allBoxes, xMin, xMax); + if (xLines.length < 2) return null; + const cols = xLines.length - 1; + // Build visual rows + const visualRows: Array<{ midY: number; boxes: TextBox[] }> = []; + const sortedBoxes = [...allBoxes].sort((a, b) => { + const ya = (a.bounds.top + a.bounds.bottom) / 2; + const yb = (b.bounds.top + b.bounds.bottom) / 2; + if (Math.abs(ya - yb) > 0.5) return yb - ya; + return a.bounds.left - b.bounds.left; + }); + for (const box of sortedBoxes) { + const cy = (box.bounds.top + box.bounds.bottom) / 2; + const last = visualRows[visualRows.length - 1]; + if (last && Math.abs(last.midY - cy) <= HONLY_ROW_TOLERANCE) { + last.boxes.push(box); + } else { + visualRows.push({ midY: cy, boxes: [box] }); + } + } + if (visualRows.length === 0) return null; + const cells: TableCell[] = []; + const consumedIds: string[] = []; + for (let rowIdx = 0; rowIdx < visualRows.length; rowIdx++) { + const vrow = visualRows[rowIdx]; + const colBoxes = new Map(); + for (const box of vrow.boxes) { + const cx = (box.bounds.left + box.bounds.right) / 2; + const col = xLines.findIndex((lineX, idx) => { + const next = xLines[idx + 1]; + return next !== undefined && cx >= lineX && cx <= next; + }); + if (col >= 0 && col < cols) { + if (!colBoxes.has(col)) colBoxes.set(col, []); + colBoxes.get(col)?.push(box); + } + } + for (let c = 0; c < cols; c++) { + const cbs = (colBoxes.get(c) ?? []).sort((a, b) => a.bounds.left - b.bounds.left); + cells.push({ + row: rowIdx, + col: c, + text: cbs.map(b => b.text).join(" "), + rowSpan: 1, + colSpan: 1, + }); + consumedIds.push(...cbs.map(b => b.id)); + } + } + const contentTopY = visualRows.length > 0 ? visualRows[0].midY : yMax; + const grid = pruneEmptyRowsAndCols({ + pageNumber, + rows: visualRows.length, + cols, + cells, + warnings: [], + topY: contentTopY, + isBorderless: false, + }); + return { grid, consumedIds }; +} + +// --------------------------------------------------------------------------- +// Pruning +// --------------------------------------------------------------------------- +function pruneEmptyRowsAndCols(table: TableGrid): TableGrid { + const occupiedRows = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.row)); + const occupiedCols = new Set(table.cells.filter(c => c.text.trim().length > 0).map(c => c.col)); + if (occupiedRows.size === 0) return table; + const rowMap = new Map(); + let newRow = 0; + for (let r = 0; r < table.rows; r++) { + if (occupiedRows.has(r)) rowMap.set(r, newRow++); + } + const colMap = new Map(); + let newCol = 0; + for (let c = 0; c < table.cols; c++) { + if (occupiedCols.has(c)) colMap.set(c, newCol++); + } + const prunedCells = table.cells + .filter(c => occupiedRows.has(c.row) && occupiedCols.has(c.col)) + .map(c => ({ + ...c, + row: rowMap.get(c.row) ?? c.row, + col: colMap.get(c.col) ?? c.col, + })); + return { ...table, rows: newRow, cols: newCol, cells: prunedCells }; +} + +// --------------------------------------------------------------------------- +// Diagram vs table discrimination +// --------------------------------------------------------------------------- +/** Maximum column count for a plausible data table. */ +const MAX_TABLE_COLS = 25; + +/** + * Returns true if a grid looks like a vector diagram rather than a data table. + * + * Heuristics (any match → diagram): + * 1. Column count > 25 (diagrams create many X-lines from box edges) + * 2. Fill ratio < 25% (most cells empty — scattered boxes) + * 3. Fill < 50% AND duplicate text ratio > 30% (repeating labels in a + * diagram layout, e.g. "Hash", "Transaction" appearing in each column) + * 4. Fill < 50% AND cols >= 6 (moderate sparseness with wide grid) + */ +function isDiagram(grid: TableGrid): boolean { + const totalCells = grid.rows * grid.cols; + if (totalCells === 0) return true; + const filled = grid.cells.filter(c => c.text.trim().length > 0); + const fillRatio = filled.length / totalCells; + // Very high column count + if (grid.cols > MAX_TABLE_COLS) return true; + // Very sparse + if (fillRatio < 0.25) return true; + // Compute duplicate text ratio among non-trivial cells. + // Exclude short values (≤3 chars) like "—", "V", "YES", "NO" which + // naturally repeat in real data tables. + const substantive = filled.filter(c => c.text.trim().length > 3); + const uniqueTexts = new Set(substantive.map(c => c.text.trim())).size; + const dupRatio = substantive.length > 2 ? 1 - uniqueTexts / substantive.length : 0; + // Sparse + highly duplicated substantive text → repeating diagram + if (fillRatio < 0.5 && dupRatio > 0.3) return true; + // High duplication + wide grid → repeating diagram even at moderate fill + if (dupRatio > 0.4 && grid.cols >= 6) return true; + // Sparse + wide grid with no substantive text to judge + if (fillRatio < 0.4 && grid.cols >= 6) return true; + return false; +} + +/** + * Detect all table grids on a single page from its text boxes and segments. + */ +export function resolveTableGrids(pageNumber: number, textBoxes: TextBox[], segments: Segment[]): GridResult { + const vertical = segments.filter(s => Math.abs(s.x1 - s.x2) <= AXIS_EPSILON); + const horizontal = segments.filter(s => Math.abs(s.y1 - s.y2) <= AXIS_EPSILON); + // Filter segments to the text's visible area + const textYValues = textBoxes.flatMap(t => [t.bounds.bottom, t.bounds.top]); + const textYMin = textYValues.length > 0 ? Math.min(...textYValues) - PAGE_MARGIN : -Infinity; + const textYMax = textYValues.length > 0 ? Math.max(...textYValues) + PAGE_MARGIN : Infinity; + const textXValues = textBoxes.flatMap(t => [t.bounds.left, t.bounds.right]); + const textXMin = textXValues.length > 0 ? Math.min(...textXValues) - 100 : -Infinity; + const textXMax = textXValues.length > 0 ? Math.max(...textXValues) + 100 : Infinity; + const filteredH = horizontal.filter( + s => s.y1 >= textYMin && s.y1 <= textYMax && s.x1 <= textXMax && s.x2 >= textXMin, + ); + const hMaxX2 = filteredH.length > 0 ? Math.max(...filteredH.map(s => s.x2)) : textXMax; + const vSegXMax = Math.max(textXMax, hMaxX2 + PAGE_MARGIN); + const filteredV = vertical.filter(s => { + const segMin = Math.min(s.y1, s.y2); + const segMax = Math.max(s.y1, s.y2); + return segMax >= textYMin && segMin <= textYMax && s.x1 >= textXMin && s.x1 <= vSegXMax; + }); + const allYLines = uniqueSorted(filteredH.flatMap(s => [s.y1, s.y2])).sort((a, b) => b - a); + if (allYLines.length < 2) { + return { grids: [], consumedIds: [] }; + } + const filteredSegments = [...filteredH, ...filteredV]; + const yGroups = splitYLinesIntoGroups(allYLines, filteredV); + const grids: TableGrid[] = []; + const gridConsumedIds: string[][] = []; + // Flat set for the alreadyConsumed check in H-line-only tables + const allConsumedIds: string[] = []; + for (const yLines of yGroups) { + if (yLines.length < 2) continue; + const yMin = yLines[yLines.length - 1]; + const yMax = yLines[0]; + const groupVerticals = filteredV.filter(s => { + const segMin = Math.min(s.y1, s.y2); + const segMax = Math.max(s.y1, s.y2); + return segMin < yMax - 1.5 && segMax > yMin + 1.5; + }); + const groupXLines = uniqueSorted(groupVerticals.flatMap(s => [s.x1, s.x2])); + if (groupXLines.length < 2) { + if (yMax - yMin < MIN_TABLE_HEIGHT) continue; + const groupHoriz = filteredH.filter(s => s.y1 >= yMin - 1.5 && s.y1 <= yMax + 1.5); + if (groupHoriz.length === 0) continue; + const hxMin = Math.min(...groupHoriz.map(s => s.x1)); + const hxMax = Math.max(...groupHoriz.map(s => s.x2)); + const result = buildHLineOnlyTable(pageNumber, yLines, hxMin, hxMax, textBoxes, new Set(allConsumedIds)); + if (result) { + grids.push(result.grid); + gridConsumedIds.push(result.consumedIds); + allConsumedIds.push(...result.consumedIds); + } + continue; + } + if (yMax - yMin < MIN_TABLE_HEIGHT) continue; + const result = buildTableGrid(pageNumber, yLines, groupXLines, filteredSegments, textBoxes); + grids.push(result.grid); + gridConsumedIds.push(result.consumedIds); + allConsumedIds.push(...result.consumedIds); + } + // Filter out grids that look like vector diagrams, not data tables. + // Their consumed text box IDs are released so the text becomes free text. + const filteredGrids: TableGrid[] = []; + const filteredConsumedIds: string[] = []; + for (let i = 0; i < grids.length; i++) { + if (isDiagram(grids[i])) continue; + filteredGrids.push(grids[i]); + filteredConsumedIds.push(...gridConsumedIds[i]); + } + return { grids: filteredGrids, consumedIds: filteredConsumedIds }; +} diff --git a/packages/coding-agent/src/markit/converters/pdf/headers.ts b/packages/coding-agent/src/markit/converters/pdf/headers.ts new file mode 100644 index 000000000..f21bfcfbd --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pdf/headers.ts @@ -0,0 +1,106 @@ +// Adapted from markit-ai (MIT). See ../../NOTICE. + +/** + * Running header/footer detection and removal. + * + * Many PDFs have repeated text at the top or bottom of every page: + * document titles, chapter names, page numbers, copyright notices. + * These pollute the markdown output as false headings or noise. + * + * Algorithm: + * 1. For each page, bucket text boxes by Y position (top/bottom zones) + * 2. Collect the text content at each zone across all pages + * 3. Text appearing on >20% of pages OR 8+ consecutive pages is a + * running header/footer + * 4. Remove matching text boxes before further processing + */ +import type { PageContent } from "./types"; + +/** Minimum number of pages to enable header/footer detection. */ +const MIN_PAGES = 5; +/** Minimum Y position for top zone (from bottom of page in PDF coords). */ +const TOP_ZONE_MIN_Y = 700; +/** Maximum Y position for bottom zone. */ +const BOTTOM_ZONE_MAX_Y = 80; +/** + * Minimum consecutive pages a text must appear on to be considered a + * running header/footer. Catches both document-wide headers (appearing + * on every page) and chapter-specific headers (appearing on 4+ consecutive + * pages within a chapter). + */ +const MIN_CONSECUTIVE_PAGES = 8; + +/** + * Detect and remove running headers and footers from all pages. + * Mutates the pages array in place, removing header/footer text boxes. + * + * Uses two strategies: + * 1. Global frequency: text appearing on > 20% of all pages + * 2. Consecutive runs: text appearing on 8+ consecutive pages + */ +export function stripHeadersFooters(pages: PageContent[]): void { + if (pages.length < MIN_PAGES) return; + // Step 1: Build per-page zone text sets + const pageZoneTexts: Set[] = []; + for (const page of pages) { + const zoneTexts = new Set(); + for (const tb of page.textBoxes) { + const midY = (tb.bounds.top + tb.bounds.bottom) / 2; + if (midY >= TOP_ZONE_MIN_Y || midY <= BOTTOM_ZONE_MAX_Y) { + const key = tb.text.trim().replace(/\s+/g, " "); + if (key.length > 0) zoneTexts.add(key); + } + } + pageZoneTexts.push(zoneTexts); + } + // Step 2: Count global frequency AND longest consecutive run for each text + const globalCount = new Map(); + const maxConsecutive = new Map(); + // Collect all unique zone texts + const allTexts = new Set(); + for (const zts of pageZoneTexts) { + for (const t of zts) allTexts.add(t); + } + for (const text of allTexts) { + let total = 0; + let consecutive = 0; + let maxRun = 0; + for (const zts of pageZoneTexts) { + if (zts.has(text)) { + total++; + consecutive++; + if (consecutive > maxRun) maxRun = consecutive; + } else { + consecutive = 0; + } + } + globalCount.set(text, total); + maxConsecutive.set(text, maxRun); + } + // Step 3: Identify running headers/footers + const globalThreshold = Math.max(3, Math.floor(pages.length * 0.2)); + const repeatedTexts = new Set(); + for (const text of allTexts) { + const gc = globalCount.get(text) ?? 0; + const mc = maxConsecutive.get(text) ?? 0; + // Global: appears on 20%+ of pages + if (gc >= globalThreshold) { + repeatedTexts.add(text); + continue; + } + // Consecutive: appears on 8+ consecutive pages (chapter-level headers) + if (mc >= MIN_CONSECUTIVE_PAGES) { + repeatedTexts.add(text); + } + } + if (repeatedTexts.size === 0) return; + // Step 4: Remove matching text boxes from each page + for (const page of pages) { + page.textBoxes = page.textBoxes.filter(tb => { + const midY = (tb.bounds.top + tb.bounds.bottom) / 2; + if (midY < TOP_ZONE_MIN_Y && midY > BOTTOM_ZONE_MAX_Y) return true; + const normalized = tb.text.trim().replace(/\s+/g, " "); + return !repeatedTexts.has(normalized); + }); + } +} diff --git a/packages/coding-agent/src/markit/converters/pdf/index.ts b/packages/coding-agent/src/markit/converters/pdf/index.ts new file mode 100644 index 000000000..726d4295e --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pdf/index.ts @@ -0,0 +1,146 @@ +// Adapted from markit-ai (MIT). See ../../NOTICE. + +/** + * PDF to Markdown converter. + * + * Uses mupdf (native WASM) for fast PDF parsing and a custom pipeline for + * table detection via vector line extraction + raycasting. + * + * Pipeline: + * 1. Extract text boxes + vector segments + image regions per page (mupdf) + * 2. Detect column layout (single vs multi-column) + * 3. Per column: detect table grids from segments (grid detection + raycasting) + * 4. Render diagrams as PNG files (if output directory provided) + * 5. Render tables as markdown tables, free text as paragraphs/headings + */ +import * as path from "node:path"; +import type { ConversionResult, Converter, StreamInfo } from "../../types"; +import { detectColumns } from "./columns"; +import { extractPages, renderImageRegion } from "./extract"; +import { resolveTableGrids } from "./grid"; +import { stripHeadersFooters } from "./headers"; +import { renderPageContent } from "./render"; +import type { Segment, TextBox } from "./types"; + +const EXTENSIONS = [".pdf"]; +const MIMETYPES = ["application/pdf", "application/x-pdf"]; + +type ImageBlock = { topY: number; markdown: string }; + +/** + * Process a set of text boxes (one column or full page): run table detection, + * separate free text, and render to markdown. + */ +function processColumn( + pageNumber: number, + textBoxes: TextBox[], + segments: Segment[], + imageBlocks: ImageBlock[], +): string { + const { grids, consumedIds } = resolveTableGrids(pageNumber, textBoxes, segments); + const consumedSet = new Set(consumedIds); + const freeTextBoxes = textBoxes.filter(tb => !consumedSet.has(tb.id)); + return renderPageContent(freeTextBoxes, grids, imageBlocks, textBoxes); +} + +export class PdfConverter implements Converter { + name = "pdf"; + + accepts(streamInfo: StreamInfo): boolean { + if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) { + return true; + } + if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) { + return true; + } + return false; + } + + async convert(input: Buffer, streamInfo: StreamInfo): Promise { + const pdfBytes = new Uint8Array(input); + const pages = await extractPages(pdfBytes); + // Remove running headers/footers before processing. + stripHeadersFooters(pages); + const imageDir = streamInfo.imageDir; + + const pageMarkdowns: string[] = []; + for (const page of pages) { + // Build image blocks for this page. + const imageBlocks: ImageBlock[] = []; + if (imageDir && page.images.length > 0) { + for (const img of page.images) { + const filename = `${img.id}.png`; + const filepath = path.join(imageDir, filename); + try { + const png = renderImageRegion(pdfBytes, img); + await Bun.write(filepath, png); + imageBlocks.push({ topY: img.topY, markdown: `![${img.id}](${filepath})` }); + } catch { + // Image rendering failed — skip. + } + } + } else if (page.images.length > 0) { + for (const img of page.images) { + imageBlocks.push({ + topY: img.topY, + markdown: ``, + }); + } + } + + // Detect column layout. + // If the page has vertical segments (tables), suppress column detection + // when one detected column is very narrow — that's a table's first column, + // not a page layout column. + const layout = detectColumns(page.textBoxes); + if (layout.columnCount > 1 && page.segments.some(s => Math.abs(s.x1 - s.x2) <= 0.8)) { + const pageXMin = Math.min(...page.textBoxes.map(tb => tb.bounds.left)); + const pageXMax = Math.max(...page.textBoxes.map(tb => tb.bounds.right)); + const pageWidth = pageXMax - pageXMin; + const minColFraction = 0.3; + const tooNarrow = layout.columns.some(col => { + const colXMin = Math.min(...col.map(tb => tb.bounds.left)); + const colXMax = Math.max(...col.map(tb => tb.bounds.right)); + return (colXMax - colXMin) / pageWidth < minColFraction; + }); + if (tooNarrow) { + layout.columnCount = 1; + layout.columns = [page.textBoxes]; + layout.boundaries = []; + } + } + + if (layout.columnCount === 1) { + // Single column — process normally. + const md = processColumn(page.pageNumber, page.textBoxes, page.segments, imageBlocks); + if (md.length > 0) pageMarkdowns.push(md); + } else { + // Multi-column — process each column independently, then join. + const columnMarkdowns: string[] = []; + for (const colBoxes of layout.columns) { + // Filter segments to those within this column's X range. + const colXMin = Math.min(...colBoxes.map(tb => tb.bounds.left)); + const colXMax = Math.max(...colBoxes.map(tb => tb.bounds.right)); + const margin = 10; + const colSegments = page.segments.filter(seg => { + const segXMin = Math.min(seg.x1, seg.x2); + const segXMax = Math.max(seg.x1, seg.x2); + return segXMax >= colXMin - margin && segXMin <= colXMax + margin; + }); + // Images go with the first column only (no X info to split by). + const md = processColumn( + page.pageNumber, + colBoxes, + colSegments, + columnMarkdowns.length === 0 ? imageBlocks : [], + ); + if (md.length > 0) columnMarkdowns.push(md); + } + const joined = columnMarkdowns.join("\n\n"); + if (joined.length > 0) pageMarkdowns.push(joined); + } + } + + return { markdown: pageMarkdowns.join("\n\n") }; + } +} diff --git a/packages/coding-agent/src/markit/converters/pdf/render.ts b/packages/coding-agent/src/markit/converters/pdf/render.ts new file mode 100644 index 000000000..a1a025f28 --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pdf/render.ts @@ -0,0 +1,501 @@ +// Adapted from markit-ai (MIT). See ../../NOTICE. + +/** + * Markdown rendering for PDF pages. + * + * Converts table grids and free text boxes into markdown, handling: + * - Table grid → markdown table (`| col | col |`) + * - Free text → paragraphs with heading detection (by font size) + * - Content ordering (top-to-bottom via Y coordinate) + * - Paragraph wrap merging (lines broken across PDF line boundaries) + * - Page number removal + * + * Ported from @oharato/pdf2md-ts, stripped of CJK/TDnet-specific logic. + */ +import type { ContentBlock, TableGrid, TextBox } from "./types"; + +/** A free-text line grouped from horizontally adjacent text boxes. */ +interface RenderLine { + text: string; + topY: number; + fontSize: number; + isBold: boolean; + isTabular: boolean; +} + +/** A content block carrying the Y of its last wrapped line during merging. */ +type WrapBlock = ContentBlock & { lastTopY: number }; + +// --------------------------------------------------------------------------- +// Utility +// --------------------------------------------------------------------------- +/** Convert full-width ASCII characters (A→A, !→! etc.) to normal ASCII. */ +function normalizeFullWidthAscii(text: string): string { + return text.replace(/[!-~]/g, ch => String.fromCharCode(ch.charCodeAt(0) - 0xfee0)); +} + +function escapePipes(text: string): string { + return normalizeFullWidthAscii(text).replaceAll("|", "\\|").replaceAll("\n", "
"); +} + +/** Parse a markdown pipe-delimited row into cell strings. */ +function parsePipeRow(line: string): string[] { + const trimmed = line.trim(); + if (!trimmed.startsWith("|") || !trimmed.endsWith("|")) return []; + return trimmed + .slice(1, -1) + .split("|") + .map(cell => cell.trim()); +} + +// --------------------------------------------------------------------------- +// Table rendering +// --------------------------------------------------------------------------- +/** + * Render a TableGrid as a markdown table. + */ +export function renderTableToMarkdown(table: TableGrid): string { + if (table.rows === 0 || table.cols === 0) return ""; + const matrix = Array.from({ length: table.rows }, () => Array.from({ length: table.cols }, () => "")); + for (const cell of table.cells) { + if (cell.row < table.rows && cell.col < table.cols) { + matrix[cell.row][cell.col] = escapePipes(cell.text.trim()); + } + } + const normalized = normalizeShiftedSparseColumns(matrix); + const promoted = promoteSubHeaderPrefixes(normalized); + const header = `| ${promoted[0].join(" | ")} |`; + const divider = `| ${Array.from({ length: promoted[0].length }, () => "---").join(" | ")} |`; + const body = promoted + .slice(1) + .map(row => `| ${row.join(" | ")} |`) + .join("\n"); + return [header, divider, body].filter(l => l.length > 0).join("\n"); +} + +/** + * Fix tables with ≥5 columns where sparse single-value columns are + * misaligned. Shifts those values to the adjacent dense column and + * removes the now-empty sparse columns. + */ +function normalizeShiftedSparseColumns(matrix: string[][]): string[][] { + if (matrix.length === 0 || matrix[0].length < 5) return matrix; + const _rows = matrix.length; + const cols = matrix[0].length; + const counts = Array.from({ length: cols }, (_, c) => + matrix.reduce((n, row) => n + (row[c].trim().length > 0 ? 1 : 0), 0), + ); + const denseCols = new Set( + counts + .map((count, col) => ({ count, col })) + .filter(({ col, count }) => col === 0 || count >= 2) + .map(({ col }) => col), + ); + const sparseCols = counts + .map((count, col) => ({ count, col })) + .filter(({ col, count }) => col > 0 && col < cols - 1 && count === 1) + .map(({ col }) => col); + if (sparseCols.length < 2 || denseCols.size < 4) return matrix; + const moves: Array<{ from: number; to: number; row: number }> = []; + for (const from of sparseCols) { + const row = matrix.findIndex(r => r[from].trim().length > 0); + const to = from + 1; + if (row < 0) return matrix; + if (!denseCols.has(to)) return matrix; + if (matrix[row][to].trim().length > 0) return matrix; + moves.push({ from, to, row }); + } + const copy = matrix.map(row => [...row]); + for (const { from, to, row } of moves) { + copy[row][to] = copy[row][to].trim().length > 0 ? `${copy[row][to]} ${copy[row][from]}` : copy[row][from]; + copy[row][from] = ""; + } + const keepCols = Array.from({ length: cols }, (_, c) => c).filter(c => copy.some(row => row[c].trim().length > 0)); + if (keepCols.length === cols) return copy; + return copy.map(row => keepCols.map(c => row[c])); +} + +/** + * When a data row has ≥2 parenthesized qualifiers in non-first columns + * (and the first column is empty), promote them into the header row. + */ +function promoteSubHeaderPrefixes(matrix: string[][]): string[][] { + if (matrix.length < 2) return matrix; + const PAREN_RE = /^\([^)]{1,40}\)$/; + const result = matrix.map(row => [...row]); + const cols = matrix[0].length; + const rowsToRemove = new Set(); + for (let r = 1; r < result.length; r++) { + if (rowsToRemove.has(r)) continue; + const promotable: Array<{ col: number; prefix: string; isFullCell: boolean }> = []; + for (let col = 1; col < cols; col++) { + const cell = (result[r][col] ?? "").trim(); + if (!cell) continue; + const parts = cell.split("
"); + if (parts.length === 1 && PAREN_RE.test(cell)) { + promotable.push({ col, prefix: cell, isFullCell: true }); + } else if (parts.length >= 2 && PAREN_RE.test(parts[0].trim())) { + promotable.push({ + col, + prefix: parts[0].trim(), + isFullCell: false, + }); + } + } + if (promotable.length < 2) continue; + if (promotable.some(p => p.isFullCell) && result[r][0].trim().length > 0) continue; + for (const { col, prefix, isFullCell } of promotable) { + result[0][col] = result[0][col].trim() ? `${result[0][col]} ${prefix}` : prefix; + if (isFullCell) { + result[r][col] = ""; + } else { + const parts = result[r][col].split("
"); + result[r][col] = parts.slice(1).join("
"); + } + } + if (result[r].every(cell => cell.trim().length === 0)) { + rowsToRemove.add(r); + } + } + return result.filter((_, r) => !rowsToRemove.has(r)); +} + +// --------------------------------------------------------------------------- +// Free text rendering +// --------------------------------------------------------------------------- +/** Y tolerance for grouping text boxes onto the same visual line. */ +const TEXT_LINE_Y_TOLERANCE = 3; +/** Minimum X gap between adjacent boxes to mark line as tabular. */ +const TABULAR_X_GAP = 30; +/** + * Minimum font size (pts) to consider when computing the modal body font. + * Tiny labels from diagrams, footnote markers, and superscripts are excluded + * so they don't skew the modal toward small sizes. + */ +const MIN_BODY_FONT_SIZE = 7; + +/** + * Compute the most frequent font size among text boxes, ignoring very small + * text that likely comes from diagrams, footnotes, or superscripts. + */ +function modalFontSize(textBoxes: TextBox[]): number { + const counts = new Map(); + for (const tb of textBoxes) { + const size = Math.round((tb.fontSize ?? 0) * 10) / 10; + if (size < MIN_BODY_FONT_SIZE) continue; + counts.set(size, (counts.get(size) ?? 0) + 1); + } + let modal = 0; + let maxCount = 0; + for (const [size, count] of counts) { + if (count > maxCount) { + maxCount = count; + modal = size; + } + } + return modal; +} + +/** Group free text boxes into horizontal lines, sorted top-to-bottom. */ +function groupFreeTextIntoLines(textBoxes: TextBox[]): RenderLine[] { + if (textBoxes.length === 0) return []; + const sorted = [...textBoxes].sort((a, b) => { + const ya = (a.bounds.top + a.bounds.bottom) / 2; + const yb = (b.bounds.top + b.bounds.bottom) / 2; + const dy = yb - ya; + if (Math.abs(dy) > TEXT_LINE_Y_TOLERANCE) return dy; + return a.bounds.left - b.bounds.left; + }); + const lines: RenderLine[] = []; + let curParts = [sorted[0].text]; + let curBoxes = [sorted[0]]; + let curY = (sorted[0].bounds.top + sorted[0].bounds.bottom) / 2; + let curTopY = curY; + let curFontSize = sorted[0].fontSize; + let curIsBold = sorted[0].isBold; + const finishLine = () => { + let isTabular = false; + for (let j = 1; j < curBoxes.length; j++) { + if (curBoxes[j].bounds.left - curBoxes[j - 1].bounds.right > TABULAR_X_GAP) { + isTabular = true; + break; + } + } + lines.push({ + text: curParts.join(" "), + topY: curTopY, + fontSize: curFontSize, + isBold: curIsBold, + isTabular, + }); + }; + for (let i = 1; i < sorted.length; i++) { + const box = sorted[i]; + const cy = (box.bounds.top + box.bounds.bottom) / 2; + if (Math.abs(cy - curY) <= TEXT_LINE_Y_TOLERANCE) { + curParts.push(box.text); + curBoxes.push(box); + curFontSize = Math.max(curFontSize, box.fontSize); + curIsBold = curIsBold || box.isBold; + } else { + finishLine(); + curParts = [box.text]; + curBoxes = [box]; + curY = cy; + curTopY = cy; + curFontSize = box.fontSize; + curIsBold = box.isBold; + } + } + finishLine(); + return lines; +} + +/** Determine markdown heading prefix based on font size relative to body. */ +function headingPrefix(fontSize: number, bodyFontSize: number, isBold: boolean): string { + if (bodyFontSize <= 0) return ""; + const ratio = fontSize / bodyFontSize; + // Large headings (>2x body size) + if (ratio >= 2.0) return "# "; + // Medium headings (~1.5x body size) + if (ratio >= 1.4) return "## "; + // Small headings (bold and slightly larger) + if (ratio >= 1.1 && isBold) return "### "; + return ""; +} + +// --------------------------------------------------------------------------- +// Block merging +// --------------------------------------------------------------------------- +/** Merge consecutive blocks with the same heading prefix (wrapped headings). */ +function mergeConsecutiveHeadings(blocks: ContentBlock[], bodyFS: number): ContentBlock[] { + if (blocks.length === 0) return []; + const HEADING_RE = /^(#{1,6} )/; + const maxGap = Math.max(bodyFS * 3, 30); + const merged: ContentBlock[] = []; + let cur: ContentBlock = { ...blocks[0] }; + for (let i = 1; i < blocks.length; i++) { + const next = blocks[i]; + const curMatch = cur.content.match(HEADING_RE); + const nextMatch = next.content.match(HEADING_RE); + const gap = cur.topY - next.topY; + if (curMatch && nextMatch && curMatch[1] === nextMatch[1] && gap <= maxGap) { + cur = { + topY: cur.topY, + content: `${cur.content} ${next.content.slice(nextMatch[1].length)}`, + isTabular: cur.isTabular || next.isTabular, + }; + } else { + merged.push(cur); + cur = { ...next }; + } + } + merged.push(cur); + return merged; +} + +/** + * Merge consecutive plain-text blocks that are wrapped lines of the same paragraph. + */ +function mergeParagraphWraps(blocks: ContentBlock[], bodyFS: number): ContentBlock[] { + if (blocks.length === 0 || bodyFS <= 0) return blocks; + const HEADING_RE = /^#{1,6} /; + const SENTENCE_END_RE = /[.!?…)\]]\s*$/; + const maxGap = bodyFS * 2.0; + const MIN_WRAP_LENGTH = 25; + const merged: ContentBlock[] = []; + let cur: WrapBlock = { ...blocks[0], lastTopY: blocks[0].topY }; + for (let i = 1; i < blocks.length; i++) { + const next = blocks[i]; + const curIsBody = !HEADING_RE.test(cur.content) && !cur.content.startsWith("|"); + const nextIsBody = !HEADING_RE.test(next.content) && !next.content.startsWith("|"); + const gap = cur.lastTopY - next.topY; + const isWrap = + curIsBody && + nextIsBody && + !cur.isTabular && + !next.isTabular && + gap > 0 && + gap <= maxGap && + cur.content.length > MIN_WRAP_LENGTH && + !SENTENCE_END_RE.test(cur.content); + if (isWrap) { + cur = { + topY: cur.topY, + lastTopY: next.topY, + content: `${cur.content.trimEnd()} ${next.content.trimStart()}`, + isTabular: false, + }; + } else { + merged.push({ topY: cur.topY, content: cur.content }); + cur = { ...next, lastTopY: next.topY }; + } + } + merged.push({ topY: cur.topY, content: cur.content }); + return merged; +} + +/** Remove page number blocks near the bottom of the page. */ +function removePageNumbers(blocks: ContentBlock[]): ContentBlock[] { + const PAGE_NUM_RE = /^(?:#{1,6}\s*)?\d+\s*$/; + const BOTTOM_Y = 120; + return blocks.filter((block, idx) => { + const isBottom = idx >= blocks.length - 3; + const isLowY = block.topY <= BOTTOM_Y; + const isPageNum = PAGE_NUM_RE.test(block.content.trim()); + return !(isBottom && isLowY && isPageNum); + }); +} + +// --------------------------------------------------------------------------- +// Detached first-column table reconstruction +// --------------------------------------------------------------------------- +/** + * Fix tables where the first column was emitted as free text blocks + * around a markdown table containing only the right-side columns. + * + * Detects: a plain-text header line with (N+1) tokens above an N-column + * markdown table, plus short label lines whose count matches the table's + * logical row count. Reconstructs into a proper (N+1)-column table. + */ +function normalizeDetachedFirstColumnTables(blocks: ContentBlock[]): ContentBlock[] { + const HEADING_RE = /^#{1,6}\s/; + const isTableBlock = (text: string) => text.trimStart().startsWith("|"); + const isPlainBlock = (text: string) => !HEADING_RE.test(text) && !isTableBlock(text); + const isShortLabel = (text: string) => { + const t = text.trim(); + return t.length > 0 && t.length <= 40; + }; + const splitTokens = (text: string) => + text + .trim() + .split(/[ \t]+/) + .filter(Boolean); + const replacements = new Map(); + const remove = new Set(); + for (let tableIdx = 0; tableIdx < blocks.length; tableIdx++) { + if (remove.has(tableIdx)) continue; + const tableBlock = blocks[tableIdx]; + if (!isTableBlock(tableBlock.content)) continue; + const tableLines = tableBlock.content + .split("\n") + .map(line => line.trim()) + .filter(line => line.startsWith("|")); + const dataRows = tableLines + .filter(line => !/^\|\s*[-: ]+\|/.test(line)) + .map(parsePipeRow) + .filter(row => row.length > 0); + if (dataRows.length === 0) continue; + const cols = dataRows[0].length; + if (cols < 2 || dataRows.some(row => row.length !== cols)) continue; + // Expand by
count to get logical row count + const logicalRows: string[][] = []; + for (const row of dataRows) { + const splitCells = row.map(cell => cell.split("
").map(p => p.trim())); + const rowSpan = Math.max(...splitCells.map(parts => parts.length)); + for (let k = 0; k < rowSpan; k++) { + logicalRows.push(splitCells.map(parts => parts[k] ?? "")); + } + } + if (logicalRows.length < 2) continue; + // Find header with (cols + 1) non-numeric tokens + let headerIdx = -1; + let headerTokens: string[] = []; + for (let i = Math.max(0, tableIdx - 4); i <= tableIdx - 1; i++) { + const text = normalizeFullWidthAscii(blocks[i].content).trim(); + if (!isPlainBlock(text)) continue; + const tokens = splitTokens(text); + if (tokens.length === cols + 1 && tokens.every(tok => !/[0-9]/.test(tok))) { + headerIdx = i; + headerTokens = tokens; + } + } + if (headerIdx < 0) continue; + // Collect short label lines above/below table + const aboveLabels: Array<{ idx: number; text: string }> = []; + for (let i = tableIdx - 1; i > headerIdx; i--) { + const text = normalizeFullWidthAscii(blocks[i].content).trim(); + if (!isPlainBlock(text) || !isShortLabel(text)) break; + aboveLabels.push({ idx: i, text }); + } + aboveLabels.reverse(); + const belowLabels: Array<{ idx: number; text: string }> = []; + for (let i = tableIdx + 1; i < blocks.length; i++) { + const text = normalizeFullWidthAscii(blocks[i].content).trim(); + if (!isPlainBlock(text) || !isShortLabel(text)) break; + belowLabels.push({ idx: i, text }); + } + const labels = [...aboveLabels, ...belowLabels]; + if (labels.length !== logicalRows.length) continue; + // Reconstruct the full table + const normalizedLines: string[] = []; + normalizedLines.push(`| ${headerTokens.join(" | ")} |`); + normalizedLines.push(`| ${Array.from({ length: cols + 1 }, () => "---").join(" | ")} |`); + for (let r = 0; r < logicalRows.length; r++) { + normalizedLines.push(`| ${labels[r].text} | ${logicalRows[r].join(" | ")} |`); + } + replacements.set(tableIdx, normalizedLines.join("\n")); + remove.add(headerIdx); + for (const label of labels) remove.add(label.idx); + } + if (replacements.size === 0 && remove.size === 0) return blocks; + const out: ContentBlock[] = []; + for (let i = 0; i < blocks.length; i++) { + if (remove.has(i)) continue; + const replaced = replacements.get(i); + if (replaced) { + out.push({ topY: blocks[i].topY, content: replaced }); + } else { + out.push(blocks[i]); + } + } + return out; +} + +// --------------------------------------------------------------------------- +// Public API +// --------------------------------------------------------------------------- +/** + * Render one page's content: free text and tables interleaved top-to-bottom. + */ +export function renderPageContent( + freeTextBoxes: TextBox[], + tables: TableGrid[], + imageBlocks: Array<{ topY: number; markdown: string }> = [], + allTextBoxes?: TextBox[], +): string { + const blocks: ContentBlock[] = []; + // Use ALL text boxes (before table/diagram filtering) for modal font size, + // so that diagram labels released as free text don't skew the body size. + const bodyFS = modalFontSize(allTextBoxes ?? freeTextBoxes); + // Free text lines + for (const line of groupFreeTextIntoLines(freeTextBoxes)) { + const prefix = headingPrefix(line.fontSize, bodyFS, line.isBold); + blocks.push({ + topY: line.topY, + content: prefix + line.text, + isTabular: prefix === "" && line.isTabular, + }); + } + // Tables + for (const table of tables) { + const md = renderTableToMarkdown(table); + if (md.length > 0) { + blocks.push({ topY: table.topY, content: md }); + } + } + // Images + for (const img of imageBlocks) { + blocks.push({ topY: img.topY, content: img.markdown }); + } + // Sort top-to-bottom (higher Y = higher on page = comes first) + blocks.sort((a, b) => b.topY - a.topY); + const cleaned = removePageNumbers(blocks); + const headingsMerged = mergeConsecutiveHeadings(cleaned, bodyFS); + const merged = mergeParagraphWraps(headingsMerged, bodyFS); + const normalized = normalizeDetachedFirstColumnTables(merged); + return normalized + .map(b => b.content) + .join("\n\n") + .trim(); +} diff --git a/packages/coding-agent/src/markit/converters/pdf/types.ts b/packages/coding-agent/src/markit/converters/pdf/types.ts new file mode 100644 index 000000000..4168adf45 --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pdf/types.ts @@ -0,0 +1,84 @@ +// Adapted from markit-ai (MIT). See ../../NOTICE. + +/** Bounding box in PDF coordinate space (origin = bottom-left). */ +export type Bounds = { + left: number; + right: number; + /** Higher value = higher on the page. */ + top: number; + bottom: number; +}; + +/** A text fragment with position and font metadata. */ +export type TextBox = { + id: string; + text: string; + bounds: Bounds; + pageNumber: number; + /** Dominant font size in points. */ + fontSize: number; + /** True if rendered bold (font name or rendering mode). */ + isBold: boolean; +}; + +/** A horizontal or vertical line segment extracted from vector graphics. */ +export type Segment = { + id: string; + x1: number; + y1: number; + x2: number; + y2: number; +}; + +/** A single cell in a resolved table grid. */ +export type TableCell = { + row: number; + col: number; + text: string; + rowSpan: number; + colSpan: number; +}; + +/** A resolved table grid ready for markdown rendering. */ +export type TableGrid = { + pageNumber: number; + rows: number; + cols: number; + cells: TableCell[]; + warnings: string[]; + /** Top Y coordinate (PDF space: larger = higher on page). */ + topY: number; + /** True for tables detected without vector borders. */ + isBorderless: boolean; +}; + +/** An image/diagram region detected on a page. */ +export type ImageRegion = { + id: string; + pageNumber: number; + /** Bounding box in mupdf coordinates (top-left origin). */ + bbox: { + x: number; + y: number; + w: number; + h: number; + }; + /** Y position in PDF coordinates (bottom-left) for ordering. */ + topY: number; +}; + +/** Result of extracting content from a single PDF page. */ +export type PageContent = { + pageNumber: number; + textBoxes: TextBox[]; + segments: Segment[]; + images: ImageRegion[]; +}; + +/** A block of rendered content (text paragraph or table). */ +export type ContentBlock = { + topY: number; + content: string; + /** True if this line has wide gaps between text boxes (column headers). */ + isTabular?: boolean; +}; diff --git a/packages/coding-agent/src/markit/converters/pptx.ts b/packages/coding-agent/src/markit/converters/pptx.ts new file mode 100644 index 000000000..996da6248 --- /dev/null +++ b/packages/coding-agent/src/markit/converters/pptx.ts @@ -0,0 +1,325 @@ +// Adapted from markit-ai (MIT). See ../NOTICE. +import * as path from "node:path"; +import { XMLParser } from "fast-xml-parser"; +import { unzip, unzipText } from "../../utils/zip"; +import type { ConversionResult, Converter, StreamInfo } from "../types"; + +const EXTENSIONS = [".pptx"]; +const MIMETYPES = ["application/vnd.openxmlformats-officedocument.presentationml.presentation"]; + +/** A text value: bare string/number, or a `{ "#text" }` node when the element carries attributes. */ +type XmlText = string | number | { "#text"?: string }; + +interface TextRun { + "a:t"?: XmlText; +} +interface Paragraph { + "a:r"?: TextRun | TextRun[]; +} +interface TextBody { + "a:p"?: Paragraph | Paragraph[]; +} +interface CNvPr { + "@_name": string; +} +interface Placeholder { + "@_type": string; +} +interface NvPr { + "p:ph"?: Placeholder; +} +interface NvSpPr { + "p:cNvPr"?: CNvPr; + "p:nvPr"?: NvPr; +} +interface NvPicPr { + "p:cNvPr"?: CNvPr; +} +interface Shape { + "p:txBody"?: TextBody; + "p:nvSpPr"?: NvSpPr; +} +interface Blip { + "@_r:embed": string; +} +interface BlipFill { + "a:blip"?: Blip; +} +interface Picture { + "p:blipFill"?: BlipFill; + "p:nvSpPr"?: NvSpPr; + "p:nvPicPr"?: NvPicPr; +} +interface TableCell { + "a:txBody"?: TextBody; +} +interface TableRow { + "a:tc"?: TableCell | TableCell[]; +} +interface Table { + "a:tr"?: TableRow | TableRow[]; +} +interface GraphicData { + "a:tbl"?: Table; +} +interface Graphic { + "a:graphicData"?: GraphicData; +} +interface GraphicFrame { + "a:graphic"?: Graphic; +} +interface SpTree { + "p:sp"?: Shape | Shape[]; + "p:pic"?: Picture | Picture[]; + "p:graphicFrame"?: GraphicFrame | GraphicFrame[]; +} +interface CSld { + "p:spTree"?: SpTree; +} +interface SlideDoc { + "p:sld"?: { "p:cSld"?: CSld }; +} +interface NotesDoc { + "p:notes"?: { "p:cSld"?: CSld }; +} +interface SldId { + "@_r:id": string; +} +interface PresentationDoc { + "p:presentation"?: { "p:sldIdLst"?: { "p:sldId"?: SldId | SldId[] } }; +} +interface Relationship { + "@_Id": string; + "@_Target": string; +} +interface RelationshipsDoc { + Relationships?: { Relationship?: Relationship | Relationship[] }; +} + +export class PptxConverter implements Converter { + name = "pptx"; + + accepts(streamInfo: StreamInfo): boolean { + if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true; + if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true; + return false; + } + + async convert(input: Buffer, streamInfo: StreamInfo): Promise { + const entries = unzip(input); + const parser = new XMLParser({ + ignoreAttributes: false, + attributeNamePrefix: "@_", + textNodeName: "#text", + processEntities: { maxTotalExpansions: 1_000_000 }, + }); + // Get slide order from presentation.xml + const presXml = unzipText(entries, "ppt/presentation.xml"); + if (!presXml) throw new Error("Invalid PPTX: missing presentation.xml"); + const pres = parser.parse(presXml) as PresentationDoc; + const sldIdList = pres["p:presentation"]?.["p:sldIdLst"]?.["p:sldId"]; + const sldIds = Array.isArray(sldIdList) ? sldIdList : sldIdList ? [sldIdList] : []; + // Get relationship mappings + const relsXml = unzipText(entries, "ppt/_rels/presentation.xml.rels"); + const rels = relsXml ? (parser.parse(relsXml) as RelationshipsDoc) : null; + const relList = rels?.Relationships?.Relationship; + const relArray = Array.isArray(relList) ? relList : relList ? [relList] : []; + const relMap = new Map(); + for (const r of relArray) { + relMap.set(r["@_Id"], r["@_Target"]); + } + // Map slide IDs to file paths in order + const slidePaths: string[] = []; + for (const sld of sldIds) { + const rId = sld["@_r:id"]; + const target = relMap.get(rId); + if (target) slidePaths.push(`ppt/${target}`); + } + // If we couldn't resolve from rels, fall back to finding slide files + if (slidePaths.length === 0) { + const slideFiles = Object.keys(entries) + .filter(f => /^ppt\/slides\/slide\d+\.xml$/.test(f)) + .sort((a, b) => { + const na = parseInt(a.match(/slide(\d+)/)?.[1] || "0", 10); + const nb = parseInt(b.match(/slide(\d+)/)?.[1] || "0", 10); + return na - nb; + }); + slidePaths.push(...slideFiles); + } + const imageDir = streamInfo.imageDir; + const sections: string[] = []; + let imageCount = 0; + for (let i = 0; i < slidePaths.length; i++) { + const slideXml = unzipText(entries, slidePaths[i]); + if (!slideXml) continue; + const slide = parser.parse(slideXml) as SlideDoc; + const spTree = slide["p:sld"]?.["p:cSld"]?.["p:spTree"]; + if (!spTree) continue; + // Parse slide-level rels for image references + const slideRelsPath = `${slidePaths[i].replace("slides/slide", "slides/_rels/slide")}.rels`; + const slideRelsXml = unzipText(entries, slideRelsPath); + const slideRelMap = new Map(); + if (slideRelsXml) { + const slideRels = parser.parse(slideRelsXml) as RelationshipsDoc; + const relItems = toList(slideRels?.Relationships?.Relationship); + for (const r of relItems) { + slideRelMap.set(r["@_Id"], r["@_Target"]); + } + } + const slideLines = [``]; + const shapes = spTree["p:sp"]; + const shapeList = Array.isArray(shapes) ? shapes : shapes ? [shapes] : []; + let isTitle = true; + for (const shape of shapeList) { + const text = this.extractText(shape); + if (!text) continue; + if (isTitle) { + slideLines.push(`# ${text}`); + isTitle = false; + } else { + slideLines.push(text); + } + } + // Extract embedded images + const pics = toList(spTree["p:pic"]); + for (const pic of pics) { + const blipFill = pic["p:blipFill"]; + const rEmbed = blipFill?.["a:blip"]?.["@_r:embed"]; + if (!rEmbed) continue; + const target = slideRelMap.get(rEmbed); + if (!target) continue; + // Resolve relative target against slide directory + const imagePath = target.startsWith("/") ? target.slice(1) : `ppt/slides/${target}`; + // Normalize path (e.g. ppt/slides/../media/image1.png → ppt/media/image1.png) + const normalizedPath = imagePath + .split("/") + .reduce((parts, seg) => { + if (seg === "..") parts.pop(); + else parts.push(seg); + return parts; + }, []) + .join("/"); + const buf = entries[normalizedPath]; + if (!buf) continue; + imageCount++; + const name = + pic["p:nvSpPr"]?.["p:cNvPr"]?.["@_name"] || + pic["p:nvPicPr"]?.["p:cNvPr"]?.["@_name"] || + `image_${imageCount}`; + if (imageDir) { + try { + const ext = normalizedPath.split(".").pop() || "png"; + const filename = `slide${i + 1}_${imageCount}.${ext}`; + const filepath = path.join(imageDir, filename); + await Bun.write(filepath, buf); + slideLines.push(`![${name}](${filepath})`); + } catch { + slideLines.push(``); + } + } else { + slideLines.push(``); + } + } + // Tables + const graphicFrames = spTree["p:graphicFrame"]; + const gfList = Array.isArray(graphicFrames) ? graphicFrames : graphicFrames ? [graphicFrames] : []; + for (const gf of gfList) { + const table = this.extractTable(gf); + if (table) slideLines.push(table); + } + // Slide notes + const noteFile = slidePaths[i].replace("slides/slide", "notesSlides/notesSlide"); + const noteXml = unzipText(entries, noteFile); + if (noteXml) { + const note = parser.parse(noteXml) as NotesDoc; + const noteSpTree = note["p:notes"]?.["p:cSld"]?.["p:spTree"]; + if (noteSpTree) { + const noteShapes = noteSpTree["p:sp"]; + const noteList = Array.isArray(noteShapes) ? noteShapes : noteShapes ? [noteShapes] : []; + const noteTexts: string[] = []; + for (const ns of noteList) { + // Skip slide image placeholder + const phType = ns["p:nvSpPr"]?.["p:nvPr"]?.["p:ph"]?.["@_type"]; + if (phType === "sldImg") continue; + const t = this.extractText(ns); + if (t) noteTexts.push(t); + } + if (noteTexts.length > 0) { + slideLines.push("\n### Notes:"); + slideLines.push(noteTexts.join("\n")); + } + } + } + sections.push(slideLines.join("\n")); + } + return { markdown: sections.join("\n\n").trim() }; + } + + extractText(shape: Shape): string { + const txBody = shape["p:txBody"]; + if (!txBody) return ""; + const paragraphs = txBody["a:p"]; + const pList = Array.isArray(paragraphs) ? paragraphs : paragraphs ? [paragraphs] : []; + const lines: string[] = []; + for (const p of pList) { + const runs = p["a:r"]; + const rList = Array.isArray(runs) ? runs : runs ? [runs] : []; + const parts: string[] = []; + for (const r of rList) { + const t = r["a:t"]; + if (t != null) parts.push(typeof t === "object" ? t["#text"] || "" : String(t)); + } + if (parts.length > 0) lines.push(parts.join("")); + } + return lines.join("\n").trim(); + } + + extractTable(gf: GraphicFrame): string | null { + const tbl = gf?.["a:graphic"]?.["a:graphicData"]?.["a:tbl"]; + if (!tbl) return null; + const rows = tbl["a:tr"]; + const rowList = Array.isArray(rows) ? rows : rows ? [rows] : []; + if (rowList.length === 0) return null; + const mdRows: string[][] = []; + for (const row of rowList) { + const cells = row["a:tc"]; + const cellList = Array.isArray(cells) ? cells : cells ? [cells] : []; + const cellTexts: string[] = []; + for (const cell of cellList) { + const txBody = cell["a:txBody"]; + if (!txBody) { + cellTexts.push(""); + continue; + } + const paragraphs = txBody["a:p"]; + const pList = Array.isArray(paragraphs) ? paragraphs : paragraphs ? [paragraphs] : []; + const parts: string[] = []; + for (const p of pList) { + const runs = p["a:r"]; + const rList = Array.isArray(runs) ? runs : runs ? [runs] : []; + for (const r of rList) { + const t = r["a:t"]; + if (t != null) parts.push(typeof t === "object" ? t["#text"] || "" : String(t)); + } + } + cellTexts.push(parts.join(" ")); + } + mdRows.push(cellTexts); + } + if (mdRows.length === 0) return null; + const [header, ...body] = mdRows; + const lines: string[] = []; + lines.push(`| ${header.join(" | ")} |`); + lines.push(`| ${header.map(() => "---").join(" | ")} |`); + for (const row of body) { + while (row.length < header.length) row.push(""); + lines.push(`| ${row.join(" | ")} |`); + } + return lines.join("\n"); + } +} + +function toList(val: T | T[] | undefined): T[] { + if (!val) return []; + return Array.isArray(val) ? val : [val]; +} diff --git a/packages/coding-agent/src/markit/converters/xlsx.ts b/packages/coding-agent/src/markit/converters/xlsx.ts new file mode 100644 index 000000000..5bfd50fda --- /dev/null +++ b/packages/coding-agent/src/markit/converters/xlsx.ts @@ -0,0 +1,173 @@ +// Adapted from markit-ai (MIT). See ../NOTICE. +import { XMLParser } from "fast-xml-parser"; +import { unzip, unzipText } from "../../utils/zip"; +import type { ConversionResult, Converter, StreamInfo } from "../types"; + +const EXTENSIONS = [".xlsx"]; +const MIMETYPES = ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"]; + +/** A text value: bare string/number, or a `{ "#text" }` node when the element carries attributes. */ +type XmlText = string | number | { "#text"?: string }; + +interface RichTextRun { + t?: XmlText; +} +interface StringItem { + t?: XmlText; + r?: RichTextRun | RichTextRun[]; +} +interface Cell { + "@_t"?: string; + v?: string | number; + is?: StringItem; +} +interface Row { + c?: Cell | Cell[]; +} +interface WorksheetDoc { + worksheet?: { sheetData?: { row?: Row | Row[] } }; +} +interface Sheet { + "@_name": string; + "@_r:id": string; +} +interface WorkbookDoc { + workbook?: { sheets?: { sheet?: Sheet | Sheet[] } }; +} +interface SharedStringsDoc { + sst?: { si?: StringItem | StringItem[] }; +} +interface Relationship { + "@_Id": string; + "@_Target": string; +} +interface RelationshipsDoc { + Relationships?: { Relationship?: Relationship | Relationship[] }; +} + +export class XlsxConverter implements Converter { + name = "xlsx"; + + accepts(streamInfo: StreamInfo): boolean { + if (streamInfo.extension && EXTENSIONS.includes(streamInfo.extension)) return true; + if (streamInfo.mimetype && MIMETYPES.some(m => streamInfo.mimetype?.startsWith(m))) return true; + return false; + } + + async convert(input: Buffer, _streamInfo: StreamInfo): Promise { + const entries = unzip(input); + const parser = new XMLParser({ + ignoreAttributes: false, + attributeNamePrefix: "@_", + textNodeName: "#text", + processEntities: { maxTotalExpansions: 1_000_000 }, + }); + // Parse shared strings + const ssXml = unzipText(entries, "xl/sharedStrings.xml"); + const ss = ssXml ? (parser.parse(ssXml) as SharedStringsDoc) : null; + const siList = ss?.sst?.si; + const shared = toArray(siList); + // Parse workbook for sheet names + const wbXml = unzipText(entries, "xl/workbook.xml"); + if (!wbXml) throw new Error("Invalid XLSX: missing workbook.xml"); + const wb = parser.parse(wbXml) as WorkbookDoc; + const sheets = toArray(wb.workbook?.sheets?.sheet); + // Parse workbook rels to map rIds to sheet files + const relsXml = unzipText(entries, "xl/_rels/workbook.xml.rels"); + const rels = relsXml ? (parser.parse(relsXml) as RelationshipsDoc) : null; + const relList = toArray(rels?.Relationships?.Relationship); + const relMap = new Map(); + for (const r of relList) { + relMap.set(r["@_Id"], r["@_Target"]); + } + const sections: string[] = []; + for (const sheet of sheets) { + const sheetName = sheet["@_name"]; + const rId = sheet["@_r:id"]; + const target = relMap.get(rId); + if (!target) continue; + const sheetPath = target.startsWith("/") ? target.slice(1) : `xl/${target}`; + const sheetXml = unzipText(entries, sheetPath); + if (!sheetXml) continue; + const parsed = parser.parse(sheetXml) as WorksheetDoc; + const rows = toArray(parsed.worksheet?.sheetData?.row); + if (rows.length === 0) continue; + // Extract all rows as string arrays + const tableRows: string[][] = []; + for (const row of rows) { + const cells = toArray(row.c); + const values: string[] = []; + for (const cell of cells) { + values.push(this.getCellValue(cell, shared)); + } + tableRows.push(values); + } + if (tableRows.length === 0) continue; + // Normalize column count + const maxCols = Math.max(...tableRows.map(r => r.length)); + for (const row of tableRows) { + while (row.length < maxCols) row.push(""); + } + sections.push(`## ${sheetName}`); + const [header, ...body] = tableRows; + const lines: string[] = []; + lines.push(`| ${header.join(" | ")} |`); + lines.push(`| ${header.map(() => "---").join(" | ")} |`); + for (const row of body) { + lines.push(`| ${row.join(" | ")} |`); + } + sections.push(lines.join("\n")); + } + return { markdown: sections.join("\n\n") }; + } + + getCellValue(cell: Cell, shared: StringItem[]): string { + // Shared string + if (cell["@_t"] === "s") { + return this.getSharedString(shared, Number(cell.v)); + } + // Inline string + if (cell["@_t"] === "inlineStr") { + const is = cell.is; + if (!is) return ""; + if (is.t != null) return textValue(is.t); + if (is.r) + return toArray(is.r) + .map(r => textValue(r.t)) + .join(""); + return ""; + } + // Boolean + if (cell["@_t"] === "b") { + return cell.v === 1 || cell.v === "1" ? "TRUE" : "FALSE"; + } + // Number or formula result + if (cell.v != null) return String(cell.v); + return ""; + } + + getSharedString(shared: StringItem[], idx: number): string { + const si = shared[idx]; + if (!si) return ""; + // Simple text + if (si.t != null) return textValue(si.t); + // Rich text runs + if (si.r) { + return toArray(si.r) + .map(r => textValue(r.t)) + .join(""); + } + return ""; + } +} + +function textValue(t: XmlText | undefined): string { + if (t == null) return ""; + if (typeof t === "object") return t["#text"] || ""; + return String(t); +} + +function toArray(val: T | T[] | undefined): T[] { + if (!val) return []; + return Array.isArray(val) ? val : [val]; +} diff --git a/packages/coding-agent/src/markit/index.ts b/packages/coding-agent/src/markit/index.ts new file mode 100644 index 000000000..558bcaf53 --- /dev/null +++ b/packages/coding-agent/src/markit/index.ts @@ -0,0 +1,2 @@ +export * from "./registry"; +export * from "./types"; diff --git a/packages/coding-agent/src/markit/registry.ts b/packages/coding-agent/src/markit/registry.ts new file mode 100644 index 000000000..1fa20f55a --- /dev/null +++ b/packages/coding-agent/src/markit/registry.ts @@ -0,0 +1,59 @@ +// Adapted from markit-ai (MIT). See ./NOTICE. +import * as path from "node:path"; +import { DocxConverter } from "./converters/docx"; +import { EpubConverter } from "./converters/epub"; +import { PdfConverter } from "./converters/pdf"; +import { PptxConverter } from "./converters/pptx"; +import { XlsxConverter } from "./converters/xlsx"; +import type { ConversionResult, Converter, MarkitOptions, StreamInfo } from "./types"; + +/** + * In-house document → markdown engine (replaces the `markit-ai` package). + * + * Only the document converters omp routes are registered (pdf, docx, pptx, + * xlsx, epub). The first converter whose `accepts()` returns true and whose + * `convert()` succeeds wins. + */ +export class Markit { + readonly #converters: readonly Converter[]; + readonly #options: MarkitOptions; + + constructor(options: MarkitOptions = {}) { + this.#options = options; + this.#converters = [ + new PdfConverter(), + new DocxConverter(), + new PptxConverter(), + new XlsxConverter(), + new EpubConverter(), + ]; + } + + async convertFile(filePath: string, extra?: { imageDir?: string }): Promise { + const buffer = Buffer.from(await Bun.file(filePath).arrayBuffer()); + const streamInfo: StreamInfo = { + localPath: filePath, + extension: path.extname(filePath).toLowerCase(), + filename: path.basename(filePath), + ...extra, + }; + return this.convert(buffer, streamInfo); + } + + async convert(input: Buffer, streamInfo: StreamInfo): Promise { + const errors: { converter: string; error: Error }[] = []; + for (const converter of this.#converters) { + if (!converter.accepts(streamInfo)) continue; + try { + return await converter.convert(input, streamInfo, this.#options); + } catch (err) { + errors.push({ converter: converter.name, error: err instanceof Error ? err : new Error(String(err)) }); + } + } + if (errors.length > 0) { + const details = errors.map(e => ` ${e.converter}: ${e.error.message}`).join("\n"); + throw new Error(`Conversion failed:\n${details}`); + } + throw new Error(`Unsupported format: ${streamInfo.extension || streamInfo.mimetype || "unknown"}`); + } +} diff --git a/packages/coding-agent/src/markit/types.ts b/packages/coding-agent/src/markit/types.ts new file mode 100644 index 000000000..1b9f6c8f7 --- /dev/null +++ b/packages/coding-agent/src/markit/types.ts @@ -0,0 +1,35 @@ +// Adapted from markit-ai (MIT). See ./NOTICE. + +export interface StreamInfo { + mimetype?: string; + extension?: string; + charset?: string; + filename?: string; + localPath?: string; + url?: string; + /** Directory to write extracted images/diagrams. */ + imageDir?: string; +} + +export interface ConversionResult { + markdown: string; + title?: string; +} + +export interface MarkitOptions { + /** Describe an image, return markdown. Receives raw bytes and mimetype. */ + describe?: (image: Buffer, mimetype: string) => Promise; + /** Transcribe audio, return text. Receives raw bytes and mimetype. */ + transcribe?: (audio: Buffer, mimetype: string) => Promise; + /** Extra instructions appended to the image description prompt. */ + prompt?: string; +} + +export interface Converter { + /** Human-readable name for error messages. */ + name: string; + /** Quick check: can this converter handle the given stream? */ + accepts(streamInfo: StreamInfo): boolean; + /** Convert the source to markdown. */ + convert(input: Buffer, streamInfo: StreamInfo, options?: MarkitOptions): Promise; +} diff --git a/packages/coding-agent/src/tools/archive-reader.ts b/packages/coding-agent/src/tools/archive-reader.ts index b004e6af5..5b44c855a 100644 --- a/packages/coding-agent/src/tools/archive-reader.ts +++ b/packages/coding-agent/src/tools/archive-reader.ts @@ -1,7 +1,7 @@ import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; -import { inflateSync, strFromU8 } from "fflate"; +import { bytesToText, inflateRaw } from "../utils/zip"; import { formatBytes } from "./render-utils"; import { ToolError } from "./tool-errors"; @@ -417,7 +417,7 @@ function parseZipCentralDirectory( throw new ToolError("Invalid ZIP archive: truncated central directory entry"); } - const rawPath = strFromU8(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0); + const rawPath = bytesToText(centralDirectory.subarray(nameStart, extraStart), (flags & ZIP_UTF8_FLAG) === 0); const normalizedPath = normalizeArchiveEntryPath(rawPath); if (normalizedPath) { const values = readZip64EntryValues( @@ -490,7 +490,7 @@ async function readZipFileBytes(storage: ZipStorage, uncompressedSize: number): } try { - return inflateSync(compressedBytes, { out: new Uint8Array(uncompressedSize) }); + return inflateRaw(compressedBytes, new Uint8Array(uncompressedSize)); } catch (error) { throw new ToolError(error instanceof Error ? error.message : String(error)); } diff --git a/packages/coding-agent/src/tools/fetch.ts b/packages/coding-agent/src/tools/fetch.ts index b3e6c40e4..42b9d75e9 100644 --- a/packages/coding-agent/src/tools/fetch.ts +++ b/packages/coding-agent/src/tools/fetch.ts @@ -51,34 +51,9 @@ const CONVERTIBLE_MIMES = new Set([ "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", "application/rtf", "application/epub+zip", - "image/png", - "image/jpeg", - "image/gif", - "image/webp", - "audio/mpeg", - "audio/wav", - "audio/ogg", ]); -const CONVERTIBLE_EXTENSIONS = new Set([ - ".pdf", - ".doc", - ".docx", - ".ppt", - ".pptx", - ".xls", - ".xlsx", - ".rtf", - ".epub", - ".png", - ".jpg", - ".jpeg", - ".gif", - ".webp", - ".mp3", - ".wav", - ".ogg", -]); +const CONVERTIBLE_EXTENSIONS = new Set([".pdf", ".doc", ".docx", ".ppt", ".pptx", ".xls", ".xlsx", ".rtf", ".epub"]); const NOTEBOOK_MIMES = new Set(["application/x-ipynb+json"]); const NOTEBOOK_EXTENSIONS = new Set([".ipynb"]); @@ -1144,45 +1119,25 @@ async function renderUrl( notes.push( `Image MIME type ${imageMimeType} is unsupported for inline model serialization; returning text metadata only`, ); - const shouldTryConvertibleFallback = isConvertible(mime, extHint); - if (shouldTryConvertibleFallback) { - notes.push("Attempting binary conversion fallback for unsupported image MIME type"); - } else { - notes.push("Falling back to textual rendering from initial response"); - } - skipConvertibleBinaryRetry = !shouldTryConvertibleFallback; + notes.push("Falling back to textual rendering from initial response"); + skipConvertibleBinaryRetry = true; } else { const binary = await fetchBinary(finalUrl, timeout, signal); if (binary.ok) { notes.push("Fetched image binary"); - const conversionExtension = getExtensionHint(finalUrl, binary.contentDisposition) || extHint; - let convertedText: string | null = null; - const converted = await convertWithMarkit(binary.buffer, conversionExtension, timeout, signal); - if (converted.ok) { - if (converted.content.trim().length > 50) { - notes.push("Converted with markit"); - convertedText = converted.content; - } else { - notes.push("markit conversion produced no usable output"); - } - } else if (converted.error) { - notes.push(`markit conversion failed: ${converted.error}`); - } else { - notes.push("markit conversion failed"); - } if (binary.buffer.byteLength > MAX_INLINE_IMAGE_SOURCE_BYTES) { notes.push( `Image exceeds inline source limit (${binary.buffer.byteLength} bytes > ${MAX_INLINE_IMAGE_SOURCE_BYTES} bytes)`, ); const output = finalizeOutput( - convertedText ?? `Fetched image content (${imageMimeType}), but it is too large to inline render.`, + `Fetched image content (${imageMimeType}), but it is too large to inline render.`, ); return { url, finalUrl, contentType: imageMimeType, - method: convertedText ? "markit" : "image-too-large", + method: "image-too-large", content: output.content, fetchedAt, truncated: output.truncated, @@ -1199,15 +1154,13 @@ async function renderUrl( if (!isDecodedImage) { notes.push(`Fetched payload could not be decoded as ${imageMimeType}; returning text metadata only`); const output = finalizeOutput( - convertedText ?? - rawContent ?? - `Fetched payload was labeled ${imageMimeType}, but bytes were not a valid image.`, + rawContent ?? `Fetched payload was labeled ${imageMimeType}, but bytes were not a valid image.`, ); return { url, finalUrl, contentType: imageMimeType, - method: convertedText ? "markit" : "image-invalid", + method: "image-invalid", content: output.content, fetchedAt, truncated: output.truncated, @@ -1219,13 +1172,13 @@ async function renderUrl( `Image exceeds inline output limit after resize (${resized.buffer.length} bytes > ${MAX_INLINE_IMAGE_OUTPUT_BYTES} bytes)`, ); const output = finalizeOutput( - convertedText ?? `Fetched image content (${imageMimeType}), but it is too large to inline render.`, + `Fetched image content (${imageMimeType}), but it is too large to inline render.`, ); return { url, finalUrl, contentType: imageMimeType, - method: convertedText ? "markit" : "image-too-large", + method: "image-too-large", content: output.content, fetchedAt, truncated: output.truncated, @@ -1234,7 +1187,7 @@ async function renderUrl( } const dimensionNote = formatDimensionNote(resized); - let imageSummary = convertedText ?? `Fetched image content (${resized.mimeType}).`; + let imageSummary = `Fetched image content (${resized.mimeType}).`; if (dimensionNote) { imageSummary += `\n${dimensionNote}`; } diff --git a/packages/coding-agent/src/tools/write.ts b/packages/coding-agent/src/tools/write.ts index 495266632..e0cfced7b 100644 --- a/packages/coding-agent/src/tools/write.ts +++ b/packages/coding-agent/src/tools/write.ts @@ -65,12 +65,6 @@ import { toolResult } from "./tool-result"; const LOOSE_HASHLINE_HEADER_RE = /^\s*\[[^#\r\n]+#[^ \t\r\n]*\]\s*$/; const EXECUTABLE_NOTICE = "[Notice: Made executable via chmod +x]"; -let fflateModulePromise: Promise | undefined; -async function loadFflate(): Promise { - if (!fflateModulePromise) fflateModulePromise = import("fflate"); - return fflateModulePromise; -} - const writeSchema = type({ path: type("string").describe("file path"), content: type("string").describe("file content"), @@ -387,8 +381,8 @@ export class WriteTool implements AgentTool void; } -declare global { - var $libmupdf_wasm_Module: MuPdfWasmModuleConfig | undefined; -} - function logMuPdfWasmOutput(stream: "stdout" | "stderr", values: unknown[]): void { const message = values.length === 1 && typeof values[0] === "string" ? values[0] : values.map(String).join(" "); logger.debug("mupdf wasm output", { stream, message }); } +// `$libmupdf_wasm_Module` is declared globally (as `any`) by the mupdf package. +// Install print hooks before the WASM module initializes so its stdout/stderr +// route to the file logger instead of corrupting the TUI. function installMuPdfWasmLogger(): void { - const moduleConfig = globalThis.$libmupdf_wasm_Module ?? {}; - moduleConfig.print = (...values) => logMuPdfWasmOutput("stdout", values); - moduleConfig.printErr = (...values) => logMuPdfWasmOutput("stderr", values); + const moduleConfig: MuPdfWasmModuleConfig = globalThis.$libmupdf_wasm_Module ?? {}; + moduleConfig.print = (...values: unknown[]) => logMuPdfWasmOutput("stdout", values); + moduleConfig.printErr = (...values: unknown[]) => logMuPdfWasmOutput("stderr", values); globalThis.$libmupdf_wasm_Module = moduleConfig; } installMuPdfWasmLogger(); let markit: () => Markit | Promise = async () => { - const promise = import("markit-ai").then(({ Markit }) => { + // Lazy: keep the document engine (mammoth/fflate/mupdf) off the startup + // import graph — it loads only when a document is first converted. + const promise = import("../markit").then(({ Markit }) => { const instance = new Markit(); markit = () => instance; return instance; diff --git a/packages/coding-agent/src/utils/turndown.ts b/packages/coding-agent/src/utils/turndown.ts new file mode 100644 index 000000000..f548372f8 --- /dev/null +++ b/packages/coding-agent/src/utils/turndown.ts @@ -0,0 +1,83 @@ +import TurndownService from "turndown"; +import { gfm } from "turndown-plugin-gfm"; + +type TurndownListParent = { + nodeName: string; + getAttribute(name: string): string | null; + children: ArrayLike; +}; + +/** + * Build a Turndown instance configured for GFM with the fixes omp relies on: + * `~~strikethrough~~`, unescaped heading periods, and single-space list markers. + * + * Shared by the web scrapers (HTML → markdown) and the markit document engine + * (`src/markit`). The rule set must stay identical across both call sites. + */ +export function createTurndown(): TurndownService { + const turndown = new TurndownService({ + headingStyle: "atx", + codeBlockStyle: "fenced", + bulletListMarker: "-", + }); + turndown.use(gfm); + // GFM spec uses ~~ (double tilde), not ~ (single) + turndown.addRule("strikethrough", { + filter: ["del", "s", "strike"], + replacement(content) { + return `~~${content}~~`; + }, + }); + // Unescape the backslash turndown inserts before periods in headings ("1." -> "1\.") + turndown.addRule("heading", { + filter: ["h1", "h2", "h3", "h4", "h5", "h6"], + replacement(content, node) { + const level = Number(node.nodeName.charAt(1)); + const prefix = "#".repeat(level); + const cleaned = content.replace(/\\([.])/g, "$1").trim(); + return `\n\n${prefix} ${cleaned}\n\n`; + }, + }); + // Single space after the marker (turndown hardcodes three) + turndown.addRule("listItem", { + filter: "li", + replacement(content, node, options) { + const body = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n "); + const parent = node.parentNode as unknown as TurndownListParent | null; + let prefix = `${options.bulletListMarker} `; + if (parent?.nodeName === "OL") { + const start = parent.getAttribute("start"); + const index = Array.prototype.indexOf.call(parent.children, node); + prefix = `${(start ? Number(start) : 1) + index}. `; + } + return prefix + body + (node.nextSibling ? "\n" : ""); + }, + }); + return turndown; +} + +/** + * Normalize HTML tables so turndown-plugin-gfm can render them: + * - strip `

` tags inside ``/`` cells (joining paragraphs with a space) + * - wrap the first row in `` when missing + */ +export function normalizeTablesHtml(html: string): string { + let result = html.replace( + /<(td|th)([^>]*)>([\s\S]*?)<\/(td|th)>/gi, + (_match, tag: string, attrs: string, inner: string, closeTag: string) => { + const stripped = inner + .replace(/^\s*

/i, "") + .replace(/<\/p>\s*$/i, "") + .replace(/<\/p>\s*

/gi, " "); + return `<${tag}${attrs}>${stripped}`; + }, + ); + result = result.replace( + /]*)>\s*(?:\s*)?()([\s\S]*?)<\/(?:tbody>\s*<\/)?table>/gi, + (_match, attrs: string, firstRow: string, rest: string) => { + const theadRow = firstRow.replace(//gi, ""); + return `${theadRow}${rest}`; + }, + ); + return result; +} diff --git a/packages/coding-agent/src/utils/zip.ts b/packages/coding-agent/src/utils/zip.ts new file mode 100644 index 000000000..d2bf06df7 --- /dev/null +++ b/packages/coding-agent/src/utils/zip.ts @@ -0,0 +1,29 @@ +// The single ZIP/DEFLATE boundary for the codebase. This is the ONLY module +// that imports `fflate`; the markit document converters, the write tool, and +// the archive reader all go through here so there is exactly one ZIP +// implementation to reason about. Do not import `fflate` (or another archive +// library) anywhere else. +import type { Unzipped } from "fflate"; +import { inflateSync, strFromU8 } from "fflate"; + +export type { Unzipped } from "fflate"; +export { unzipSync as unzip, zipSync as zip } from "fflate"; + +/** Read a single ZIP entry as UTF-8 text, or `undefined` when the entry is absent. */ +export function unzipText(entries: Unzipped, entryPath: string): string | undefined { + const data = entries[entryPath]; + return data ? strFromU8(data) : undefined; +} + +/** + * Inflate a raw DEFLATE stream (a single deflate-compressed ZIP member). Pass a + * preallocated `into` buffer when the uncompressed size is known up front. + */ +export function inflateRaw(bytes: Uint8Array, into?: Uint8Array): Uint8Array { + return into ? inflateSync(bytes, { out: into }) : inflateSync(bytes); +} + +/** Decode raw bytes as text — UTF-8 by default, latin1 when `latin1` is set. */ +export function bytesToText(bytes: Uint8Array, latin1?: boolean): string { + return strFromU8(bytes, latin1); +} diff --git a/packages/coding-agent/src/web/scrapers/types.ts b/packages/coding-agent/src/web/scrapers/types.ts index ae985a74a..cd84db68e 100644 --- a/packages/coding-agent/src/web/scrapers/types.ts +++ b/packages/coding-agent/src/web/scrapers/types.ts @@ -243,58 +243,15 @@ export async function loadPage(url: string, options: LoadPageOptions = {}): Prom /** Module-level Turndown instance — built lazily on first use. */ let turndownPromise: Promise | undefined; -type TurndownListParent = { - nodeName: string; - getAttribute(name: string): string | null; - children: ArrayLike; -}; - function getTurndown(): Promise { turndownPromise ||= initTurndown(); return turndownPromise; } async function initTurndown(): Promise { - const [{ default: TurndownService }, { gfm }] = await Promise.all([ - import("turndown"), - import("turndown-plugin-gfm"), - ]); - const turndown = new TurndownService({ - headingStyle: "atx", - codeBlockStyle: "fenced", - bulletListMarker: "-", - }); - turndown.use(gfm); - turndown.addRule("strikethrough", { - filter: ["del", "s", "strike"], - replacement(content) { - return `~~${content}~~`; - }, - }); - turndown.addRule("heading", { - filter: ["h1", "h2", "h3", "h4", "h5", "h6"], - replacement(content, node) { - const level = Number(node.nodeName.charAt(1)); - const prefix = "#".repeat(level); - const cleaned = content.replace(/\\([.])/g, "$1").trim(); - return `\n\n${prefix} ${cleaned}\n\n`; - }, - }); - turndown.addRule("listItem", { - filter: "li", - replacement(content, node, options) { - content = content.replace(/^\n+/, "").replace(/\n+$/, "\n").replace(/\n/gm, "\n "); - const parent = node.parentNode as unknown as TurndownListParent | null; - let prefix = `${options.bulletListMarker} `; - if (parent?.nodeName === "OL") { - const start = parent.getAttribute("start"); - const index = Array.prototype.indexOf.call(parent.children, node); - prefix = `${(start ? Number(start) : 1) + index}. `; - } - return prefix + content + (node.nextSibling ? "\n" : ""); - }, - }); - return turndown; + // Lazy import keeps turndown/turndown-plugin-gfm off the startup graph. + const { createTurndown } = await import("../../utils/turndown"); + return createTurndown(); } /** diff --git a/packages/coding-agent/test/markit-converters.test.ts b/packages/coding-agent/test/markit-converters.test.ts new file mode 100644 index 000000000..66ade4156 --- /dev/null +++ b/packages/coding-agent/test/markit-converters.test.ts @@ -0,0 +1,160 @@ +/** + * Runtime coverage for the in-house markit document engine (src/markit), which + * replaced the `markit-ai` package. Each format is generated in-memory via the + * shared zip util (src/utils/zip) — no external fixtures — and converted through + * the public wrapper (src/utils/markit), locking: docx text, xlsx tables, pptx + * slides, epub metadata+spine, shared HTML-table normalization, image + * extraction, the nested/relative zip path resolution (the JSZip→fflate + * regression surface), and the unsupported-format error contract. + */ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { convertBufferWithMarkit, convertFileWithMarkit } from "@oh-my-pi/pi-coding-agent/utils/markit"; +import { zip } from "@oh-my-pi/pi-coding-agent/utils/zip"; + +const enc = (s: string): Uint8Array => new TextEncoder().encode(s); +const WML = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"; + +function makeDocx(bodyXml: string): Uint8Array { + return zip({ + "[Content_Types].xml": enc( + ``, + ), + "_rels/.rels": enc( + ``, + ), + "word/document.xml": enc( + `${bodyXml}`, + ), + }); +} + +describe("markit converters", () => { + it("converts docx paragraphs to markdown", async () => { + const docx = makeDocx( + `First paragraph.Second paragraph.`, + ); + const result = await convertBufferWithMarkit(docx, ".docx"); + expect(result.ok).toBe(true); + expect(result.content).toBe("First paragraph.\n\nSecond paragraph."); + }); + + it("converts xlsx sheets to markdown tables", async () => { + const xlsx = zip({ + "xl/workbook.xml": enc( + ``, + ), + "xl/_rels/workbook.xml.rels": enc( + ``, + ), + "xl/worksheets/sheet1.xml": enc( + `NameAgeAlice30`, + ), + }); + const result = await convertBufferWithMarkit(xlsx, ".xlsx"); + expect(result.ok).toBe(true); + expect(result.content).toContain("## People"); + expect(result.content).toContain("| Name | Age |"); + expect(result.content).toContain("| --- | --- |"); + expect(result.content).toContain("| Alice | 30 |"); + }); + + it("reads an xlsx worksheet through an absolute (/-prefixed) rel target", async () => { + const xlsx = zip({ + "xl/workbook.xml": enc( + ``, + ), + "xl/_rels/workbook.xml.rels": enc( + ``, + ), + "xl/worksheets/sheet1.xml": enc( + `Header42`, + ), + }); + const result = await convertBufferWithMarkit(xlsx, ".xlsx"); + expect(result.ok).toBe(true); + expect(result.content).toContain("| Header |"); + expect(result.content).toContain("| 42 |"); + }); + + it("converts pptx slides with a title heading and body text", async () => { + const pptx = zip({ + "ppt/presentation.xml": enc( + ``, + ), + "ppt/_rels/presentation.xml.rels": enc( + ``, + ), + "ppt/slides/slide1.xml": enc( + `The TitleBody line`, + ), + }); + const result = await convertBufferWithMarkit(pptx, ".pptx"); + expect(result.ok).toBe(true); + expect(result.content).toContain("# The Title"); + expect(result.content).toContain("Body line"); + }); + + it("extracts a pptx image through a ../media relative rel target into imageDir", async () => { + const dir = await fs.mkdtemp(path.join(os.tmpdir(), "markit-pptx-")); + try { + const pptx = zip({ + "ppt/presentation.xml": enc( + ``, + ), + "ppt/_rels/presentation.xml.rels": enc( + ``, + ), + "ppt/slides/slide1.xml": enc( + ``, + ), + "ppt/slides/_rels/slide1.xml.rels": enc( + ``, + ), + "ppt/media/image1.png": new Uint8Array([137, 80, 78, 71, 13, 10, 26, 10]), + }); + const pptxPath = path.join(dir, "deck.pptx"); + const imageDir = path.join(dir, "imgs"); + await Bun.write(pptxPath, pptx); + const result = await convertFileWithMarkit(pptxPath, undefined, { imageDir }); + expect(result.ok).toBe(true); + const written = await fs.readdir(imageDir); + expect(written).toHaveLength(1); + expect(result.content).toContain(`](${path.join(imageDir, written[0]!)})`); + } finally { + await fs.rm(dir, { recursive: true, force: true }); + } + }); + + it("converts epub spine, normalizes HTML tables, and resolves a non-root OPF basePath", async () => { + const epub = zip({ + "META-INF/container.xml": enc( + ``, + ), + "OEBPS/content.opf": enc( + `Nested BookAda`, + ), + "OEBPS/text/ch1.xhtml": enc( + `

Chapter One

Body text.

AB
12
`, + ), + }); + const result = await convertBufferWithMarkit(epub, ".epub"); + expect(result.ok).toBe(true); + expect(result.content).toContain("**Title:** Nested Book"); + expect(result.content).toContain("**Authors:** Ada"); + expect(result.content).toContain("## Chapter One"); + expect(result.content).toContain("Body text."); + // normalizeTablesHtml promotes the first row to a header so GFM renders a table. + expect(result.content).toContain("| A | B |"); + expect(result.content).toContain("| --- | --- |"); + }); + + it("reports an unsupported format instead of emitting garbage", async () => { + const rtf = enc("{\\rtf1\\ansi binary-ish}"); + const result = await convertBufferWithMarkit(rtf, ".rtf"); + expect(result.ok).toBe(false); + expect(result.error).toContain("Unsupported format"); + }); +}); diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index 6d829548a..e66c57c04 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -18,8 +18,8 @@ import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; import { DEFAULT_FILE_LIMIT, MULTI_FILE_PER_FILE_MATCHES, SearchTool } from "@oh-my-pi/pi-coding-agent/tools/search"; import * as toolTimeouts from "@oh-my-pi/pi-coding-agent/tools/tool-timeouts"; import { WriteTool } from "@oh-my-pi/pi-coding-agent/tools/write"; +import { unzip } from "@oh-my-pi/pi-coding-agent/utils/zip"; import { $which, Snowflake } from "@oh-my-pi/pi-utils"; -import { unzipSync } from "fflate"; // Helper to extract text from content blocks function getTextOutput(result: any): string { @@ -931,7 +931,7 @@ describe("Coding Agent Tools", () => { `Successfully wrote ${content.length} bytes to ${path.basename(archivePath)}:pkg/README.md`, ); - const unzipped = unzipSync(new Uint8Array(fs.readFileSync(archivePath))); + const unzipped = unzip(new Uint8Array(fs.readFileSync(archivePath))); expect(new TextDecoder().decode(unzipped["pkg/README.md"])).toBe(content); expect(new TextDecoder().decode(unzipped["pkg/src/index.ts"])).toBe("export const archiveValue = 1;\n"); }); diff --git a/packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts b/packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts index cab602c17..5d5bdae76 100644 --- a/packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts +++ b/packages/coding-agent/test/tools/fetch-binary-dispatch.test.ts @@ -7,10 +7,10 @@ import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; +import { zip } from "@oh-my-pi/pi-coding-agent/utils/zip"; import * as scrapers from "@oh-my-pi/pi-coding-agent/web/scrapers/types"; import * as scraperUtils from "@oh-my-pi/pi-coding-agent/web/scrapers/utils"; import { Snowflake } from "@oh-my-pi/pi-utils"; -import { zipSync } from "fflate"; function makeSession(testDir: string): ToolSession { const sessionFile = path.join(testDir, "session.jsonl"); @@ -112,7 +112,7 @@ describe("read URL binary dispatch", () => { }); it("lists a remote zip instead of dumping decoded bytes", async () => { - const zipBytes = zipSync({ + const zipBytes = zip({ "root.txt": Buffer.from("root file\n"), "nested/data.txt": Buffer.from("nested file\n"), }); diff --git a/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts b/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts index 56c13c2b5..de67c08b7 100644 --- a/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts +++ b/packages/coding-agent/test/tools/fetch-kagi-toggle.test.ts @@ -167,11 +167,6 @@ describe("read tool URL handling", () => { ok: true, buffer: imageBytes, }); - vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({ - ok: false, - content: "", - error: "markit unavailable", - }); vi.spyOn(imageResize, "resizeImage").mockResolvedValue({ buffer: imageBytes, mimeType: "image/png", @@ -222,11 +217,7 @@ describe("read tool URL handling", () => { ok: true, buffer: new Uint8Array([137, 80, 78, 71]), }); - vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({ - ok: false, - content: "", - error: "markit unavailable", - }); + const convertSpy = vi.spyOn(scraperUtils, "convertWithMarkit"); const result = await tool.execute("fetch-image-resized", { path: "https://example.com/image.png" }); const imageBlock = result.content.find( @@ -240,52 +231,9 @@ describe("read tool URL handling", () => { expect(imageBlock?.data).toBe("cmVzaXplZA=="); expect(textBlock?.type).toBe("text"); expect(textBlock?.text).toContain("displayed at 1000x500"); + expect(convertSpy).not.toHaveBeenCalled(); }); - it("keeps markit extracted text for image responses", async () => { - const session = createSession(); - const tool = new ReadTool(session); - const extractedText = "Converted image text content that is definitely longer than fifty characters."; - vi.spyOn(imageResize, "resizeImage").mockResolvedValue({ - buffer: new Uint8Array([1, 2, 3]), - mimeType: "image/png", - originalWidth: 100, - originalHeight: 100, - width: 100, - height: 100, - wasResized: false, - get data() { - return "aW1hZ2U="; - }, - }); - vi.spyOn(scrapers, "loadPage").mockResolvedValue({ - ok: true, - status: 200, - contentType: "image/png", - finalUrl: "https://example.com/image.png", - content: "", - }); - vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({ - ok: true, - buffer: new Uint8Array([137, 80, 78, 71]), - }); - vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({ - ok: true, - content: extractedText, - }); - - const result = await tool.execute("fetch-image-with-ocr", { path: "https://example.com/image.png" }); - const textBlock = result.content.find(content => content.type === "text"); - const imageBlock = result.content.find( - (content): content is { type: "image"; data: string; mimeType: string } => content.type === "image", - ); - - expect(result.details?.method).toBe("image"); - expect(textBlock?.type).toBe("text"); - expect(textBlock?.text).toContain(extractedText); - expect(imageBlock?.mimeType).toBe("image/png"); - expect(imageBlock?.data).toBe("aW1hZ2U="); - }); it("falls back to text-only output for unsupported image MIME types", async () => { const session = createSession(); const tool = new ReadTool(session); @@ -309,39 +257,6 @@ describe("read tool URL handling", () => { expect(textBlock?.text).toContain(""); }); - it("uses binary conversion fallback for unsupported image MIME when extension is convertible", async () => { - const session = createSession(); - const tool = new ReadTool(session); - const convertedText = "Converted image text from markit fallback with sufficient length to pass threshold."; - const fetchBinarySpy = vi.spyOn(scraperUtils, "fetchBinary").mockResolvedValue({ - ok: true, - buffer: new Uint8Array([255, 216, 255, 224]), - }); - const convertSpy = vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({ - ok: true, - content: convertedText, - }); - vi.spyOn(scrapers, "loadPage").mockResolvedValue({ - ok: true, - status: 200, - contentType: "image/jpg", - finalUrl: "https://example.com/image.jpg", - content: "\u0000\u0001garbage", - }); - - const result = await tool.execute("fetch-image-jpg-fallback", { path: "https://example.com/image.jpg" }); - const imageBlock = result.content.find(content => content.type === "image"); - const textBlock = result.content.find(content => content.type === "text"); - - expect(result.details?.method).toBe("markit"); - expect(fetchBinarySpy).toHaveBeenCalledTimes(1); - expect(convertSpy).toHaveBeenCalledTimes(1); - expect(result.details?.notes).toContain("Attempting binary conversion fallback for unsupported image MIME type"); - expect(imageBlock).toBeUndefined(); - expect(textBlock?.type).toBe("text"); - expect(textBlock?.text).toContain(convertedText); - }); - it("does not treat text/html at .png paths as inline images", async () => { const session = createSession(); const tool = new ReadTool(session); @@ -404,11 +319,6 @@ describe("read tool URL handling", () => { ok: true, buffer: new Uint8Array([60, 104, 116, 109, 108]), }); - vi.spyOn(scraperUtils, "convertWithMarkit").mockResolvedValue({ - ok: false, - content: "", - error: "conversion failed", - }); vi.spyOn(imageResize, "resizeImage").mockResolvedValue({ buffer: new Uint8Array([60, 104, 116, 109, 108]), mimeType: "image/png",