From ea632a518b076a6610196332512ceded6863d43f Mon Sep 17 00:00:00 2001 From: can1357 Date: Sat, 11 Jul 2026 06:58:56 +0200 Subject: [PATCH] feat(coding-agent): expanded search capabilities and scraping reliability - Added six new search providers (Bing, Yahoo, Ecosia, Startpage, Mojeek, and Public) to expand coverage and parallel search capabilities. - Implemented a unified `browserFetch` utility with headless-browser fallback and randomized Chrome profiles to improve scrape reliability. - Integrated automated bot-defense mechanisms including CAPTCHA detection, ALTCHA proof-of-work, and homepage-token flows. - Fixed hanging search CLI commands by ensuring proper closure of AuthStorage connections. --- README.md | 13 +- bun.lock | 20 ++ docs/tools/web_search.md | 33 ++- package.json | 1 + packages/coding-agent/CHANGELOG.md | 7 +- packages/coding-agent/package.json | 1 + .../coding-agent/src/web/search/provider.ts | 33 ++- .../src/web/search/providers/bing.ts | 197 ++++++++++++++ .../web/search/providers/browser-headers.ts | 78 +++++- .../src/web/search/providers/browser-page.ts | 123 +++++++++ .../src/web/search/providers/duckduckgo.ts | 27 +- .../src/web/search/providers/ecosia.ts | 178 +++++++++++++ .../src/web/search/providers/google.ts | 104 ++------ .../src/web/search/providers/mojeek.ts | 206 +++++++++++++++ .../src/web/search/providers/public.ts | 201 +++++++++++++++ .../src/web/search/providers/startpage.ts | 213 +++++++++++++++ .../src/web/search/providers/yahoo.ts | 179 +++++++++++++ packages/coding-agent/src/web/search/types.ts | 30 +++ .../test/tools/web-search-bing.test.ts | 184 +++++++++++++ .../tools/web-search-browser-headers.test.ts | 69 +++++ .../test/tools/web-search-ecosia.test.ts | 172 +++++++++++++ .../test/tools/web-search-google.test.ts | 2 +- .../test/tools/web-search-mojeek.test.ts | 161 ++++++++++++ .../test/tools/web-search-public.test.ts | 198 ++++++++++++++ .../test/tools/web-search-startpage.test.ts | 243 ++++++++++++++++++ .../test/tools/web-search-yahoo.test.ts | 188 ++++++++++++++ 26 files changed, 2757 insertions(+), 104 deletions(-) create mode 100644 packages/coding-agent/src/web/search/providers/bing.ts create mode 100644 packages/coding-agent/src/web/search/providers/browser-page.ts create mode 100644 packages/coding-agent/src/web/search/providers/ecosia.ts create mode 100644 packages/coding-agent/src/web/search/providers/mojeek.ts create mode 100644 packages/coding-agent/src/web/search/providers/public.ts create mode 100644 packages/coding-agent/src/web/search/providers/startpage.ts create mode 100644 packages/coding-agent/src/web/search/providers/yahoo.ts create mode 100644 packages/coding-agent/test/tools/web-search-bing.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-browser-headers.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-ecosia.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-mojeek.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-public.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-startpage.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-yahoo.test.ts diff --git a/README.md b/README.md index ec9f6074f..49a3fb43f 100644 --- a/README.md +++ b/README.md @@ -309,13 +309,13 @@ Ollama `local` · Ollama Cloud · LM Studio `local` · llama.cpp `local` · vLLM Full provider & routing reference at [omp.sh/docs/providers](https://omp.sh/docs/providers). -## Eighteen backends. _One tool the agent already knows_. +## Twenty-five backends. _One tool the agent already knows_. -`web_search` is built in, not bolted on. `auto` walks an eighteen-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive. +`web_search` is built in, not bolted on. `auto` walks a twenty-five-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive. ### Search providers -Eighteen backends. Pin one, or let `auto` walk the chain in order. +Twenty-five backends. Pin one, or let `auto` walk the chain in order. | provider | auth | | ------------ | ---------------------- | @@ -338,6 +338,13 @@ Eighteen backends. Pin one, or let `auto` walk the chain in order. | `synthetic` | `SYNTHETIC_API_KEY` | | `searxng` | self-hosted | | `duckduckgo` | no key | +| `bing` | no key | +| `yahoo` | no key | +| `startpage` | no key | +| `google` | no key (browser) | +| `ecosia` | no key (browser) | +| `mojeek` | no key (browser) | +| `public` | no key (all of the above, consolidated) | ### Specialised handlers diff --git a/bun.lock b/bun.lock index 3d54b7172..568ba0939 100644 --- a/bun.lock +++ b/bun.lock @@ -102,6 +102,7 @@ "diff": "catalog:", "fast-xml-parser": "catalog:", "handlebars": "catalog:", + "header-generator": "catalog:", "linkedom": "catalog:", "lru-cache": "catalog:", "mammoth": "catalog:", @@ -380,6 +381,7 @@ "fflate": "0.8.3", "ghostty-web": "^0.4.0", "handlebars": "^4.7.9", + "header-generator": "^2.1.82", "linkedom": "^0.18.12", "lint-staged": "^17.0.7", "lru-cache": "11.5.1", @@ -850,6 +852,8 @@ "@rolldown/pluginutils": ["@rolldown/pluginutils@1.0.1", "", {}, "sha512-2j9bGt5Jh8hj+vPtgzPtl72j0yRxHAyumoo6TNfAjsLB04UtpSvPbPcDcBMxz7n+9CYB0c1GxQFxYRg2jimqGw=="], + "@sindresorhus/is": ["@sindresorhus/is@4.6.0", "", {}, "sha512-t09vSN3MdfsyCHoFcTRCH/iUtG7OJ0CsjzB8cjAmKc/va/kIgeDI/TxsigdncE/4be734m0cvIYwNaV4i2XqAw=="], + "@so-ric/colorspace": ["@so-ric/colorspace@1.1.6", "", { "dependencies": { "color": "^5.0.2", "text-hex": "1.0.x" } }, "sha512-/KiKkpHNOBgkFJwu9sh48LkHSMYGyuTcSFK/qMBdnOAlrRJzRSXAOFB5qwzaVQuDl8wAvHVMkaASQDReTahxuw=="], "@tailwindcss/node": ["@tailwindcss/node@4.3.2", "", { "dependencies": { "@jridgewell/remapping": "^2.3.5", "enhanced-resolve": "5.21.6", "jiti": "^2.7.0", "lightningcss": "1.32.0", "magic-string": "^0.30.21", "source-map-js": "^1.2.1", "tailwindcss": "4.3.2" } }, "sha512-yWP/sqEcBLaD8JuA6zNwxoYKr75qxTioYwlRwekj5Jr/I5GXnoJfjetH/psLUIv74cYTH2lBUEzBkinthoYcBg=="], @@ -970,6 +974,8 @@ "bun-types": ["bun-types@1.3.14", "", { "dependencies": { "@types/node": "*" } }, "sha512-4N0ig0fEomHt5R0KCFWjovxow98rIoRwKolrYdCcknNwMekCXRnWEUvgu5soYV8QXtVsrUD8B95MBOZGPvr6KQ=="], + "callsites": ["callsites@3.1.0", "", {}, "sha512-P8BjAsXvZS+VIDUI11hHCQEv74YT67YUi5JJFNWIqL235sBmjX4+qx9Muvls5ivyNENctx46xQLQ3aTuE7ssaQ=="], + "caniuse-lite": ["caniuse-lite@1.0.30001803", "", {}, "sha512-g/uHREV2ZpK9qMalCsWaxmA6ol+DX8GYhuf3T40RKoP+oL7vhRJh8LNt73PCjpnR6l14FzfPrB5Yux4PKm2meg=="], "chalk": ["chalk@5.6.2", "", {}, "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA=="], @@ -1046,6 +1052,8 @@ "domutils": ["domutils@4.0.2", "", { "dependencies": { "dom-serializer": "^3.0.0", "domelementtype": "^3.0.0", "domhandler": "^6.0.0" } }, "sha512-qI4JLRKnSzqFqr7hAlS5xQDusBCjKSEG4t4+7aNrIQMHBcsC2TGEhuyABJdYkgSewL57PNLYEiibY2iPKhKpaA=="], + "dot-prop": ["dot-prop@6.0.1", "", { "dependencies": { "is-obj": "^2.0.0" } }, "sha512-tE7ztYzXHIeyvc7N+hR3oi7FIbf/NIjVP9hmAt3yMXzrQ072/fpjGLx2GxNxGxUl5V73MEqYzioOMoVhGMJ5cA=="], + "duck": ["duck@0.1.12", "", { "dependencies": { "underscore": "^1.13.1" } }, "sha512-wkctla1O6VfP89gQ+J/yDesM0S7B7XLXjKGzXxMDVFg7uEn706niAtyYovKbyq1oT9YwDcly721/iUWoc8MVRg=="], "electron-to-chromium": ["electron-to-chromium@1.5.389", "", {}, "sha512-cEto7aeOqBfU1D+c5py5pE+ooscKE75JifxLBdFUZsqAxRS6y7kebtxAZvICszSl05gPjYHDTjY+lXpyGvpJbg=="], @@ -1104,6 +1112,8 @@ "gearhash-jit": ["gearhash-jit@1.0.2", "", {}, "sha512-UhzJL4KXSdqAKepy/tZwmi2Rcy0YMmtiC4DQS4SURCuIWdh8ECZtnXK2ePRMLigfB61hRKdLK/Vgg2bSw73izQ=="], + "generative-bayesian-network": ["generative-bayesian-network@2.1.83", "", { "dependencies": { "adm-zip": "^0.5.9", "tslib": "^2.4.0" } }, "sha512-LssI9es+oUoezoHloFGw0Hts0YEfujjBOE8KNl70oBt4HPjD/4rpUqcgZ/M7RCmgKvkmCyPB2KWowyJHuKyhfw=="], + "gensync": ["gensync@1.0.0-beta.2", "", {}, "sha512-3hN7NaskYvMDLQY55gnW3NQ+mesEAepTqlg+VEbj7zzqEMBVNhzcGYYeqFo/TlYz6eQiFcp1HcsCZO+nGgS8zg=="], "get-caller-file": ["get-caller-file@2.0.5", "", {}, "sha512-DyFP3BM/3YHTQOCUL/w0OZHR0lpKeGrxotcHWcqNEdnltqFwXVfhEBQ94eIo34AfQpo0rGki4cyIiftY06h2Fg=="], @@ -1126,6 +1136,8 @@ "has-property-descriptors": ["has-property-descriptors@1.0.2", "", { "dependencies": { "es-define-property": "^1.0.0" } }, "sha512-55JNKuIW+vq4Ke1BjOTjM2YctQIvCT7GFzHwmfZPGo5wnrgkid0YQtnAleFSqumZm4az3n2BS+erby5ipJdgrg=="], + "header-generator": ["header-generator@2.1.82", "", { "dependencies": { "browserslist": "^4.21.1", "generative-bayesian-network": "^2.1.82", "ow": "^0.28.1", "tslib": "^2.4.0" } }, "sha512-4NjPB0+bAKjPoponSmTOkK58IEF2W22sOJA5O48k/MxbCZgOm+jrU4WVR53Z2I6xFgIPkVrQmKtt1LAbWtfqXw=="], + "html-entities": ["html-entities@2.3.3", "", {}, "sha512-DV5Ln36z34NNTDgnz0EWGBLZENelNAtkiFA4kyNOG2tDI6Mz1uSWiq1wAKdyjnJwyDiDO7Fa2SO1CTxPXL8VxA=="], "html-escaper": ["html-escaper@3.0.3", "", {}, "sha512-RuMffC89BOWQoY0WKGpIhn5gX3iI54O6nRA0yC124NYVtzjmFWBIiFd8M0x+ZdX0P9R4lADg1mgP8C7PxGOWuQ=="], @@ -1140,6 +1152,8 @@ "is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], + "is-obj": ["is-obj@2.0.0", "", {}, "sha512-drqDG3cbczxxEJRoOXcOjtdp1J/lyp1mNn0xaznRs8+muBhgQcrnbspox5X5fOw0HnMnbfDzvnEMEtqDEJEo8w=="], + "is-stream": ["is-stream@2.0.1", "", {}, "sha512-hFoiJiTl63nn+kstHGBtewWSKnQLpyb155KHheA1l39uvtO9nWIop1p3udqPcUd/xbF1VLMO4n7OI6p7RbngDg=="], "is-unsafe": ["is-unsafe@1.0.1", "", {}, "sha512-CLK2+VdgERgD96EYm5lUQssZYlRg2tkZnbsxZoacmSiRxiFJ4Nk4SzjCl+Ur+v3kXIY9dTIdb3IH22y1mZ56LA=="], @@ -1198,6 +1212,8 @@ "listr2": ["listr2@10.2.2", "", { "dependencies": { "cli-truncate": "^5.2.0", "eventemitter3": "^5.0.4", "log-update": "^6.1.0", "rfdc": "^1.4.1", "wrap-ansi": "^10.0.0" } }, "sha512-JtNtbZj8q5BnDMR7trpwvwk3RIrANtIVzEUm8w7amp6xelLgyuq+4WZoTH913XaQAoH/cNdYhaNzBPA2U3xbDw=="], + "lodash.isequal": ["lodash.isequal@4.5.0", "", {}, "sha512-pDo3lu8Jhfjqls6GkMgpahsF9kCyayhgykjyLMNFTKWrpVdAQtYyB4muAMWozBB4ig/dtWAmsMxLEI8wuz+DYQ=="], + "log-update": ["log-update@6.1.0", "", { "dependencies": { "ansi-escapes": "^7.0.0", "cli-cursor": "^5.0.0", "slice-ansi": "^7.1.0", "strip-ansi": "^7.1.0", "wrap-ansi": "^9.0.0" } }, "sha512-9ie8ItPR6tjY5uYJh8K/Zrv/RMZ5VOlOWvtZdEHYSTFKZfIBPQa9tOAEeAWhd+AnIneLJ22w5fjOYtoutpWq5w=="], "logform": ["logform@2.7.0", "", { "dependencies": { "@colors/colors": "1.6.0", "@types/triple-beam": "^1.3.2", "fecha": "^4.2.0", "ms": "^2.1.1", "safe-stable-stringify": "^2.3.1", "triple-beam": "^1.3.0" } }, "sha512-TFYA4jnP7PVbmlBIfhlSe+WKxs9dklXMTEGcBCIvLhE/Tn3H6Gk1norupVW7m5Cnd4bLcr08AytbyV/xj7f/kQ=="], @@ -1270,6 +1286,8 @@ "option": ["option@0.2.4", "", {}, "sha512-pkEqbDyl8ou5cpq+VsnQbe/WlEy5qS7xPzMS1U55OCG9KPvwFD46zDbxQIj3egJSFc3D+XhYOPUzz49zQAVy7A=="], + "ow": ["ow@0.28.2", "", { "dependencies": { "@sindresorhus/is": "^4.2.0", "callsites": "^3.1.0", "dot-prop": "^6.0.1", "lodash.isequal": "^4.5.0", "vali-date": "^1.0.0" } }, "sha512-dD4UpyBh/9m4X2NVjA+73/ZPBRF+uF4zIMFvvQsabMiEK8x41L3rQ8EENOi35kyyoaJwNxEeJcP6Fj1H4U409Q=="], + "pako": ["pako@1.0.11", "", {}, "sha512-4hLB8Py4zZce5s4yd9XzopqwVv/yGNhV1Bl8NTmCq1763HeK2+EwVTv+leGeL13Dnh2wfbqowVPXCIO0z4taYw=="], "parse5": ["parse5@7.3.0", "", { "dependencies": { "entities": "^6.0.0" } }, "sha512-IInvU7fabl34qmi9gY8XOVxhYyMyuH2xUNpb2q8/Y+7552KlejkRvqvD19nMoUW/uQGGbqNpA6Tufu5FL5BZgw=="], @@ -1426,6 +1444,8 @@ "util-deprecate": ["util-deprecate@1.0.2", "", {}, "sha512-EPD5q1uXyFxJpCrLnCc1nHnq3gOa6DZBocAIiI2TaSCA7VCJ1UJDMagCzIkXNsUYfD1daK//LTEQ8xiIbrHtcw=="], + "vali-date": ["vali-date@1.0.0", "", {}, "sha512-sgECfZthyaCKW10N0fm27cg8HYTFK5qMWgypqkXMQ4Wbl/zZKx7xZICgcoxIIE+WFAP/MBL2EFwC/YvLxw3Zeg=="], + "vite": ["vite@8.1.3", "", { "dependencies": { "lightningcss": "^1.32.0", "picomatch": "^4.0.4", "postcss": "^8.5.16", "rolldown": "~1.1.3", "tinyglobby": "^0.2.17" }, "optionalDependencies": { "fsevents": "~2.3.3" }, "peerDependencies": { "@types/node": "^20.19.0 || >=22.12.0", "@vitejs/devtools": "^0.3.0", "esbuild": "^0.27.0 || ^0.28.0", "jiti": ">=1.21.0", "less": "^4.0.0", "sass": "^1.70.0", "sass-embedded": "^1.70.0", "stylus": ">=0.54.8", "sugarss": "^5.0.0", "terser": "^5.16.0", "tsx": "^4.8.1", "yaml": "^2.4.2" }, "optionalPeers": ["@types/node", "@vitejs/devtools", "esbuild", "jiti", "less", "sass", "sass-embedded", "stylus", "sugarss", "terser", "tsx", "yaml"], "bin": { "vite": "bin/vite.js" } }, "sha512-Ds+gBRbj0lwRO2Y5hwnUBdxSwlAve9LeRyU4sNnAr0ewW0gWF0n5bgXgUzbgZ49MV9BVUAQUFYVcDUcilUExMA=="], "vite-plugin-solid": ["vite-plugin-solid@2.11.12", "", { "dependencies": { "@babel/core": "^7.23.3", "@types/babel__core": "^7.20.4", "babel-preset-solid": "^1.8.4", "merge-anything": "^5.1.7", "solid-refresh": "^0.6.3", "vitefu": "^1.0.4" }, "peerDependencies": { "@testing-library/jest-dom": "^5.16.6 || ^5.17.0 || ^6.*", "solid-js": "^1.7.2", "vite": "^3.0.0 || ^4.0.0 || ^5.0.0 || ^6.0.0 || ^7.0.0 || ^8.0.0" }, "optionalPeers": ["@testing-library/jest-dom"] }, "sha512-FgjPcx2OwX9h6f28jli7A4bG7PP3te8uyakE5iqsmpq3Jqi1TWLgSroC9N6cMfGRU2zXsl4Q6ISvTr2VL0QHpA=="], diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 679b4001b..dad35b046 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -11,23 +11,32 @@ - `packages/coding-agent/src/web/search/render.ts` — TUI renderer details type. - `packages/coding-agent/src/web/search/providers/base.ts` — provider interface and shared params contract. - `packages/coding-agent/src/web/search/providers/utils.ts` — credential lookup; source normalization. + - `packages/coding-agent/src/web/search/providers/browser-headers.ts` — shared Chromium navigation headers for scrape providers. + - `packages/coding-agent/src/web/search/providers/browser-page.ts` — shared fetch/headless-browser page loader for scrape providers. - `packages/coding-agent/src/web/search/providers/anthropic.ts` — Claude web-search provider. + - `packages/coding-agent/src/web/search/providers/bing.ts` — Bing HTML SERP scraper. - `packages/coding-agent/src/web/search/providers/brave.ts` — Brave Search API adapter. - `packages/coding-agent/src/web/search/providers/codex.ts` — OpenAI Codex SSE adapter. - `packages/coding-agent/src/web/search/providers/duckduckgo.ts` — DuckDuckGo HTML frontend scraper. + - `packages/coding-agent/src/web/search/providers/ecosia.ts` — Ecosia browser-backed scraper. - `packages/coding-agent/src/web/search/providers/exa.ts` — Exa API or MCP adapter. - `packages/coding-agent/src/web/search/providers/firecrawl.ts` — Firecrawl search adapter. - `packages/coding-agent/src/web/search/providers/gemini.ts` — Gemini grounding SSE adapter. + - `packages/coding-agent/src/web/search/providers/google.ts` — Google browser-backed SERP scraper. - `packages/coding-agent/src/web/search/providers/jina.ts` — Jina Reader search adapter. - `packages/coding-agent/src/web/search/providers/kagi.ts` — Kagi provider wrapper. - `packages/coding-agent/src/web/search/providers/kimi.ts` — Kimi search adapter. + - `packages/coding-agent/src/web/search/providers/mojeek.ts` — Mojeek browser-backed scraper (independent index). - `packages/coding-agent/src/web/search/providers/parallel.ts` — Parallel provider wrapper. - `packages/coding-agent/src/web/search/providers/perplexity.ts` — Perplexity API / OAuth adapter. + - `packages/coding-agent/src/web/search/providers/public.ts` — Public Web aggregate over all credential-free engines. - `packages/coding-agent/src/web/search/providers/searxng.ts` — self-hosted SearXNG adapter. + - `packages/coding-agent/src/web/search/providers/startpage.ts` — Startpage (Google-proxied) form-flow scraper. - `packages/coding-agent/src/web/search/providers/synthetic.ts` — Synthetic search adapter. - `packages/coding-agent/src/web/search/providers/tavily.ts` — Tavily search adapter. - `packages/coding-agent/src/web/search/providers/tinyfish.ts` — TinyFish search adapter. - `packages/coding-agent/src/web/search/providers/xai.ts` — xAI Responses web-search adapter. + - `packages/coding-agent/src/web/search/providers/yahoo.ts` — Yahoo HTML SERP scraper. - `packages/coding-agent/src/web/search/providers/zai.ts` — Z.AI remote MCP adapter. - `packages/coding-agent/src/web/parallel.ts` — Parallel search/extract HTTP client. - `packages/coding-agent/src/web/kagi.ts` — Kagi HTTP client. @@ -95,7 +104,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - **Forced provider**: internal callers may pass `provider`; unavailable forced providers fall back to the auto chain instead of hard-failing (`packages/coding-agent/src/web/search/index.ts`). This field is not in the model-facing schema. - **Preferred provider**: `setPreferredSearchProvider()` sets a module-global default used by `resolveProviderChain()`. `packages/coding-agent/src/sdk.ts` and `packages/coding-agent/src/modes/controllers/selector-controller.ts` wire this from settings. - **Excluded providers**: `setExcludedSearchProviders()` records providers `resolveProviderChain()` must never return, including as fallbacks. Wired from the `providers.webSearchExclude` setting (`providers.webSearch` drives the preferred provider) in `packages/coding-agent/src/sdk.ts`, `packages/coding-agent/src/modes/interactive-mode.ts`, and `packages/coding-agent/src/modes/controllers/selector-controller.ts`. - - **Auto chain order** (18 providers): `perplexity`, `gemini`, `anthropic`, `codex`, `xai`, `zai`, `exa`, `tinyfish`, `jina`, `kagi`, `tavily`, `firecrawl`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng`, `duckduckgo` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). + - **Auto chain order** (25 providers): `perplexity`, `gemini`, `anthropic`, `codex`, `xai`, `zai`, `exa`, `tinyfish`, `jina`, `kagi`, `tavily`, `firecrawl`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng`, `duckduckgo`, `bing`, `yahoo`, `startpage`, `google`, `ecosia`, `mojeek`, `public` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). `public` is explicit-only: its `isAvailable()` returns `false` so the auto chain never fans out implicitly. - **Provider adapters** - **Perplexity** — `packages/coding-agent/src/web/search/providers/perplexity.ts` - Availability: auth precedence is `PERPLEXITY_COOKIES` -> OAuth token in `agent.db` -> `PERPLEXITY_API_KEY` / `PPLX_API_KEY` -> anonymous ask-endpoint fallback. `isAvailable()` gates the auto chain on credentials, but `isExplicitlyAvailable()` is always true, so explicit selection works unauthenticated. @@ -202,6 +211,20 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - `recency` maps to `df`; values outside `day|week|month|year` are ignored. - `limit` / `num_search_results`: collapsed and clamped to `1..20`, default `10`; output exposes `sources` only (DuckDuckGo's HTML page does not return a standalone abstract). - DuckDuckGo serves a bot-detection challenge (HTTP 200/202 with an `anomaly-modal` body) when it throttles datacenter or shared-egress IPs. The adapter detects this and raises a `SearchProviderError` so the orchestrator can fall through to the next configured provider with a clear cause. + - **Bing / Yahoo / Startpage** — `providers/bing.ts`, `providers/yahoo.ts`, `providers/startpage.ts` + - Availability: always available; no API key. Plain fetch with shared browser navigation headers. + - Bing: GET `https://www.bing.com/search`; unwraps `bing.com/ck/a?...&u=a1` redirect hrefs; `recency` maps to `filters=ex1:"ez1|ez2|ez3"` and a computed `ez5` epoch-day range for `year`. + - Yahoo: GET `https://search.yahoo.com/search`; unwraps `r.search.yahoo.com/.../RU=` tracker hrefs; `recency` maps to `btf=d|w|m` (`year` dropped). + - Startpage: proxies Google's index; GET homepage to lift the `sc` anti-bot form token, then POST `/sp/search` (tokenless GET fallback); `recency` maps to `with_date=d|w|m|y`. + - Each detects its engine's bot-challenge/consent page and raises a provider-tagged `SearchProviderError` (429) so the chain advances. + - **Google / Ecosia / Mojeek** — `providers/google.ts`, `providers/ecosia.ts`, `providers/mojeek.ts` + - Availability: always available; no API key. `browserFetch` (`providers/browser-page.ts`) tries a browser-profiled plain fetch first and escalates fetch failures, non-2xx statuses, and challenge bodies to the shared stealth headless browser (`acquireBrowser`); an injected `params.fetch` (tests) never escalates. + - Google: seeds cookies via the homepage, then loads the rendered SERP; `recency` maps to `tbs=qdr:*`. Ecosia sits behind Cloudflare (hence the browser); its organic results are Google-backed; `recency` is a server-side no-op and silently ignored. Mojeek fronts an ALTCHA proof-of-work wall that the browser path auto-solves; `recency` maps to `since=day|week|month|year`. + - Challenge pages (Google `unusual traffic`, Ecosia Firewall, Mojeek ALTCHA/robot 403) raise provider-tagged `SearchProviderError`s (429). + - **Public Web** — `packages/coding-agent/src/web/search/providers/public.ts` + - Availability: explicit selection only (`isAvailable()` is `false`; `isExplicitlyAvailable()` is `true`). + - Querying: fans out to every credential-free engine in parallel (`duckduckgo`, `bing`, `yahoo`, `startpage`, `google`, `ecosia`, `mojeek`, minus excluded ones), then consolidates: URLs deduplicated on a canonical key (host without `www.`, no trailing slash, no fragment), ranked by cross-engine consensus, then best per-engine rank; the longest snippet wins. + - Deadline race: returns at the earliest of all engines settled, 5s soft deadline with at least one success, or 30s hard cap; stragglers are aborted. Individual engine failures are tolerated; it fails only when every engine fails (aggregated 503). ## Side Effects - Network @@ -217,12 +240,14 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Many provider adapters accept `AbortSignal`; `WebSearchTool.execute()` passes the tool call signal into `executeSearch()`, which forwards it as `params.signal` to providers and rethrows cancellation during fallback. ## Limits & Caps -- Provider auto-order length: 18 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). +- Provider auto-order length: 25 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). - `formatForLLM()` truncates source snippets and citation text to 240 chars (`packages/coding-agent/src/web/search/index.ts`). - `formatForLLM()` emits at most 3 search queries, each truncated to 120 chars (`packages/coding-agent/src/web/search/index.ts`). - Brave result count: default `10`, max `20` (`DEFAULT_NUM_RESULTS`, `MAX_NUM_RESULTS` in `packages/coding-agent/src/web/search/providers/brave.ts`). - TinyFish local result count: default `10`, max `20`; the API has no count parameter and returns at most 10 results per page, so the adapter fetches documented pages (`page=0`, then `page=1` when needed) and slices locally (`packages/coding-agent/src/web/search/providers/tinyfish.ts`). - DuckDuckGo result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/duckduckgo.ts`). +- Bing / Yahoo / Startpage / Google / Ecosia / Mojeek result count: default `10`, max `20` (their `providers/*.ts` modules). +- Public Web result count: default `15`, max `30`; fan-out soft deadline `5s`, hard cap `30s` (`packages/coding-agent/src/web/search/providers/public.ts`). - Tavily result count: default `5`, max `20` (`packages/coding-agent/src/web/search/providers/tavily.ts`). - Firecrawl result count: default `10`, max `100` (`packages/coding-agent/src/web/search/providers/firecrawl.ts`). - Kimi result count: default `10`, max `20`; request timeout field fixed to `30` seconds (`packages/coding-agent/src/web/search/providers/kimi.ts`). @@ -251,7 +276,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. - `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. - Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts. TinyFish uses the collapsed value as a local cap, serializes `num_results` per page, and paginates with `page` when more results are needed. xAI sends that collapsed value as `search_parameters.max_search_results` and applies the same precedence locally after parsing to cap returned sources/citations (`10` default, `30` max). -- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, Firecrawl, and xAI. The model-facing prompt does not name specific providers. +- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, Firecrawl, xAI, DuckDuckGo, Bing, Yahoo, Startpage, Google, and Mojeek (Ecosia ignores it; Public Web passes it through). The model-facing prompt does not name specific providers. - `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain. -- DuckDuckGo is intentionally last in the auto chain because it is always available without credentials. +- The credential-free scrapers close the auto chain, cheap plain-fetch engines first (`duckduckgo`, `bing`, `yahoo`, `startpage`) and browser-backed ones after (`google`, `ecosia`, `mojeek`); `public` is listed last and never auto-selected. - Exa uses `authStorage.getApiKey("exa")`, then `EXA_API_KEY`, then unauthenticated `https://mcp.exa.ai/mcp` fallback. diff --git a/package.json b/package.json index 1ea076b57..b1691f038 100644 --- a/package.json +++ b/package.json @@ -64,6 +64,7 @@ "fast-xml-parser": "^5.9.0", "ghostty-web": "^0.4.0", "handlebars": "^4.7.9", + "header-generator": "^2.1.82", "linkedom": "^0.18.12", "lint-staged": "^17.0.7", "lru-cache": "11.5.1", diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 751d7d44c..fa5b3d9d6 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -8,6 +8,8 @@ - Added `/vibe` mode: the model becomes a director whose toolset is stripped to `read` plus five new session tools (`vibe_spawn`, `vibe_send`, `vibe_wait`, `vibe_kill`, `vibe_list`) for driving persistent background worker sessions. Workers come in two flavors mapped to existing model tiers — `fast` (sonic/`pi/smol`) and `good` (task/`pi/task`) — retain their conversation across turns via the subagent keep-alive lifecycle, deliver each turn's result asynchronously with a compressed tool-call trace plus the worker's response, and are killed when the mode exits. The TUI renders sends as a mini CLI composer and waits as a live stacked-screen view of every worker's tool calls and streamed output. - `omp acp` now prints a short hint on stderr when launched from an interactive terminal (stdin is a TTY): the command speaks JSON-RPC over stdout and is meant to be spawned by an ACP client such as Zed, so running it by hand previously showed nothing at all - Added a credential-free Google web search provider that loads rendered Search results in stealth Chromium, parses titles, URLs, and snippets, and supports recency filters. +- Added five credential-free web search providers scraped from public engines: Bing, Yahoo, and Startpage over plain fetch, plus Ecosia and Mojeek with stealth-browser escalation (Cloudflare and ALTCHA proof-of-work walls); all detect bot challenges and map recency filters where the engine supports them. +- Added a `public` ("Public Web") web search provider that fans out to every credential-free engine in parallel and consolidates results — URLs deduplicated across engines, ranked by cross-engine consensus then best per-engine rank — returning at the earliest of all engines settled, a 5s soft deadline with at least one success, or a 30s hard cap. Explicit selection only; never auto-selected. - Added PCRE2 fallback for grep lookaround and backreferences when Rust regex rejects a pattern. ### Changed @@ -16,14 +18,15 @@ - Refined agent delegation logic to prioritize top-level planning and scoping by the primary agent - Optimized subagent usage to discourage single-agent delegation and improve parallel execution flows - Clarified that prerequisite work for subagent tasks should be handled inline by the main agent +- HTML search requests now use a randomized, internally consistent desktop Chrome profile per request; Google, Ecosia, and Mojeek try fetch first and escalate blocked or failed production responses to the shared stealth browser. +- Reordered credential-free web search engines from live quality measurements: Startpage (Google-backed, fastest and most reliable in testing) now leads the auto chain, Ecosia precedes the flakier browser-backed Google, and Mojeek stays last; the Public Web fan-out tiebreak now favors Google-index engines (Startpage, Google) so their ranking wins equal-consensus ties. ### Removed - Removed the bundled `plan` subagent from available task agents -- Removed the bundled `plan` subagent from available task agents. ### Fixed - +- Fixed the `omp search` / `omp q` CLI command hanging instead of exiting cleanly after search execution by ensuring the internally discovered `AuthStorage` connection is properly closed. - Fixed `write` blocking for the full 3-second LSP diagnostics poll in main-agent sessions by wiring it into the deferred late-diagnostics channel; slow diagnostics now return after the short inline window and arrive as an aside. - Fixed `tab.fill`/`tab.click` (and every puppeteer Locator action) timing out after 15s on all pages: the stealth patch routes default `Frame.evaluate`/`waitForFunction` through the isolated world, but `waitForSelector`/Locator results were still transferred to the main world, so Locator's enabled-precondition (`handle.frame.waitForFunction(pred, opts, handle)`) and `page.evaluate(fn, handle)` threw a cross-context handle error that Locators retried silently until timeout. `QueryHandler.waitFor` now returns its result in the isolated world, matching the patched default realm; explicit `//!world=main` evaluation still adopts handles via ElementHandle - Fixed browser runs silently dropping `display("string")`, `console.log`, and `print` output: the runtime emits those as stream text, which the browser embedders (worker and cmux) routed to the debug log only, so the tool result showed a bare "Ran code on tab". Stream text is now buffered and surfaced as ordered display entries alongside `display()` payloads and screenshots. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index e84b6bac7..3e74c712d 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -81,6 +81,7 @@ "diff": "catalog:", "fast-xml-parser": "catalog:", "handlebars": "catalog:", + "header-generator": "catalog:", "linkedom": "catalog:", "lru-cache": "catalog:", "mammoth": "catalog:", diff --git a/packages/coding-agent/src/web/search/provider.ts b/packages/coding-agent/src/web/search/provider.ts index 0a043a1a9..4c2ee6f28 100644 --- a/packages/coding-agent/src/web/search/provider.ts +++ b/packages/coding-agent/src/web/search/provider.ts @@ -119,6 +119,36 @@ const PROVIDER_META: Record = { label: SEARCH_PROVIDER_LABELS.google, load: async () => new (await import("./providers/google")).GoogleProvider(), }, + bing: { + id: "bing", + label: SEARCH_PROVIDER_LABELS.bing, + load: async () => new (await import("./providers/bing")).BingProvider(), + }, + yahoo: { + id: "yahoo", + label: SEARCH_PROVIDER_LABELS.yahoo, + load: async () => new (await import("./providers/yahoo")).YahooProvider(), + }, + ecosia: { + id: "ecosia", + label: SEARCH_PROVIDER_LABELS.ecosia, + load: async () => new (await import("./providers/ecosia")).EcosiaProvider(), + }, + startpage: { + id: "startpage", + label: SEARCH_PROVIDER_LABELS.startpage, + load: async () => new (await import("./providers/startpage")).StartpageProvider(), + }, + mojeek: { + id: "mojeek", + label: SEARCH_PROVIDER_LABELS.mojeek, + load: async () => new (await import("./providers/mojeek")).MojeekProvider(), + }, + public: { + id: "public", + label: SEARCH_PROVIDER_LABELS.public, + load: async () => new (await import("./providers/public")).PublicWebProvider(), + }, }; const instanceCache = new Map(); @@ -185,7 +215,8 @@ export function setExcludedSearchProviders(providers: readonly SearchProviderId[ excludedProvIds = new Set(providers); } -function isSearchProviderExcluded(id: SearchProviderId): boolean { +/** `true` when settings exclude `id` from web search (auto chain and the Public Web fan-out). */ +export function isSearchProviderExcluded(id: SearchProviderId): boolean { return excludedProvIds.has(id); } diff --git a/packages/coding-agent/src/web/search/providers/bing.ts b/packages/coding-agent/src/web/search/providers/bing.ts new file mode 100644 index 000000000..aaad2d427 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/bing.ts @@ -0,0 +1,197 @@ +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import { parseHTML } from "linkedom"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { browserFetch } from "./browser-page"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +/** + * Bing's HTML search frontend. A plain GET with browser navigation headers + * returns a fully server-rendered results page — no JavaScript challenge on + * the organic path — so we parse it directly without a real browser. + */ +const BING_HOME_URL = "https://www.bing.com/"; +const BING_SEARCH_URL = "https://www.bing.com/search"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; +const MS_PER_DAY = 86_400_000; + +/** + * Recency → Bing `filters=ex1:"…"` freshness codes, as emitted by Bing's own + * "Any time" dropdown. `year` has no fixed code; the dropdown emits a custom + * epoch-day range (`ez5__`, days since 1970-01-01) which + * {@link recencyToFilters} computes. Bing parses the parameter (the SERP + * filter UI reflects it) but enforcement is server-side and vantage-dependent. + */ +const RECENCY_TO_BING_EZ: Record, "year">, string> = { + day: "ez1", + week: "ez2", + month: "ez3", +}; + +/** Snippet containers observed on Bing result blocks, in preference order. */ +const BING_SNIPPET_SELECTORS: readonly string[] = [".b_caption p", "p[class*='b_lineclamp']", ".b_algoSlug"]; + +interface ParsedResult { + title: string; + url: string; + snippet?: string; +} + +/** Build the `filters` value for a recency window, mirroring Bing's dropdown URLs. */ +function recencyToFilters(recency: NonNullable): string { + if (recency === "year") { + const epochDay = Math.floor(Date.now() / MS_PER_DAY); + return `ex1:"ez5_${epochDay - 365}_${epochDay}"`; + } + return `ex1:"${RECENCY_TO_BING_EZ[recency]}"`; +} + +/** + * Resolve a Bing result href to the underlying target URL. + * + * Organic hrefs are usually wrapped as `https://www.bing.com/ck/a?…&u=a1` + * where the payload after the literal `a1` prefix is the unpadded base64url + * encoding of the target URL. Direct external hrefs also occur; Bing-internal + * links (vertical tabs, ads plumbing) and non-http(s) schemes are rejected. + */ +function unwrapResultUrl(href: string): string | undefined { + let url: URL; + try { + url = new URL(href, BING_HOME_URL); + } catch { + return undefined; + } + + if (url.hostname === "bing.com" || url.hostname.endsWith(".bing.com")) { + if (url.pathname !== "/ck/a") return undefined; + const wrapped = url.searchParams.get("u"); + if (!wrapped?.startsWith("a1")) return undefined; + try { + url = new URL(Buffer.from(wrapped.slice(2), "base64url").toString("utf-8")); + } catch { + return undefined; + } + } + + if (url.protocol !== "http:" && url.protocol !== "https:") return undefined; + return url.href; +} + +function findSnippet(item: Element): string | undefined { + for (const selector of BING_SNIPPET_SELECTORS) { + const text = (item.querySelector(selector)?.textContent ?? "").replace(/\s+/g, " ").trim(); + if (text) return text; + } + return undefined; +} + +/** + * Pull organic result blocks out of the page in document order. + * + * Each organic hit is an `
  • ` with the title link in + * `h2 > a[href]` (sitelink/attribution anchors live outside the `h2`) and the + * preview text in one of {@link BING_SNIPPET_SELECTORS}. Ads, answer cards, + * and the "no results" row use other classes and fall out naturally. + */ +function parseHtmlResults(html: string): ParsedResult[] { + const { document } = parseHTML(html); + const results: ParsedResult[] = []; + for (const item of document.querySelectorAll("li.b_algo")) { + const anchor = item.querySelector("h2 a[href]"); + const href = anchor?.getAttribute("href"); + if (!href) continue; + const url = unwrapResultUrl(href); + if (!url) continue; + const title = (anchor?.textContent ?? "").replace(/\s+/g, " ").trim(); + if (!title) continue; + results.push({ title, url, snippet: findSnippet(item) }); + } + return results; +} + +/** + * `true` when Bing answered with its CAPTCHA/consent interstitial instead of + * a results page. The challenge redirects to `/turing/captcha/…`; body + * markers are only trusted when no organic result block is present so a + * search *about* CAPTCHAs never trips the detector. + */ +function isChallengeResponse(html: string, finalUrl: string): boolean { + if (finalUrl.includes("/turing/captcha")) return true; + if (html.includes('class="b_algo"')) return false; + return /turing\/captcha|b_captcha|px-captcha|verify (?:that )?you are (?:a )?human/i.test(html); +} + +function buildSearchUrl(params: SearchParams, numResults: number): string { + const url = new URL(BING_SEARCH_URL); + url.searchParams.set("q", params.query); + url.searchParams.set("count", String(numResults)); + url.searchParams.set("mkt", "en-US"); + url.searchParams.set("setlang", "en"); + if (params.recency) url.searchParams.set("filters", recencyToFilters(params.recency)); + return url.href; +} + +async function callBingHtml(params: SearchParams, numResults: number): Promise { + const url = buildSearchUrl(params, numResults); + const page = await browserFetch(url, { + fetch: params.fetch ?? fetch, + signal: withHardTimeout(params.signal), + referer: BING_HOME_URL, + }); + + const body = page.html; + if (isChallengeResponse(body, page.url)) { + throw new SearchProviderError( + "bing", + "Bing blocked the request with a CAPTCHA challenge. Bing throttles automated searches from datacenter/shared-egress IPs; try the duckduckgo or mojeek provider, or configure a credentialed provider such as Brave, Tavily, Exa, or Kagi.", + 429, + ); + } + if (page.status < 200 || page.status >= 300) { + const classified = classifyProviderHttpError("bing", page.status, body); + if (classified) throw classified; + throw new SearchProviderError("bing", `Bing HTML error (${page.status})`, page.status); + } + + return body; +} + +/** Execute a Bing web search via the server-rendered HTML results page. */ +export async function searchBing(params: SearchParams): Promise { + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const html = await callBingHtml(params, numResults); + const parsed = parseHtmlResults(html); + + const sources: SearchSource[] = []; + const seen = new Set(); + for (const result of parsed) { + if (seen.has(result.url)) continue; + seen.add(result.url); + sources.push({ title: result.title, url: result.url, snippet: result.snippet }); + if (sources.length >= numResults) break; + } + + return { provider: "bing", sources }; +} + +/** Search provider for Bing (no API key required). */ +export class BingProvider extends SearchProvider { + readonly id = "bing"; + readonly label = "Bing"; + + isAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchBing(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/browser-headers.ts b/packages/coding-agent/src/web/search/providers/browser-headers.ts index ecb6e8052..a3a5c578a 100644 --- a/packages/coding-agent/src/web/search/providers/browser-headers.ts +++ b/packages/coding-agent/src/web/search/providers/browser-headers.ts @@ -1,5 +1,20 @@ -/** Chromium document-navigation headers shared by HTML search providers. */ -export const BROWSER_NAVIGATION_HEADERS: Readonly> = { +import { HeaderGenerator } from "header-generator"; + +// Instantiate the singleton header generator. +// This matches modern browsers from real-world query statistics +// and randomizes between Chrome, Firefox, Safari, and Edge. +const generator = new HeaderGenerator({ + browserListQuery: "last 3 versions", + devices: ["desktop"], + operatingSystems: ["windows", "macos", "linux"], + locales: ["en-US", "en"], + httpVersion: "2", + strict: false, +}); + +// A fallback desktop Mac Chrome navigation fingerprint matching +// the previous static default setup for deterministic or non-randomized calls. +const CHROME_FALLBACK_HEADERS: Record = { Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7", "Accept-Language": "en-US,en;q=0.9", @@ -16,3 +31,62 @@ export const BROWSER_NAVIGATION_HEADERS: Readonly> = { "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36", }; + +function canonicalizeHeaderNames(headers: Record): Record { + const canonicalized: Record = {}; + + for (const key in headers) { + const value = headers[key]; + if (value === undefined) continue; + + // Retain Client Hints (sec-ch-ua*) in their standard lower-case representation + if (key.startsWith("sec-ch-ua")) { + canonicalized[key] = value; + continue; + } + + // Retain diagnostics or other HTTP/2 custom lower-case keys + if (["dnt", "rtt", "ect"].includes(key)) { + canonicalized[key.toUpperCase()] = value; + continue; + } + + // Retain HTTP/2 specific pseudo headers if any, or general standard casing overrides + if (key === "te") { + canonicalized.TE = value; + continue; + } + + // Pascalize words separated by hyphens (e.g. accept-language -> Accept-Language) + const pascalized = key + .split("-") + .map(part => (part[0] ? part[0].toUpperCase() + part.slice(1).toLowerCase() : "")) + .join("-"); + + canonicalized[pascalized] = value; + } + + return canonicalized; +} + +/** + * Build a fresh, internally consistent desktop navigation fingerprint for one HTTP request. + * By default, this randomizes across valid modern versions of Chrome, Firefox, Edge, and Safari + * using real-world traffic data. Set `randomized` to `false` when a fetch must preserve a + * stable Mac Chrome identity. + */ +export function buildBrowserNavigationHeaders(options?: { randomized?: boolean }): Record { + const randomized = options?.randomized !== false; + if (!randomized) { + return { ...CHROME_FALLBACK_HEADERS }; + } + + try { + // Generate realistic, consistent headers with the Bayesian generator + const generated = generator.getHeaders(); + return canonicalizeHeaderNames(generated); + } catch { + // Gracefully recover to the robust default profile on unexpected generator errors + return { ...CHROME_FALLBACK_HEADERS }; + } +} diff --git a/packages/coding-agent/src/web/search/providers/browser-page.ts b/packages/coding-agent/src/web/search/providers/browser-page.ts new file mode 100644 index 000000000..2be17fbca --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/browser-page.ts @@ -0,0 +1,123 @@ +import type { FetchImpl } from "@oh-my-pi/pi-ai"; +import { untilAborted } from "@oh-my-pi/pi-utils"; +import type { Page } from "puppeteer-core"; +import { applyStealthPatches, applyViewport } from "../../../tools/browser/launch"; +import { acquireBrowser, holdBrowser, releaseBrowser } from "../../../tools/browser/registry"; +import { buildBrowserNavigationHeaders } from "./browser-headers"; +import { SEARCH_HARD_TIMEOUT_MS } from "./utils"; + +/** HTML plus the response status and final URL after redirects or browser navigation. */ +export interface LoadedHtmlPage { + html: string; + status: number; + url: string; +} + +interface BrowserFallbackOptions { + homeUrl?: string; + ready?: { selector: string; timeoutMs: number }; + afterNavigation?: (page: Page, signal: AbortSignal) => Promise; + shouldFallback: (page: LoadedHtmlPage) => boolean; + attempts?: number; + retryDelayMs?: number; +} + +/** Controls a browser-profiled fetch and its optional headless-browser fallback. */ +export interface BrowserFetchOptions { + fetch?: FetchImpl; + signal: AbortSignal; + randomizeHeaders?: boolean; + referer?: string; + init?: Omit; + headers?: Readonly>; + browser?: BrowserFallbackOptions; +} + +async function fetchHtmlPage(url: string, options: BrowserFetchOptions, fetchImpl: FetchImpl): Promise { + const response = await fetchImpl(url, { + ...options.init, + headers: { + ...buildBrowserNavigationHeaders({ randomized: options.randomizeHeaders }), + ...(options.referer ? { Referer: options.referer, "Sec-Fetch-Site": "same-origin" } : {}), + ...options.headers, + }, + signal: options.signal, + }); + return { html: await response.text(), status: response.status, url: response.url || url }; +} + +async function browseHtmlPage( + url: string, + options: BrowserFallbackOptions, + signal: AbortSignal, +): Promise { + const { homeUrl, ready } = options; + const attempts = Math.max(1, options.attempts ?? 1); + const handle = await untilAborted(signal, () => + acquireBrowser( + { kind: "headless", headless: true }, + { + cwd: process.cwd(), + signal, + }, + ), + ); + if (!("browser" in handle)) { + await releaseBrowser(handle, { kill: false }); + throw new Error("Headless browser acquisition returned a non-Puppeteer browser"); + } + + holdBrowser(handle); + let page: Page | undefined; + try { + const activePage = await untilAborted(signal, () => handle.browser.newPage()); + page = activePage; + await applyViewport(activePage); + await applyStealthPatches(handle.browser, activePage, handle.stealth); + if (homeUrl) { + await untilAborted(signal, () => + activePage.goto(homeUrl, { waitUntil: "domcontentloaded", timeout: SEARCH_HARD_TIMEOUT_MS }), + ); + } + for (let attempt = 0; attempt < attempts; attempt++) { + if (attempt > 0 && options.retryDelayMs) await Bun.sleep(options.retryDelayMs); + + const response = await untilAborted(signal, () => + activePage.goto(url, { waitUntil: "domcontentloaded", timeout: SEARCH_HARD_TIMEOUT_MS }), + ); + if (options.afterNavigation) await options.afterNavigation(activePage, signal); + if (ready) { + await untilAborted(signal, () => + activePage.waitForSelector(ready.selector, { timeout: ready.timeoutMs }).catch(() => null), + ); + } + const loaded = { + html: await untilAborted(signal, () => activePage.content()), + status: response?.status() ?? 200, + url: activePage.url(), + }; + if (!options.shouldFallback(loaded) || attempt === attempts - 1) return loaded; + } + throw new Error("Browser fallback exhausted without a response"); + } finally { + await page?.close().catch(() => undefined); + await releaseBrowser(handle, { kill: false }); + } +} + +/** Fetch with a fresh browser profile, escalating rejected production responses to the stealth browser. */ +export async function browserFetch(url: string, options: BrowserFetchOptions): Promise { + const fetchImpl = options.fetch ?? fetch; + let page: LoadedHtmlPage; + try { + page = await fetchHtmlPage(url, options, fetchImpl); + } catch (error) { + if (options.fetch || !options.browser) throw error; + return browseHtmlPage(url, options.browser, options.signal); + } + + if (!options.browser || options.fetch) return page; + const isSuccessful = page.status >= 200 && page.status < 300; + if (isSuccessful && !options.browser.shouldFallback(page)) return page; + return browseHtmlPage(url, options.browser, options.signal); +} diff --git a/packages/coding-agent/src/web/search/providers/duckduckgo.ts b/packages/coding-agent/src/web/search/providers/duckduckgo.ts index 5b19abb0c..c77972375 100644 --- a/packages/coding-agent/src/web/search/providers/duckduckgo.ts +++ b/packages/coding-agent/src/web/search/providers/duckduckgo.ts @@ -4,7 +4,7 @@ import { SearchProviderError } from "../../../web/search/types"; import { clampNumResults } from "../utils"; import type { SearchParams } from "./base"; import { SearchProvider } from "./base"; -import { BROWSER_NAVIGATION_HEADERS } from "./browser-headers"; +import { browserFetch } from "./browser-page"; import { classifyProviderHttpError, withHardTimeout } from "./utils"; /** @@ -125,23 +125,22 @@ async function callDuckDuckGoHtml(params: SearchParams): Promise { // Add b: "" parameter as specified in the browser fetch template to match real browser form submission form.set("b", ""); - const response = await (params.fetch ?? fetch)(DUCKDUCKGO_HTML_URL, { - method: "POST", - body: form.toString(), - headers: { - ...BROWSER_NAVIGATION_HEADERS, - "Content-Type": "application/x-www-form-urlencoded", - "Sec-Fetch-Site": "same-origin", - Referer: "https://html.duckduckgo.com/", - }, + const page = await browserFetch(DUCKDUCKGO_HTML_URL, { + fetch: params.fetch ?? fetch, signal: withHardTimeout(params.signal), + referer: "https://html.duckduckgo.com/", + init: { + method: "POST", + body: form.toString(), + }, + headers: { "Content-Type": "application/x-www-form-urlencoded" }, }); - const body = await response.text(); - if (!response.ok && response.status !== 202) { - const classified = classifyProviderHttpError("duckduckgo", response.status, body); + const body = page.html; + if (page.status < 200 || page.status >= 300) { + const classified = classifyProviderHttpError("duckduckgo", page.status, body); if (classified) throw classified; - throw new SearchProviderError("duckduckgo", `DuckDuckGo HTML error (${response.status})`, response.status); + throw new SearchProviderError("duckduckgo", `DuckDuckGo HTML error (${page.status})`, page.status); } if (isAnomalyResponse(body)) { diff --git a/packages/coding-agent/src/web/search/providers/ecosia.ts b/packages/coding-agent/src/web/search/providers/ecosia.ts new file mode 100644 index 000000000..a41cca27a --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/ecosia.ts @@ -0,0 +1,178 @@ +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import { parseHTML } from "linkedom"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import type { LoadedHtmlPage } from "./browser-page"; +import { browserFetch } from "./browser-page"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +/** + * Ecosia serves a server-rendered Vue/Nuxt results page (no `__NUXT_DATA__` + * JSON island — probed 2026-07), so both load paths parse the same markup: + * `
    ` blocks whose title anchor carries + * the final target URL directly (no redirect wrapper). The site fronts search + * with Cloudflare. Requests start with a browser-profiled fetch and escalate + * to the shared stealth browser only when the response is blocked or fails. + * + * Recency is ignored: Ecosia's web results expose no date filter in the UI + * and the legacy Bing-era `freshness` param is a server-side no-op (verified + * live), so per the {@link SearchParams.recency} contract the field must not + * be approximated. + */ +const ECOSIA_HOME_URL = "https://www.ecosia.org/"; +const ECOSIA_SEARCH_URL = "https://www.ecosia.org/search"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; +const RESULT_RENDER_TIMEOUT_MS = 10_000; + +interface ParsedResult { + title: string; + url: string; + snippet?: string; +} + +/** + * Validate a result anchor href. Organic anchors carry the target URL + * directly; anything non-http(s) or pointing back at ecosia.org (internal + * navigation such as the images/news verticals) is rejected. + */ +function resolveResultUrl(href: string): string | undefined { + let url: URL; + try { + url = new URL(href, ECOSIA_HOME_URL); + } catch { + return undefined; + } + if (url.protocol !== "http:" && url.protocol !== "https:") return undefined; + if (url.hostname === "ecosia.org" || url.hostname === "www.ecosia.org") return undefined; + return url.href; +} + +/** + * Pull organic results out of the page in document order. Each result is an + * `
    ` with the title inside + * `

    ` wrapped by the target link, and the + * preview text inside `

    `. The inner + * paragraph is preferred over its `result-description` container because the + * container also holds screen-reader-only thumbnail captions on video rows. + * Ad slots (`data-test-id="ad-google"`) and entity/infobox cards use + * different test-ids and never match. + */ +function parseHtmlResults(html: string): ParsedResult[] { + const { document } = parseHTML(html); + const results: ParsedResult[] = []; + for (const article of document.querySelectorAll('article[data-test-id="organic-result"]')) { + const heading = article.querySelector('[data-test-id="result-title"]'); + const href = heading?.closest("a")?.getAttribute("href"); + if (!heading || !href) continue; + const url = resolveResultUrl(href); + if (!url) continue; + const title = (heading.textContent ?? "").replace(/\s+/g, " ").trim(); + if (!title) continue; + const description = + article.querySelector('[data-test-id="web-result-description"]') ?? + article.querySelector('[data-test-id="result-description"]'); + const snippet = (description?.textContent ?? "").replace(/\s+/g, " ").trim(); + results.push({ title, url, snippet: snippet || undefined }); + } + return results; +} + +/** + * `true` when Ecosia's Cloudflare front answered with the managed challenge + * instead of results. The observed page is a 403 titled "Ecosia Firewall" + * carrying the `_cf_chl_opt` bootstrap and the challenge-platform loader. + */ +function isBlockedPage(page: LoadedHtmlPage): boolean { + return ( + page.status === 403 || + page.status === 429 || + page.html.includes("Ecosia Firewall") || + page.html.includes("_cf_chl_opt") || + page.html.includes("/cdn-cgi/challenge-platform/") || + /confirm you.{0,3}re not a robot/i.test(page.html) + ); +} + +async function callEcosiaHtml(params: SearchParams): Promise { + const signal = withHardTimeout(params.signal); + const url = new URL(ECOSIA_SEARCH_URL); + url.searchParams.set("q", params.query); + + let page: LoadedHtmlPage; + try { + page = await browserFetch(url.href, { + fetch: params.fetch, + signal, + referer: ECOSIA_HOME_URL, + browser: { + homeUrl: ECOSIA_HOME_URL, + ready: { + selector: 'article[data-test-id="organic-result"]', + timeoutMs: RESULT_RENDER_TIMEOUT_MS, + }, + shouldFallback: isBlockedPage, + }, + }); + } catch (error) { + if (error instanceof SearchProviderError || params.signal?.aborted) throw error; + if (signal.aborted) { + throw new SearchProviderError("ecosia", "Ecosia search timed out.", 504); + } + const message = error instanceof Error ? error.message : String(error); + throw new SearchProviderError("ecosia", `Ecosia search failed: ${message}`, 503); + } + + if (isBlockedPage(page)) { + throw new SearchProviderError( + "ecosia", + "Ecosia blocked the request with a Cloudflare bot challenge. Ecosia's firewall throttles automated searches from datacenter/shared-egress IPs; try another web search provider such as DuckDuckGo, Brave, or Tavily.", + 429, + ); + } + if (page.status < 200 || page.status >= 300) { + const classified = classifyProviderHttpError("ecosia", page.status, page.html); + if (classified) throw classified; + throw new SearchProviderError("ecosia", `Ecosia HTML error (${page.status})`, page.status); + } + return page.html; +} + +/** Execute an Ecosia web search and parse the server-rendered result page. */ +export async function searchEcosia(params: SearchParams): Promise { + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const html = await callEcosiaHtml(params); + const parsed = parseHtmlResults(html); + + const sources: SearchSource[] = []; + const seen = new Set(); + for (const result of parsed) { + if (seen.has(result.url)) continue; + seen.add(result.url); + sources.push({ title: result.title, url: result.url, snippet: result.snippet }); + if (sources.length >= numResults) break; + } + + return { provider: "ecosia", sources }; +} + +/** Search provider for Ecosia (no API key required). */ +export class EcosiaProvider extends SearchProvider { + readonly id = "ecosia"; + readonly label = "Ecosia"; + + isAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchEcosia(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/google.ts b/packages/coding-agent/src/web/search/providers/google.ts index 67a3aa19b..574f2dcaf 100644 --- a/packages/coding-agent/src/web/search/providers/google.ts +++ b/packages/coding-agent/src/web/search/providers/google.ts @@ -1,16 +1,13 @@ -import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; -import { untilAborted } from "@oh-my-pi/pi-utils"; +import type { AuthStorage } from "@oh-my-pi/pi-ai"; import { parseHTML } from "linkedom"; -import type { Page } from "puppeteer-core"; -import { applyStealthPatches, applyViewport } from "../../../tools/browser/launch"; -import { acquireBrowser, holdBrowser, releaseBrowser } from "../../../tools/browser/registry"; import type { SearchResponse, SearchSource } from "../../../web/search/types"; import { SearchProviderError } from "../../../web/search/types"; import { clampNumResults } from "../utils"; import type { SearchParams } from "./base"; import { SearchProvider } from "./base"; -import { BROWSER_NAVIGATION_HEADERS } from "./browser-headers"; -import { SEARCH_HARD_TIMEOUT_MS, withHardTimeout } from "./utils"; +import type { LoadedHtmlPage } from "./browser-page"; +import { browserFetch } from "./browser-page"; +import { withHardTimeout } from "./utils"; const GOOGLE_HOME_URL = "https://www.google.com/"; const GOOGLE_SEARCH_URL = "https://www.google.com/search"; @@ -38,12 +35,6 @@ interface ParsedResult { snippet?: string; } -interface LoadedGooglePage { - html: string; - status: number; - url: string; -} - function normalizeText(value: string | null | undefined): string { return (value ?? "").replace(/\s+/g, " ").trim(); } @@ -111,76 +102,34 @@ function buildSearchUrl(params: SearchParams, numResults: number): string { return url.href; } -async function loadWithFetch(url: string, fetchImpl: FetchImpl, signal: AbortSignal): Promise { - const response = await fetchImpl(url, { - headers: { - ...BROWSER_NAVIGATION_HEADERS, - Referer: GOOGLE_HOME_URL, - "Sec-Fetch-Site": "same-origin", - }, - signal, - }); - return { html: await response.text(), status: response.status, url: response.url || url }; -} - -async function loadWithBrowser(url: string, signal: AbortSignal): Promise { - const handle = await untilAborted(signal, () => - acquireBrowser( - { kind: "headless", headless: true }, - { - cwd: process.cwd(), - signal, - }, - ), - ); - if (!("browser" in handle)) { - await releaseBrowser(handle, { kill: false }); - throw new Error("Headless browser acquisition returned a non-Puppeteer browser"); - } - - holdBrowser(handle); - let page: Page | undefined; - try { - const activePage = await untilAborted(signal, () => handle.browser.newPage()); - page = activePage; - await applyViewport(activePage); - await applyStealthPatches(handle.browser, activePage, handle.stealth); - // Seed Google's same-origin cookies and referrer; a cold direct navigation gets the enable-JavaScript interstitial. - await untilAborted(signal, () => - activePage.goto(GOOGLE_HOME_URL, { waitUntil: "domcontentloaded", timeout: SEARCH_HARD_TIMEOUT_MS }), - ); - const response = await untilAborted(signal, () => - activePage.goto(url, { waitUntil: "domcontentloaded", timeout: SEARCH_HARD_TIMEOUT_MS }), - ); - await untilAborted(signal, () => - activePage.waitForSelector("a h3", { timeout: RESULT_RENDER_TIMEOUT_MS }).catch(() => null), - ); - return { - html: await untilAborted(signal, () => activePage.content()), - status: response?.status() ?? 200, - url: activePage.url(), - }; - } finally { - await page?.close().catch(() => undefined); - await releaseBrowser(handle, { kill: false }); - } -} - -function isBlockedPage(page: LoadedGooglePage): boolean { - return ( +function blockReason(page: LoadedHtmlPage): "javascript" | "traffic" | undefined { + if (page.html.includes("/httpservice/retry/enablejs") && !/ { const signal = withHardTimeout(params.signal); const url = buildSearchUrl(params, numResults); - let page: LoadedGooglePage; + let page: LoadedHtmlPage; try { - page = params.fetch ? await loadWithFetch(url, params.fetch, signal) : await loadWithBrowser(url, signal); + page = await browserFetch(url, { + fetch: params.fetch, + signal, + referer: GOOGLE_HOME_URL, + browser: { + homeUrl: GOOGLE_HOME_URL, + ready: { selector: "a h3", timeoutMs: RESULT_RENDER_TIMEOUT_MS }, + shouldFallback: candidate => blockReason(candidate) !== undefined, + }, + }); } catch (error) { if (error instanceof SearchProviderError || params.signal?.aborted) throw error; if (signal.aborted) { @@ -190,7 +139,8 @@ async function callGoogleHtml(params: SearchParams, numResults: number): Promise throw new SearchProviderError("google", `Google browser search failed: ${message}`, 503); } - if (isBlockedPage(page)) { + const blocked = blockReason(page); + if (blocked === "traffic") { throw new SearchProviderError( "google", "Google blocked the browser search with an automated-traffic challenge. Try another web search provider or retry later.", @@ -200,7 +150,7 @@ async function callGoogleHtml(params: SearchParams, numResults: number): Promise if (page.status < 200 || page.status >= 300) { throw new SearchProviderError("google", `Google HTML error (${page.status})`, page.status); } - if (page.html.includes("/httpservice/retry/enablejs") && !/ { const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); const html = await callGoogleHtml(params, numResults); @@ -228,7 +178,7 @@ export async function searchGoogle(params: SearchParams): Promise li` with the title in + * `h2 > a.title` (href is the direct target URL) and the preview text in + * `p.s`. Clustered sub-results (`li.clu-result`) share the same shape; rows + * without a title anchor (infoboxes, spelling suggestions) are skipped. + */ +function parseHtmlResults(html: string): ParsedResult[] { + const { document } = parseHTML(html); + const results: ParsedResult[] = []; + for (const item of document.querySelectorAll("ul.results-standard > li")) { + const anchor = item.querySelector("h2 a.title") ?? item.querySelector("a.title"); + const href = anchor?.getAttribute("href"); + if (!href) continue; + const url = normalizeResultUrl(href); + if (!url) continue; + const title = (anchor?.textContent ?? "").replace(/\s+/g, " ").trim(); + if (!title) continue; + const snippet = (item.querySelector("p.s")?.textContent ?? "").replace(/\s+/g, " ").trim(); + results.push({ title, url, snippet: snippet || undefined }); + } + return results; +} + +function buildSearchUrl(params: SearchParams, numResults: number): string { + const url = new URL(MOJEEK_SEARCH_URL); + url.searchParams.set("q", params.query); + url.searchParams.set("t", String(numResults)); + url.searchParams.set("arc", "none"); + url.searchParams.set("lang", "en"); + url.searchParams.set("lb", "en"); + url.searchParams.set("theme", "dark"); + // Mojeek's `since` filter accepts the relative tokens day/week/month/year + // verbatim — the same vocabulary as `recency` (verified live: each window + // returns a near-disjoint, fresher result set). Dates reflect crawl or + // last-modification time per Mojeek's operator docs. + if (params.recency) url.searchParams.set("since", params.recency); + return url.href; +} + +/** Solve Mojeek's ALTCHA interstitial and wait for its verified redirect to populate results. */ +async function solveCaptcha(page: Page, signal: AbortSignal): Promise { + if (await untilAborted(signal, () => page.$("ul.results-standard li"))) return; + + const checkbox = await untilAborted(signal, () => page.$("altcha-widget input[type=checkbox]")); + if (!checkbox) return; + + const navigation = page + .waitForNavigation({ waitUntil: "domcontentloaded", timeout: CAPTCHA_SOLVE_TIMEOUT_MS }) + .catch(() => null); + await untilAborted(signal, () => checkbox.click()); + await untilAborted(signal, () => navigation); + await untilAborted(signal, () => + page.waitForSelector("ul.results-standard li", { timeout: CAPTCHA_SOLVE_TIMEOUT_MS }).catch(() => null), + ); +} + +function isRobotPage(page: LoadedHtmlPage): boolean { + return ( + (page.html.includes("altcha-widget") || + page.html.includes("captcha-wrap") || + /sending automated queries/i.test(page.html)) && + !page.html.includes("results-standard") + ); +} + +async function callMojeekHtml(params: SearchParams, numResults: number): Promise { + const signal = withHardTimeout(params.signal); + const url = buildSearchUrl(params, numResults); + let page: LoadedHtmlPage; + try { + page = await browserFetch(url, { + fetch: params.fetch, + signal, + randomizeHeaders: false, + referer: MOJEEK_HOME_URL, + browser: { + homeUrl: MOJEEK_HOME_URL, + afterNavigation: solveCaptcha, + shouldFallback: isRobotPage, + attempts: 2, + retryDelayMs: 1_000, + }, + }); + } catch (error) { + if (error instanceof SearchProviderError || params.signal?.aborted) throw error; + if (signal.aborted) { + throw new SearchProviderError("mojeek", "Mojeek search timed out.", 504); + } + const message = error instanceof Error ? error.message : String(error); + throw new SearchProviderError("mojeek", `Mojeek search failed: ${message}`, 503); + } + + // Robot walls: the ALTCHA proof-of-work captcha page arrives as HTTP 200 + // (`Captcha`, `altcha-widget`) and the "automated queries" + // refusal as HTTP 403. Both bodies are more actionable than their raw + // statuses, so check them before the generic status handling. + if (isRobotPage(page)) { + throw new SearchProviderError( + "mojeek", + "Mojeek blocked the request with its automated-queries wall. Mojeek rate-limits scripted searches from datacenter/shared-egress IPs; retry later or configure another provider such as Brave, Tavily, Exa, or Kagi.", + 429, + ); + } + if (page.status < 200 || page.status >= 300) { + const classified = classifyProviderHttpError("mojeek", page.status, page.html); + if (classified) throw classified; + throw new SearchProviderError("mojeek", `Mojeek HTML error (${page.status})`, page.status); + } + return page.html; +} + +/** Execute a Mojeek web search against the standard HTML results page. */ +export async function searchMojeek(params: SearchParams): Promise { + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const html = await callMojeekHtml(params, numResults); + const parsed = parseHtmlResults(html); + + const sources: SearchSource[] = []; + const seen = new Set(); + for (const result of parsed) { + if (seen.has(result.url)) continue; + seen.add(result.url); + sources.push({ title: result.title, url: result.url, snippet: result.snippet }); + if (sources.length >= numResults) break; + } + + return { provider: "mojeek", sources }; +} + +/** Search provider for Mojeek (independent index, no API key required). */ +export class MojeekProvider extends SearchProvider { + readonly id = "mojeek"; + readonly label = "Mojeek"; + + isAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchMojeek(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/public.ts b/packages/coding-agent/src/web/search/providers/public.ts new file mode 100644 index 000000000..4e0fb7ce8 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/public.ts @@ -0,0 +1,201 @@ +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import { formatSearchProviderFailures, getSearchProvider, isSearchProviderExcluded } from "../provider"; +import type { SearchProviderId, SearchResponse, SearchSource } from "../types"; +import { SearchProviderError } from "../types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { withHardTimeout } from "./utils"; + +/** + * Credential-free engines the Public Web aggregate fans out to. Order is the + * tiebreak for merged ranking (earlier engines win equal consensus/rank), so + * engines with the best ranking quality when they answer come first: + * Google-index engines (startpage, google) lead, Bing-backed scrapers follow, + * and Mojeek's independent index breaks remaining ties (measured 2026-07). + */ +const PUBLIC_ENGINE_IDS = [ + "startpage", + "google", + "duckduckgo", + "bing", + "yahoo", + "ecosia", + "mojeek", +] as const satisfies readonly SearchProviderId[]; + +/** Aggregates get a wider default window than single engines: consensus needs breadth. */ +const DEFAULT_NUM_RESULTS = 15; +const MAX_NUM_RESULTS = 30; + +/** + * Soft deadline for the fan-out: past this point the aggregate returns as + * soon as it has at least one engine's results. Fast HTML engines answer + * well under this; browser-backed engines (google, ecosia, mojeek) routinely + * exceed it and are treated as bonus coverage rather than latency floor. + */ +const SOFT_DEADLINE_MS = 5_000; + +/** + * Hard deadline for the fan-out: the aggregate returns whatever it has, even + * nothing, so one pathologically slow engine can never pin the tool call to + * the per-request 60s ceiling. + */ +const HARD_DEADLINE_MS = 30_000; + +/** Deadline overrides — test seam; production callers use the defaults. */ +export interface PublicWebDeadlines { + softMs?: number; + hardMs?: number; +} + +/** Accumulator for one deduplicated URL across engines. */ +interface MergedSource { + source: SearchSource; + /** Number of engines that returned this URL — the primary ranking signal. */ + engines: number; + /** Best (lowest) per-engine rank observed. */ + bestRank: number; + /** First-seen insertion index; final tiebreak keeps ordering deterministic. */ + order: number; +} + +/** + * Canonical dedup key for a result URL: case-normalized host without a + * leading `www.`, path without a trailing slash, query preserved, fragment + * dropped. Engines disagree on exactly these variations for the same page. + */ +function dedupKey(rawUrl: string): string { + try { + const url = new URL(rawUrl); + const host = url.hostname.toLowerCase().replace(/^www\./, ""); + let path = url.pathname; + if (path.length > 1 && path.endsWith("/")) path = path.slice(0, -1); + return `${host}${path}${url.search}`; + } catch { + return rawUrl; + } +} + +/** Merge one engine's ranked sources into the accumulator map. */ +function mergeSources(merged: Map, sources: readonly SearchSource[]): void { + for (const [rank, source] of sources.entries()) { + const key = dedupKey(source.url); + const existing = merged.get(key); + if (!existing) { + merged.set(key, { source: { ...source }, engines: 1, bestRank: rank, order: merged.size }); + continue; + } + existing.engines += 1; + if (rank < existing.bestRank) { + existing.bestRank = rank; + existing.source.title = source.title; + existing.source.url = source.url; + } + // Keep the most informative snippet regardless of which engine ranked it best. + if (source.snippet && source.snippet.length > (existing.source.snippet?.length ?? 0)) { + existing.source.snippet = source.snippet; + } + existing.source.publishedDate ??= source.publishedDate; + existing.source.ageSeconds ??= source.ageSeconds; + } +} + +/** + * Execute a web search against every credential-free engine in parallel and + * consolidate the results: URLs are deduplicated across engines, ranked by + * cross-engine consensus (how many engines returned them), then by best + * per-engine rank. + * + * The fan-out races three exits and returns at the earliest: every engine + * settled; the soft deadline elapsed with at least one success in hand; the + * hard deadline elapsed regardless. If the soft deadline fires before any + * engine has delivered, the aggregate keeps waiting (up to the hard cap) for + * the first success, so a slow field degrades to fewer engines rather than + * an empty answer. Stragglers are aborted once the race resolves. Individual + * engine failures (bot challenges, timeouts) are tolerated; the call fails + * only when every engine fails. + */ +export async function searchPublicWeb( + params: SearchParams, + deadlines: PublicWebDeadlines = {}, +): Promise { + const softMs = deadlines.softMs ?? SOFT_DEADLINE_MS; + const hardMs = deadlines.hardMs ?? HARD_DEADLINE_MS; + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const engineIds = PUBLIC_ENGINE_IDS.filter(id => !isSearchProviderExcluded(id)); + if (engineIds.length === 0) { + throw new SearchProviderError("public", "Every credential-free engine is excluded by settings.", 400); + } + + // Each engine composes its own per-request ceiling on top of the shared + // hard deadline; the straggler controller lets the aggregate cancel + // still-running engines once it decides to return. + const straggler = new AbortController(); + const signal = AbortSignal.any([withHardTimeout(params.signal), straggler.signal]); + + const responses: (SearchResponse | undefined)[] = new Array(engineIds.length); + const failures: { provider: { id: SearchProviderId; label: string }; error: unknown }[] = []; + const firstSuccess = Promise.withResolvers(); + const all = Promise.all( + engineIds.map(async (id, index) => { + try { + const provider = await getSearchProvider(id); + responses[index] = await provider.search({ ...params, signal }); + firstSuccess.resolve(); + } catch (error) { + failures.push({ provider: { id, label: id }, error }); + } + }), + ); + + await Promise.race([all, Bun.sleep(softMs)]); + if (!responses.some(response => response !== undefined) && failures.length < engineIds.length) { + await Promise.race([all, firstSuccess.promise, Bun.sleep(Math.max(0, hardMs - softMs))]); + } + straggler.abort(); + + // Merge in engine-priority order (not settlement order) so ranking + // tiebreaks stay deterministic. + const merged = new Map(); + for (const response of responses) { + if (response) mergeSources(merged, response.sources); + } + + if (merged.size === 0 && failures.length === engineIds.length) { + throw new SearchProviderError( + "public", + `All public engines failed: ${formatSearchProviderFailures(failures)}`, + 503, + ); + } + + const sources = [...merged.values()] + .sort((a, b) => b.engines - a.engines || a.bestRank - b.bestRank || a.order - b.order) + .slice(0, numResults) + .map(entry => entry.source); + + return { provider: "public", sources }; +} + +/** + * Aggregate meta-provider over every credential-free engine. Explicit-only: + * the auto chain already walks the individual engines sequentially, so + * fanning out to all of them is a deliberate user choice, not a fallback. + */ +export class PublicWebProvider extends SearchProvider { + readonly id = "public"; + readonly label = "Public Web"; + + isAvailable(_authStorage: AuthStorage): boolean { + return false; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchPublicWeb(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/startpage.ts b/packages/coding-agent/src/web/search/providers/startpage.ts new file mode 100644 index 000000000..be8cacb5f --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/startpage.ts @@ -0,0 +1,213 @@ +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { parseHTML } from "linkedom"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import type { LoadedHtmlPage } from "./browser-page"; +import { browserFetch } from "./browser-page"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +/** + * Startpage proxies Google's index behind a privacy frontend and serves fully + * server-rendered result pages — no JS challenge on the happy path. Its bot + * defense keys on requests that skip the homepage handshake: the search form + * carries a session token (`sc`) plus sibling hidden inputs, and posting the + * form with a stale/absent token 302s to the `/en/errors/` CAPTCHA shell. + * The robust flow is therefore the same dance a real browser performs: GET + * the homepage, lift the form's hidden inputs, POST them back with the query. + */ +const STARTPAGE_HOME_URL = "https://www.startpage.com/"; +const STARTPAGE_SEARCH_URL = "https://www.startpage.com/sp/search"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; + +/** + * Recency → Startpage `with_date` param. Accepts single letters; an absent + * value returns the unfiltered default. + */ +const RECENCY_TO_STARTPAGE_WITH_DATE: Record, string> = { + day: "d", + week: "w", + month: "m", + year: "y", +}; + +/** One organic result lifted from the Startpage results page. */ +interface ParsedResult { + title: string; + url: string; + snippet?: string; +} + +function normalizeText(value: string | null | undefined): string { + return (value ?? "").replace(/\s+/g, " ").trim(); +} + +/** + * `true` when Startpage answered with its CAPTCHA/error shell instead of + * results. Rejected requests 302 to `/en/errors/` (legacy: `/sp/captcha`), a + * Gatsby SPA whose chunk map names the captcha page components; the body + * marker matters because mocked fetch responses carry no final URL. A bare + * "captcha" substring is deliberately not used — result snippets for + * captcha-related queries would false-positive. + */ +function isChallengeResponse(page: LoadedHtmlPage): boolean { + if (/\/(?:errors|captcha)\//.test(page.url) || page.url.includes("/sp/captcha")) return true; + return page.html.includes("component---src-pages-captcha") || page.html.includes("/sp/captcha"); +} + +/** + * Lift the hidden inputs from the homepage's `/sp/search` form. Returns + * `undefined` when the form or its `sc` anti-bot token cannot be found so the + * caller can degrade to a tokenless GET instead of posting a doomed form. + */ +function parseSearchFormInputs(html: string): Record | undefined { + const { document } = parseHTML(html); + const form = document.querySelector('form[action="/sp/search"]'); + if (!form) return undefined; + const inputs: Record = {}; + for (const input of form.querySelectorAll('input[type="hidden"]')) { + const name = input.getAttribute("name"); + if (name) inputs[name] = input.getAttribute("value") ?? ""; + } + return inputs.sc ? inputs : undefined; +} + +/** Accept only http(s) result targets that point away from Startpage itself. */ +function sanitizeResultUrl(href: string | null | undefined): string | undefined { + if (!href) return undefined; + let url: URL; + try { + url = new URL(href, STARTPAGE_HOME_URL); + } catch { + return undefined; + } + if (url.protocol !== "http:" && url.protocol !== "https:") return undefined; + if (url.hostname === "startpage.com" || url.hostname.endsWith(".startpage.com")) return undefined; + return url.href; +} + +/** + * Walk the server-rendered results page in document order. + * + * Each organic hit lives in a `div.result` container holding the title + * anchor `a.result-link` (with an `h2.wgl-title` heading) and an optional + * `p.description` snippet. Hrefs are direct target URLs — Startpage does not + * wrap outbound clicks. The offscreen adblock-honeypot div uses the class + * token `a-bg-result`, which a CSS class selector correctly ignores, and + * sponsored placements render outside `div.result` containers. + */ +function parseHtmlResults(html: string): ParsedResult[] { + const { document } = parseHTML(html); + const results: ParsedResult[] = []; + for (const block of document.querySelectorAll("div.result")) { + const anchor = block.querySelector("a.result-link"); + if (!anchor) continue; + const url = sanitizeResultUrl(anchor.getAttribute("href")); + if (!url) continue; + const title = normalizeText(anchor.querySelector("h2, h3")?.textContent ?? anchor.textContent); + if (!title) continue; + const snippet = normalizeText(block.querySelector("p.description")?.textContent); + results.push({ title, url, snippet: snippet || undefined }); + } + return results; +} + +/** + * Fetch the homepage and lift the search form's hidden inputs. Best effort: + * any failure (network, non-OK status, challenge shell, markup drift) yields + * `undefined` and the caller falls back to a direct GET. + */ +async function fetchFormInputs(fetchImpl: FetchImpl, signal: AbortSignal): Promise | undefined> { + let page: LoadedHtmlPage; + try { + page = await browserFetch(STARTPAGE_HOME_URL, { fetch: fetchImpl, signal }); + } catch (error) { + if (signal.aborted) throw error; + return undefined; + } + if (page.status < 200 || page.status >= 300 || isChallengeResponse(page)) return undefined; + return parseSearchFormInputs(page.html); +} + +async function callStartpageHtml(params: SearchParams): Promise { + const fetchImpl = params.fetch ?? fetch; + const signal = withHardTimeout(params.signal); + const withDate = params.recency ? RECENCY_TO_STARTPAGE_WITH_DATE[params.recency] : undefined; + + const formInputs = await fetchFormInputs(fetchImpl, signal); + let page: LoadedHtmlPage; + if (formInputs) { + const form = new URLSearchParams(formInputs); + form.set("query", params.query); + if (withDate) form.set("with_date", withDate); + page = await browserFetch(STARTPAGE_SEARCH_URL, { + fetch: fetchImpl, + signal, + referer: STARTPAGE_HOME_URL, + init: { method: "POST", body: form.toString() }, + headers: { "Content-Type": "application/x-www-form-urlencoded" }, + }); + } else { + const url = new URL(STARTPAGE_SEARCH_URL); + url.searchParams.set("query", params.query); + if (withDate) url.searchParams.set("with_date", withDate); + page = await browserFetch(url.href, { + fetch: fetchImpl, + signal, + referer: STARTPAGE_HOME_URL, + }); + } + + if (isChallengeResponse(page)) { + throw new SearchProviderError( + "startpage", + "Startpage blocked the request with a CAPTCHA challenge. Startpage rate-limits automated searches from datacenter/shared-egress IPs; try another provider such as DuckDuckGo or Mojeek, or retry later.", + 429, + ); + } + if (page.status < 200 || page.status >= 300) { + const classified = classifyProviderHttpError("startpage", page.status, page.html); + if (classified) throw classified; + throw new SearchProviderError("startpage", `Startpage HTML error (${page.status})`, page.status); + } + return page.html; +} + +/** Execute a Startpage web search via the homepage-token form flow. */ +export async function searchStartpage(params: SearchParams): Promise { + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const html = await callStartpageHtml(params); + const parsed = parseHtmlResults(html); + + const sources: SearchSource[] = []; + const seen = new Set(); + for (const result of parsed) { + if (seen.has(result.url)) continue; + seen.add(result.url); + sources.push({ title: result.title, url: result.url, snippet: result.snippet }); + if (sources.length >= numResults) break; + } + + return { provider: "startpage", sources }; +} + +/** Search provider for Startpage (no API key required). */ +export class StartpageProvider extends SearchProvider { + readonly id = "startpage"; + readonly label = "Startpage"; + + isAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchStartpage(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/yahoo.ts b/packages/coding-agent/src/web/search/providers/yahoo.ts new file mode 100644 index 000000000..bdc1e312f --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/yahoo.ts @@ -0,0 +1,179 @@ +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import { parseHTML } from "linkedom"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { browserFetch } from "./browser-page"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +/** + * Yahoo Search's server-rendered results page. A plain GET with browser + * navigation headers returns the full SERP without any JavaScript challenge, + * so no headless-browser fallback is needed (verified live 2026-07). + */ +const YAHOO_HOME_URL = "https://search.yahoo.com/"; +const YAHOO_SEARCH_URL = "https://search.yahoo.com/search"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; + +/** + * Recency → Yahoo `btf` query param. Yahoo's time filter only offers + * day/week/month; `year` has no equivalent and is silently dropped per the + * {@link SearchParams.recency} contract. + */ +const RECENCY_TO_YAHOO_BTF: Partial, string>> = { + day: "d", + week: "w", + month: "m", +}; + +interface ParsedResult { + title: string; + url: string; + snippet?: string; +} + +/** + * Resolve a Yahoo result href back to the underlying target URL. + * + * Organic hrefs are wrapped through the click tracker + * `https://r.search.yahoo.com/_ylt=…/RU=/RK=…/RS=…`; + * the `/RU=` path segment carries the destination. Older layouts emit plain + * absolute hrefs, so both shapes are handled. Tracker links without a + * recoverable target and Yahoo-internal navigation are rejected. + */ +function unwrapResultUrl(href: string): string | undefined { + let url: URL; + try { + url = new URL(href, YAHOO_HOME_URL); + } catch { + return undefined; + } + if (url.protocol !== "http:" && url.protocol !== "https:") return undefined; + + const wrapped = /\/RU=([^/]+)/.exec(url.pathname); + if (wrapped) { + let target: string; + try { + target = decodeURIComponent(wrapped[1]); + } catch { + return undefined; + } + return target.startsWith("http://") || target.startsWith("https://") ? target : undefined; + } + // A tracker link without an RU segment has no recoverable destination. + if (url.hostname === "r.search.yahoo.com") return undefined; + // Relative hrefs resolve against the search host: internal navigation. + if (url.hostname === "search.yahoo.com") return undefined; + return url.href; +} + +/** + * Walk the SERP and pull organic result blocks in document order. + * + * Organics render as `

    ` blocks (inside `#web`'s + * `
      `): the title `

      ` sits inside the tracker `` in the current + * layout, while legacy layouts nested the `` inside `

      ` + * — both are handled. The preview text lives in a sibling + * `
      `. Module headers ("Videos", "People also ask") + * carry `

      `s outside `.algo` blocks and are excluded by construction. + */ +function parseHtmlResults(html: string): ParsedResult[] { + const { document } = parseHTML(html); + const results: ParsedResult[] = []; + for (const block of document.querySelectorAll("div.algo")) { + const heading = block.querySelector("h3"); + if (!heading) continue; + const anchor = heading.querySelector("a") ?? heading.closest("a"); + const href = anchor?.getAttribute("href"); + if (!href) continue; + const url = unwrapResultUrl(href); + if (!url) continue; + const title = (heading.textContent ?? "").replace(/\s+/g, " ").trim(); + if (!title) continue; + const snippet = (block.querySelector(".compText")?.textContent ?? "").replace(/\s+/g, " ").trim() || undefined; + results.push({ title, url, snippet }); + } + return results; +} + +/** + * `true` when Yahoo answered with its EU consent interstitial instead of + * results: either the request was redirected to consent.yahoo.com / + * guce.yahoo.com, or the body carries the consent form. The normal SERP + * mentions guce.yahoo.com only in a meta tag, so detection keys on the + * consent-host redirect and the `collectConsent` form action. + */ +function isConsentInterstitial(finalUrl: string, html: string): boolean { + if (/^https?:\/\/(?:[^/]*\.)?(?:consent|guce)\.yahoo\.com\//i.test(finalUrl)) return true; + return html.includes("consent.yahoo.com") || html.includes("collectConsent"); +} + +async function callYahooHtml(params: SearchParams, numResults: number): Promise { + const url = new URL(YAHOO_SEARCH_URL); + url.searchParams.set("p", params.query); + url.searchParams.set("n", String(numResults)); + const btf = params.recency ? RECENCY_TO_YAHOO_BTF[params.recency] : undefined; + if (btf) url.searchParams.set("btf", btf); + + const page = await browserFetch(url.href, { + fetch: params.fetch ?? fetch, + signal: withHardTimeout(params.signal), + referer: YAHOO_HOME_URL, + }); + + const body = page.html; + if (page.status < 200 || page.status >= 300) { + const classified = classifyProviderHttpError("yahoo", page.status, body); + if (classified) throw classified; + throw new SearchProviderError("yahoo", `Yahoo HTML error (${page.status})`, page.status); + } + + if (isConsentInterstitial(page.url, body)) { + throw new SearchProviderError( + "yahoo", + "Yahoo served its GDPR consent interstitial instead of search results. This typically affects EU egress IPs; use another web search provider such as DuckDuckGo, Brave, or Mojeek.", + 429, + ); + } + + return body; +} + +/** Execute a Yahoo web search via the server-rendered HTML results page. */ +export async function searchYahoo(params: SearchParams): Promise { + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const html = await callYahooHtml(params, numResults); + const parsed = parseHtmlResults(html); + + const sources: SearchSource[] = []; + const seen = new Set(); + for (const result of parsed) { + if (seen.has(result.url)) continue; + seen.add(result.url); + sources.push({ title: result.title, url: result.url, snippet: result.snippet }); + if (sources.length >= numResults) break; + } + + return { provider: "yahoo", sources }; +} + +/** Search provider for Yahoo (no API key required). */ +export class YahooProvider extends SearchProvider { + readonly id = "yahoo"; + readonly label = "Yahoo"; + + isAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchYahoo(params); + } +} diff --git a/packages/coding-agent/src/web/search/types.ts b/packages/coding-agent/src/web/search/types.ts index b7c555803..477525ea1 100644 --- a/packages/coding-agent/src/web/search/types.ts +++ b/packages/coding-agent/src/web/search/types.ts @@ -43,16 +43,46 @@ export const SEARCH_PROVIDER_OPTIONS = [ { value: "parallel", label: "Parallel", description: "Requires PARALLEL_API_KEY" }, { value: "synthetic", label: "Synthetic", description: "Requires SYNTHETIC_API_KEY" }, { value: "searxng", label: "SearXNG", description: "Requires SEARXNG_ENDPOINT or searxng.endpoint" }, + { + value: "startpage", + label: "Startpage", + description: "Credential-free scrape of Startpage (Google-backed) results; may be bot-challenged", + }, { value: "duckduckgo", label: "DuckDuckGo", description: "Credential-free best-effort fallback; may be bot-challenged on datacenter/shared-egress IPs", }, + { + value: "bing", + label: "Bing", + description: "Credential-free HTML scrape of Bing results; may be bot-challenged", + }, + { + value: "yahoo", + label: "Yahoo", + description: "Credential-free HTML scrape of Yahoo (Bing-backed) results", + }, + { + value: "ecosia", + label: "Ecosia", + description: "Credential-free browser-backed scrape of Ecosia (Google-backed) results", + }, { value: "google", label: "Google", description: "Credential-free browser-backed fallback; slower and may be bot-challenged", }, + { + value: "mojeek", + label: "Mojeek", + description: "Credential-free browser-backed scrape of Mojeek's independent index", + }, + { + value: "public", + label: "Public Web", + description: "Queries every credential-free engine in parallel and consolidates deduplicated results", + }, ] as const; /** Supported web search providers (every option except `auto`). */ diff --git a/packages/coding-agent/test/tools/web-search-bing.test.ts b/packages/coding-agent/test/tools/web-search-bing.test.ts new file mode 100644 index 000000000..eb5766a00 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-bing.test.ts @@ -0,0 +1,184 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import type { SearchParams } from "@oh-my-pi/pi-coding-agent/web/search/providers/base"; +import { searchBing } from "@oh-my-pi/pi-coding-agent/web/search/providers/bing"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("Bing search must not request API keys"); + }, + resolver() { + throw new Error("Bing search must not request credential resolvers"); + }, + hasAuth() { + throw new Error("Bing search must not check auth"); + }, +} as unknown as AuthStorage; + +function makeParams(query: string, fetch: FetchImpl): SearchParams { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "Bing search test prompt", + fetch, + }; +} + +/** Wrap a target URL the way Bing's `/ck/a` click-tracking redirect does (`u=a1`). */ +function wrapBingHref(target: string): string { + const payload = Buffer.from(target, "utf-8").toString("base64url"); + return `https://www.bing.com/ck/a?!&&p=6ddfcabc8528ae9bbd50e99e9ccfcb85&ptn=3&ver=2&hsh=4&fclid=2a06eaa3&u=a1${payload}&ntb=1`; +} + +/** Render a `b_algo` block matching Bing's live markup (entity-escaped href, sitelink anchor outside `h2`). */ +function algoResult(href: string, title: string, snippet?: string): string { + const escaped = href.replace(/&/g, "&"); + return `
    1. + +

      ${title}

      + ${snippet ? `

      ${snippet}

      ` : ""} +
    2. `; +} + +function resultsPage(...items: string[]): string { + return `
        ${items.join("\n")}
      `; +} + +describe("Bing web search provider", () => { + it("requests the HTML result page with browser navigation headers and recency filter", async () => { + let capturedUrl = ""; + let capturedInit: RequestInit | undefined; + const fetchMock: FetchImpl = (input, init) => { + capturedUrl = typeof input === "string" ? input : input.toString(); + capturedInit = init; + return Promise.resolve( + new Response( + resultsPage(algoResult(wrapBingHref("https://example.com/result"), "Result", "Search snippet")), + { status: 200, headers: { "Content-Type": "text/html" } }, + ), + ); + }; + + const response = await searchBing({ + ...makeParams("browser headers & parsing", fetchMock), + numSearchResults: 99, + recency: "week", + }); + + const url = new URL(capturedUrl); + expect(url.origin + url.pathname).toBe("https://www.bing.com/search"); + expect(url.searchParams.get("q")).toBe("browser headers & parsing"); + expect(url.searchParams.get("count")).toBe("20"); + expect(url.searchParams.get("mkt")).toBe("en-US"); + expect(url.searchParams.get("setlang")).toBe("en"); + expect(url.searchParams.get("filters")).toBe('ex1:"ez2"'); + expect(capturedInit?.method).toBeUndefined(); + const headers = new Headers(capturedInit?.headers); + expect(headers.get("accept")).toContain("text/html"); + expect(headers.get("user-agent")).toMatch(/Chrome\/\d+\.0\.0\.0/); + expect(headers.get("referer")).toBe("https://www.bing.com/"); + expect(headers.get("sec-fetch-dest")).toBe("document"); + expect(headers.get("sec-fetch-mode")).toBe("navigate"); + expect(headers.get("sec-fetch-site")).toBe("same-origin"); + expect(response.sources).toEqual([ + { title: "Result", url: "https://example.com/result", snippet: "Search snippet" }, + ]); + }); + + it("maps every recency window to Bing's native freshness codes and omits the param otherwise", async () => { + const filtersFor = async (recency?: SearchParams["recency"]): Promise => { + let captured = ""; + const fetchMock: FetchImpl = input => { + captured = typeof input === "string" ? input : input.toString(); + return Promise.resolve(new Response(resultsPage(), { status: 200 })); + }; + await searchBing({ ...makeParams("recency mapping", fetchMock), recency }); + return new URL(captured).searchParams.get("filters"); + }; + + expect(await filtersFor("day")).toBe('ex1:"ez1"'); + expect(await filtersFor("month")).toBe('ex1:"ez3"'); + expect(await filtersFor(undefined)).toBeNull(); + + // "Past year" has no fixed code; Bing's own dropdown emits an epoch-day range. + const year = await filtersFor("year"); + const match = year?.match(/^ex1:"ez5_(\d+)_(\d+)"$/); + expect(match).toBeTruthy(); + const [, start, end] = match as RegExpMatchArray; + expect(Number(end) - Number(start)).toBe(365); + expect(Math.abs(Number(end) - Math.floor(Date.now() / 86_400_000))).toBeLessThanOrEqual(1); + }); + + it("unwraps ck/a redirects, keeps direct links, deduplicates targets, and skips junk rows", async () => { + const target = "https://example.com/docs?a=1&b=2"; + const html = resultsPage( + algoResult( + wrapBingHref(target), + "Bun — A fast runtime", + `Jan 3, 2026 · Bundle & run JavaScript. How to verify you are human on CAPTCHA walls.`, + ), + // Direct external href; snippet only in the b_algoSlug fallback container. + `
    3. Direct result

      Slug snippet
    4. `, + algoResult(wrapBingHref(target), "Duplicate target", "duplicate"), + // Bing-internal navigation link must be dropped. + algoResult("https://www.bing.com/images/search?q=x", "Images tab"), + // Unknown u= payload version must be dropped, not garbage-decoded. + `
    5. Future wrapper

    6. `, + algoResult(wrapBingHref("https://example.com/bare"), "No snippet row"), + ); + const fetchMock: FetchImpl = () => Promise.resolve(new Response(html, { status: 200 })); + + const response = await searchBing({ ...makeParams("mixed markup", fetchMock), numSearchResults: 10 }); + + expect(response.provider).toBe("bing"); + expect(response.sources).toEqual([ + { + title: "Bun — A fast runtime", + url: target, + snippet: "Jan 3, 2026 · Bundle & run JavaScript. How to verify you are human on CAPTCHA walls.", + }, + { title: "Direct result", url: "https://example.com/direct", snippet: "Slug snippet" }, + { title: "No snippet row", url: "https://example.com/bare", snippet: undefined }, + ]); + }); + + it("returns empty sources for Bing's genuine no-results page", async () => { + const html = `
      1. There are no results for xzqv

      `; + const fetchMock: FetchImpl = () => Promise.resolve(new Response(html, { status: 200 })); + + const response = await searchBing(makeParams("xzqv", fetchMock)); + + expect(response).toEqual({ provider: "bing", sources: [] }); + }); + + it("surfaces Bing's CAPTCHA challenge as a provider-tagged 429", async () => { + const challenge = `
      Please solve the challenge
      `; + const fetchMock: FetchImpl = () => Promise.resolve(new Response(challenge, { status: 200 })); + + try { + await searchBing(makeParams("blocked", fetchMock)); + expect.unreachable("Bing CAPTCHA challenge should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "bing", status: 429 }); + expect((error as SearchProviderError).message).toContain("CAPTCHA"); + } + }); + + it("propagates non-OK HTTP statuses as provider errors", async () => { + const fetchMock: FetchImpl = () => Promise.resolve(new Response("Service Unavailable", { status: 503 })); + + try { + await searchBing(makeParams("outage", fetchMock)); + expect.unreachable("HTTP 503 should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "bing", + status: 503, + message: "Bing HTML error (503)", + }); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-browser-headers.test.ts b/packages/coding-agent/test/tools/web-search-browser-headers.test.ts new file mode 100644 index 000000000..6db7405b5 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-browser-headers.test.ts @@ -0,0 +1,69 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { buildBrowserNavigationHeaders } from "@oh-my-pi/pi-coding-agent/web/search/providers/browser-headers"; +import { browserFetch } from "@oh-my-pi/pi-coding-agent/web/search/providers/browser-page"; + +afterEach(() => { + vi.restoreAllMocks(); +}); + +describe("browser navigation headers", () => { + it("builds a randomized, internally consistent browser profile", () => { + const headers = buildBrowserNavigationHeaders(); + + // Ensure core navigation headers are always populated + expect(headers["User-Agent"]).toBeDefined(); + expect(headers.Accept).toBeDefined(); + expect(headers["Accept-Language"]).toBeDefined(); + expect(headers["Accept-Encoding"]).toBeDefined(); + + const ua = headers["User-Agent"] || ""; + + // Ensure generated headers match standard conventions for the resolved browser type + if (ua.includes("Firefox/")) { + // Firefox doesn't support Client Hints and has unique accept values + expect(headers["Sec-Ch-Ua"]).toBeUndefined(); + expect(headers["Sec-Ch-Ua-Platform"]).toBeUndefined(); + expect(headers.Accept).toContain("text/html"); + } else if (ua.includes("Chrome/")) { + // Chrome, Edge, and Opera support Client Hints + expect(headers["Sec-Ch-Ua"]).toBeDefined(); + expect(headers["Sec-Ch-Ua-Mobile"]).toBeDefined(); + expect(headers["Sec-Ch-Ua-Platform"]).toBeDefined(); + + if (ua.includes("Edg/")) { + expect(headers["Sec-Ch-Ua"]).toContain("Microsoft Edge"); + } else if (ua.includes("OPR/")) { + expect(headers["Sec-Ch-Ua"]).toContain("Opera"); + } else { + expect(headers["Sec-Ch-Ua"]).toContain("Google Chrome"); + } + } + }); + + it("falls back gracefully to robust Mac Chrome profile when randomized option is disabled", () => { + const headers = buildBrowserNavigationHeaders({ randomized: false }); + + expect(headers["User-Agent"]).toContain("Chrome/149.0.0.0"); + expect(headers["User-Agent"]).toContain("Macintosh; Intel Mac OS X 10_15_7"); + expect(headers["Sec-Ch-Ua"]).toContain('v="149"'); + expect(headers["Sec-Ch-Ua-Platform"]).toBe('"macOS"'); + }); + + it("uses ordinary fetch before considering the browser fallback", async () => { + const fetchSpy = vi + .spyOn(globalThis, "fetch") + .mockResolvedValue(new Response("results", { status: 200 })); + + const page = await browserFetch("https://search.example/results", { + signal: new AbortController().signal, + browser: { shouldFallback: () => false }, + }); + + expect(fetchSpy).toHaveBeenCalledTimes(1); + expect(page).toEqual({ + html: "results", + status: 200, + url: "https://search.example/results", + }); + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-ecosia.test.ts b/packages/coding-agent/test/tools/web-search-ecosia.test.ts new file mode 100644 index 000000000..af5544b72 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-ecosia.test.ts @@ -0,0 +1,172 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import type { SearchParams } from "@oh-my-pi/pi-coding-agent/web/search/providers/base"; +import { searchEcosia } from "@oh-my-pi/pi-coding-agent/web/search/providers/ecosia"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("Ecosia search must not request API keys"); + }, + resolver() { + throw new Error("Ecosia search must not request credential resolvers"); + }, + hasAuth() { + throw new Error("Ecosia search must not check auth"); + }, +} as unknown as AuthStorage; + +function makeParams(query: string, fetch: FetchImpl): SearchParams { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "Ecosia search test prompt", + fetch, + }; +} + +/** + * Trimmed organic result matching Ecosia's server-rendered markup: two + * `result-link` anchors per article (breadcrumb row + title row) and a + * description container that mixes a screen-reader thumbnail caption with + * the real `web-result-description` paragraph. + */ +function organicResult(url: string, title: string, snippet?: string): string { + return `
      + +
      `; +} + +/** Trimmed Cloudflare managed-challenge page as served by Ecosia's firewall. */ +const CLOUDFLARE_CHALLENGE = `Ecosia Firewall
      +

      Confirm you’re not a robot

      +

      Our system has detected unusual traffic from your network. Please solve the challenge below to show you’re not a robot.

      + +
      `; + +describe("Ecosia web search provider", () => { + it("requests the results page with browser navigation headers and ignores recency", async () => { + let capturedUrl = ""; + let capturedInit: RequestInit | undefined; + const fetchMock: FetchImpl = (input, init) => { + capturedUrl = typeof input === "string" ? input : input.toString(); + capturedInit = init; + return Promise.resolve( + new Response(organicResult("https://example.com/result", "Result", "Search snippet"), { + status: 200, + headers: { "Content-Type": "text/html" }, + }), + ); + }; + + const response = await searchEcosia({ + ...makeParams("browser headers & parsing", fetchMock), + numSearchResults: 99, + recency: "week", + }); + + const url = new URL(capturedUrl); + expect(url.origin + url.pathname).toBe("https://www.ecosia.org/search"); + expect(url.searchParams.get("q")).toBe("browser headers & parsing"); + // Ecosia has no confirmed time filter; recency must not leak into the request. + expect([...url.searchParams.keys()]).toEqual(["q"]); + expect(capturedInit?.method).toBeUndefined(); + const headers = new Headers(capturedInit?.headers); + expect(headers.get("accept")).toContain("text/html"); + expect(headers.get("user-agent")).toMatch(/Chrome\/\d+\.0\.0\.0/); + expect(headers.get("referer")).toBe("https://www.ecosia.org/"); + expect(headers.get("sec-fetch-dest")).toBe("document"); + expect(headers.get("sec-fetch-mode")).toBe("navigate"); + expect(headers.get("sec-fetch-site")).toBe("same-origin"); + expect(response.sources).toEqual([ + { title: "Result", url: "https://example.com/result", snippet: "Search snippet" }, + ]); + }); + + it("parses organic articles, skips junk rows, and deduplicates targets", async () => { + const html = [ + organicResult("https://example.com/first", "First & decoded", "Leading snippet"), + // Ad slot: rendered client-side into an empty container, never an organic article. + `
      `, + // Internal navigation and non-http targets must be rejected. + organicResult("https://www.ecosia.org/images?q=first", "Images vertical", "internal"), + organicResult("javascript:void(0)", "Script link", "junk"), + // Article without a title heading is skipped. + `
      `, + organicResult("https://example.com/first", "Duplicate of first", "duplicate"), + organicResult("https://example.com/bare", "Snippetless result"), + ].join("\n"); + const fetchMock: FetchImpl = () => Promise.resolve(new Response(html, { status: 200 })); + + const response = await searchEcosia({ ...makeParams("mixed markup", fetchMock), numSearchResults: 10 }); + + expect(response.provider).toBe("ecosia"); + expect(response.sources).toEqual([ + { title: "First & decoded", url: "https://example.com/first", snippet: "Leading snippet" }, + { title: "Snippetless result", url: "https://example.com/bare", snippet: undefined }, + ]); + }); + + it("clamps sources to the requested count", async () => { + const html = [ + organicResult("https://example.com/1", "One", "s1"), + organicResult("https://example.com/2", "Two", "s2"), + organicResult("https://example.com/3", "Three", "s3"), + ].join("\n"); + const fetchMock: FetchImpl = () => Promise.resolve(new Response(html, { status: 200 })); + + const response = await searchEcosia({ ...makeParams("clamped", fetchMock), numSearchResults: 2 }); + + expect(response.sources.map(source => source.url)).toEqual(["https://example.com/1", "https://example.com/2"]); + }); + + it("surfaces the Cloudflare challenge as a provider-tagged 429, regardless of HTTP status", async () => { + for (const status of [403, 200]) { + const fetchMock: FetchImpl = () => Promise.resolve(new Response(CLOUDFLARE_CHALLENGE, { status })); + try { + await searchEcosia(makeParams("blocked", fetchMock)); + expect.unreachable(`Cloudflare challenge with status ${status} should reject`); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "ecosia", status: 429 }); + expect((error as SearchProviderError).message).toContain("Cloudflare bot challenge"); + } + } + }); + + it("maps non-challenge HTTP failures to a provider-tagged error with the upstream status", async () => { + const fetchMock: FetchImpl = () => + Promise.resolve(new Response("Internal Server Error", { status: 500 })); + + try { + await searchEcosia(makeParams("broken", fetchMock)); + expect.unreachable("HTTP 500 should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "ecosia", + status: 500, + message: "Ecosia HTML error (500)", + }); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-google.test.ts b/packages/coding-agent/test/tools/web-search-google.test.ts index 9bf570b61..0976ce381 100644 --- a/packages/coding-agent/test/tools/web-search-google.test.ts +++ b/packages/coding-agent/test/tools/web-search-google.test.ts @@ -65,7 +65,7 @@ describe("Google web search provider", () => { expect(capturedInit?.method).toBeUndefined(); const headers = new Headers(capturedInit?.headers); expect(headers.get("accept")).toContain("text/html"); - expect(headers.get("user-agent")).toContain("Chrome/149"); + expect(headers.get("user-agent")).toMatch(/Chrome\/\d+\.0\.0\.0/); expect(headers.get("referer")).toBe("https://www.google.com/"); expect(headers.get("sec-fetch-dest")).toBe("document"); expect(headers.get("sec-fetch-mode")).toBe("navigate"); diff --git a/packages/coding-agent/test/tools/web-search-mojeek.test.ts b/packages/coding-agent/test/tools/web-search-mojeek.test.ts new file mode 100644 index 000000000..a13acdfc9 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-mojeek.test.ts @@ -0,0 +1,161 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import type { SearchParams } from "@oh-my-pi/pi-coding-agent/web/search/providers/base"; +import { searchMojeek } from "@oh-my-pi/pi-coding-agent/web/search/providers/mojeek"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("Mojeek search must not request API keys"); + }, + resolver() { + throw new Error("Mojeek search must not request credential resolvers"); + }, + hasAuth() { + throw new Error("Mojeek search must not check auth"); + }, +} as unknown as AuthStorage; + +function makeParams(query: string, fetch: FetchImpl): SearchParams { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "Mojeek search test prompt", + fetch, + }; +} + +/** One organic result row in the shape Mojeek's results page renders live. */ +function resultItem(href: string, title: string, snippet?: string, liClass = "r1"): string { + return `
    7. ${href}

      ${title}

      ${snippet ? `

      ${snippet}

      ` : ""}
    8. `; +} + +function resultsPage(items: string): string { + return `
        ${items}
      `; +} + +describe("Mojeek web search provider", () => { + it("requests the configured public search route with browser navigation headers, locale, count, and recency", async () => { + let capturedUrl = ""; + let capturedInit: RequestInit | undefined; + const fetchMock: FetchImpl = (input, init) => { + capturedUrl = typeof input === "string" ? input : input.toString(); + capturedInit = init; + return Promise.resolve( + new Response(resultsPage(resultItem("https://example.com/result", "Result", "Search snippet")), { + status: 200, + headers: { "Content-Type": "text/html" }, + }), + ); + }; + + const response = await searchMojeek({ + ...makeParams("browser headers & parsing", fetchMock), + numSearchResults: 99, + recency: "week", + }); + + const url = new URL(capturedUrl); + expect(url.origin + url.pathname).toBe("https://www.mojeek.de/search"); + expect(url.searchParams.get("q")).toBe("browser headers & parsing"); + expect(url.searchParams.get("t")).toBe("20"); + expect(url.searchParams.get("arc")).toBe("none"); + expect(url.searchParams.get("lang")).toBe("en"); + expect(url.searchParams.get("lb")).toBe("en"); + expect(url.searchParams.get("theme")).toBe("dark"); + expect(url.searchParams.get("since")).toBe("week"); + expect(capturedInit?.method).toBeUndefined(); + const headers = new Headers(capturedInit?.headers); + expect(headers.get("accept")).toContain("text/html"); + expect(headers.get("user-agent")).toMatch(/Chrome\/\d+\.0\.0\.0/); + expect(headers.get("referer")).toBe("https://www.mojeek.de/?arc=none&lang=en&lb=en&theme=dark"); + expect(headers.get("sec-fetch-dest")).toBe("document"); + expect(headers.get("sec-fetch-mode")).toBe("navigate"); + expect(headers.get("sec-fetch-site")).toBe("same-origin"); + expect(response.sources).toEqual([ + { title: "Result", url: "https://example.com/result", snippet: "Search snippet" }, + ]); + }); + + it("omits the since filter and requests the default count when recency is absent", async () => { + let capturedUrl = ""; + const fetchMock: FetchImpl = input => { + capturedUrl = typeof input === "string" ? input : input.toString(); + return Promise.resolve(new Response(resultsPage(""), { status: 200 })); + }; + + await searchMojeek(makeParams("plain query", fetchMock)); + + const url = new URL(capturedUrl); + expect(url.searchParams.get("t")).toBe("10"); + expect(url.searchParams.get("since")).toBeNull(); + }); + + it("parses result rows, deduplicates targets, and skips junk and intra-Mojeek rows", async () => { + const html = resultsPage( + [ + resultItem( + "https://bun.sh/", + "Bun & friends — a fast runtime", + "Bun is a JavaScript\n\t runtime. ... built from scratch.", + ), + resultItem("https://bun.com/docs/runtime", "Bun Runtime - Bun", "Execute files with Bun.", "r2 clu-result"), + resultItem("https://bun.sh/", "Duplicate of the first target", "duplicate"), + `
    9. Row without a title anchor is skipped

    10. `, + resultItem("/search?q=bun&s=11", "Next page"), + resultItem("https://www.mojeek.com/about/", "About Mojeek"), + resultItem("https://www.mojeek.de/about/", "German About Mojeek"), + resultItem("https://example.com/bare", "Bare result without snippet"), + ].join(""), + ); + const fetchMock: FetchImpl = () => Promise.resolve(new Response(html, { status: 200 })); + + const response = await searchMojeek({ ...makeParams("bun", fetchMock), numSearchResults: 10 }); + + expect(response.provider).toBe("mojeek"); + expect(response.sources).toEqual([ + { + title: "Bun & friends — a fast runtime", + url: "https://bun.sh/", + snippet: "Bun is a JavaScript runtime. ... built from scratch.", + }, + { + title: "Bun Runtime - Bun", + url: "https://bun.com/docs/runtime", + snippet: "Execute files with Bun.", + }, + { + title: "Bare result without snippet", + url: "https://example.com/bare", + snippet: undefined, + }, + ]); + }); + + it("surfaces the ALTCHA captcha interstitial as a provider-tagged 429", async () => { + const captcha = `Captcha

      Verification required

      Please complete the challenge to continue.

      `; + const fetchMock: FetchImpl = () => Promise.resolve(new Response(captcha, { status: 200 })); + + try { + await searchMojeek(makeParams("blocked", fetchMock)); + expect.unreachable("Mojeek captcha interstitial should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "mojeek", status: 429 }); + expect((error as SearchProviderError).message).toContain("Mojeek"); + } + }); + + it("maps the 403 automated-queries wall to the robot 429 rather than a generic 403", async () => { + const refusal = `403 - Forbidden

      403 - Forbidden

      Sorry your network appears to be sending automated queries so we can't process your search at this time.

      `; + const fetchMock: FetchImpl = () => Promise.resolve(new Response(refusal, { status: 403 })); + + try { + await searchMojeek(makeParams("rate limited", fetchMock)); + expect.unreachable("Mojeek automated-queries wall should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "mojeek", status: 429 }); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-public.test.ts b/packages/coding-agent/test/tools/web-search-public.test.ts new file mode 100644 index 000000000..a3c58b53f --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-public.test.ts @@ -0,0 +1,198 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { setExcludedSearchProviders } from "@oh-my-pi/pi-coding-agent/web/search/provider"; +import type { SearchParams } from "@oh-my-pi/pi-coding-agent/web/search/providers/base"; +import { searchPublicWeb } from "@oh-my-pi/pi-coding-agent/web/search/providers/public"; +import { SearchProviderError, type SearchProviderId } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("Public web search must not request API keys"); + }, + resolver() { + throw new Error("Public web search must not request credential resolvers"); + }, + hasAuth() { + throw new Error("Public web search must not check auth"); + }, +} as unknown as AuthStorage; + +/** Restrict the fan-out to the two engines these tests provide fixtures for. */ +const NON_TEST_ENGINES: readonly SearchProviderId[] = ["bing", "yahoo", "ecosia", "startpage", "mojeek"]; + +function makeParams(query: string, fetch: FetchImpl): SearchParams { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "Public web search test prompt", + fetch, + }; +} + +function ddgResult(url: string, title: string, snippet?: string): string { + return ``; +} + +function googleResult(url: string, title: string, snippet?: string): string { + return `
      +

      ${title}

      + ${snippet ? `
      ${snippet}
      ` : ""} +
      `; +} + +/** Dispatch fixture bodies per engine host. */ +function makeFetchMock(bodies: { ddg: string; google: string }): FetchImpl { + return input => { + const url = typeof input === "string" ? input : input.toString(); + if (url.includes("duckduckgo.com")) { + return Promise.resolve(new Response(bodies.ddg, { status: 200 })); + } + if (url.includes("google.com")) { + return Promise.resolve(new Response(bodies.google, { status: 200 })); + } + return Promise.reject(new Error(`Unexpected fetch in public web test: ${url}`)); + }; +} + +const GOOGLE_CHALLENGE = `Our systems have detected unusual traffic from your computer network.`; +const DDG_CHALLENGE = `
      `; + +afterEach(() => { + setExcludedSearchProviders([]); +}); + +describe("Public Web aggregate provider", () => { + it("consolidates engines: dedups URL variants, ranks by consensus, keeps the best snippet", async () => { + setExcludedSearchProviders(NON_TEST_ENGINES); + const fetchMock = makeFetchMock({ + ddg: [ + ddgResult("https://example.com/shared", "Shared result", "short"), + ddgResult("https://a.example/one", "Alpha", "alpha snippet"), + ].join("\n"), + google: [ + googleResult("https://www.example.com/shared/", "Shared (google)", "a much longer consolidated snippet"), + googleResult("https://c.example/three", "Gamma", "gamma snippet"), + ].join("\n"), + }); + + const response = await searchPublicWeb(makeParams("consensus ranking", fetchMock)); + + expect(response.provider).toBe("public"); + expect(response.sources).toEqual([ + // Two-engine consensus outranks single-engine results; www/trailing-slash + // variants merge. Google merges first (higher tiebreak priority), so its + // title/url win the equal-rank tie; the longer snippet wins regardless. + { + title: "Shared (google)", + url: "https://www.example.com/shared/", + snippet: "a much longer consolidated snippet", + }, + { title: "Gamma", url: "https://c.example/three", snippet: "gamma snippet" }, + { title: "Alpha", url: "https://a.example/one", snippet: "alpha snippet" }, + ]); + }); + + it("tolerates individual engine failures and returns the surviving results", async () => { + setExcludedSearchProviders(NON_TEST_ENGINES); + const fetchMock = makeFetchMock({ + ddg: ddgResult("https://a.example/one", "Alpha", "alpha snippet"), + google: GOOGLE_CHALLENGE, + }); + + const response = await searchPublicWeb(makeParams("partial failure", fetchMock)); + + expect(response.sources).toEqual([{ title: "Alpha", url: "https://a.example/one", snippet: "alpha snippet" }]); + }); + + it("returns at the soft deadline with delivered results and aborts stragglers", async () => { + setExcludedSearchProviders(NON_TEST_ENGINES); + let stragglerAborted = false; + const fetchMock: FetchImpl = (input, init) => { + const url = typeof input === "string" ? input : input.toString(); + if (url.includes("duckduckgo.com")) { + return Promise.resolve( + new Response(ddgResult("https://a.example/one", "Alpha", "alpha snippet"), { status: 200 }), + ); + } + // google: hangs until the aggregate cancels it at the deadline. + const { promise, reject } = Promise.withResolvers(); + init?.signal?.addEventListener("abort", () => { + stragglerAborted = true; + reject(new Error("aborted")); + }); + return promise; + }; + + const response = await searchPublicWeb(makeParams("deadline race", fetchMock), { softMs: 50 }); + + expect(response.sources).toEqual([{ title: "Alpha", url: "https://a.example/one", snippet: "alpha snippet" }]); + expect(stragglerAborted).toBe(true); + }); + + it("waits past the soft deadline for the first success instead of returning empty", async () => { + setExcludedSearchProviders(NON_TEST_ENGINES); + const fetchMock: FetchImpl = async input => { + const url = typeof input === "string" ? input : input.toString(); + if (url.includes("duckduckgo.com")) { + await Bun.sleep(60); + return new Response(ddgResult("https://a.example/one", "Alpha", "alpha snippet"), { status: 200 }); + } + return new Response(GOOGLE_CHALLENGE, { status: 200 }); + }; + + const response = await searchPublicWeb(makeParams("slow first success", fetchMock), { softMs: 10 }); + + expect(response.sources).toEqual([{ title: "Alpha", url: "https://a.example/one", snippet: "alpha snippet" }]); + }); + + it("returns whatever it has at the hard deadline even with zero successes", async () => { + setExcludedSearchProviders(NON_TEST_ENGINES); + const fetchMock: FetchImpl = input => { + const url = typeof input === "string" ? input : input.toString(); + if (url.includes("duckduckgo.com")) { + return Promise.resolve(new Response(DDG_CHALLENGE, { status: 200 })); + } + // google: never settles and ignores abort — only the hard cap can end the wait. + const { promise } = Promise.withResolvers(); + return promise; + }; + + const response = await searchPublicWeb(makeParams("hard cap", fetchMock), { softMs: 10, hardMs: 40 }); + + expect(response.provider).toBe("public"); + expect(response.sources).toEqual([]); + }); + + it("fails with an aggregated provider-tagged error when every engine fails", async () => { + setExcludedSearchProviders(NON_TEST_ENGINES); + const fetchMock = makeFetchMock({ ddg: DDG_CHALLENGE, google: GOOGLE_CHALLENGE }); + + try { + await searchPublicWeb(makeParams("all blocked", fetchMock)); + expect.unreachable("all-engine failure should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + const providerError = error as SearchProviderError; + expect(providerError.provider).toBe("public"); + expect(providerError.status).toBe(503); + expect(providerError.message).toContain("duckduckgo:"); + expect(providerError.message).toContain("google:"); + } + }); + + it("rejects when settings exclude every credential-free engine", async () => { + setExcludedSearchProviders([...NON_TEST_ENGINES, "duckduckgo", "google"]); + const fetchMock: FetchImpl = () => Promise.reject(new Error("no engine should be queried")); + + try { + await searchPublicWeb(makeParams("nothing left", fetchMock)); + expect.unreachable("fully excluded fan-out should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "public", status: 400 }); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-startpage.test.ts b/packages/coding-agent/test/tools/web-search-startpage.test.ts new file mode 100644 index 000000000..a826a20d8 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-startpage.test.ts @@ -0,0 +1,243 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import type { SearchParams } from "@oh-my-pi/pi-coding-agent/web/search/providers/base"; +import { searchStartpage } from "@oh-my-pi/pi-coding-agent/web/search/providers/startpage"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("Startpage search must not request API keys"); + }, + resolver() { + throw new Error("Startpage search must not request credential resolvers"); + }, + hasAuth() { + throw new Error("Startpage search must not check auth"); + }, +} as unknown as AuthStorage; + +function makeParams(query: string, fetch: FetchImpl): SearchParams { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "Startpage search test prompt", + fetch, + }; +} + +const SC_TOKEN = "2sbbv9IndMZLVHNdHqjDurhVAo8vrQSiu3q8EtaSRZMbFzmQ0Qt1"; + +/** Homepage shell trimmed from a live capture: the `/sp/search` form with its hidden inputs. */ +function homepageHtml(sc: string): string { + return ` + + `; +} + +/** Organic result block trimmed from a live capture. */ +function resultHtml(url: string, title: string, snippet?: string): string { + return `
      +
      +
      + +
      + +

      ${title}

      +
      + ${snippet ? `

      ${snippet}

      ` : ""} +
      `; +} + +function resultsPage(...blocks: string[]): string { + return `Startpage Search Results +
      ${blocks.join("\n")}
      +
      + `; +} + +/** CAPTCHA/error SPA shell trimmed from a live `/en/errors/` capture. */ +const CHALLENGE_HTML = `
      `; + +interface CapturedRequest { + url: string; + init: RequestInit | undefined; +} + +/** Dispatch mocked responses by URL; records every request for assertions. */ +function dispatchFetch(routes: Record Response>, captured: CapturedRequest[]): FetchImpl { + return (input, init) => { + const url = typeof input === "string" ? input : input.toString(); + captured.push({ url, init }); + const pathname = new URL(url).pathname; + const route = routes[pathname]; + if (!route) throw new Error(`Unexpected request: ${url}`); + return Promise.resolve(route()); + }; +} + +describe("Startpage web search provider", () => { + it("performs the homepage-token dance: GET home, then POST the form inputs with query and recency", async () => { + const captured: CapturedRequest[] = []; + const fetchMock = dispatchFetch( + { + "/": () => new Response(homepageHtml(SC_TOKEN), { status: 200 }), + "/sp/search": () => + new Response(resultsPage(resultHtml("https://example.com/result", "Result", "Search snippet")), { + status: 200, + }), + }, + captured, + ); + + const response = await searchStartpage({ + ...makeParams("browser headers & parsing", fetchMock), + numSearchResults: 99, + recency: "week", + }); + + expect(captured.map(r => r.url)).toEqual(["https://www.startpage.com/", "https://www.startpage.com/sp/search"]); + + const homeInit = captured[0].init; + expect(homeInit?.method).toBeUndefined(); + const homeHeaders = new Headers(homeInit?.headers); + expect(homeHeaders.get("accept")).toContain("text/html"); + expect(homeHeaders.get("user-agent")).toMatch(/Chrome\/\d+\.0\.0\.0/); + expect(homeHeaders.get("sec-fetch-site")).toBe("none"); + + const searchInit = captured[1].init; + expect(searchInit?.method).toBe("POST"); + const searchHeaders = new Headers(searchInit?.headers); + expect(searchHeaders.get("content-type")).toBe("application/x-www-form-urlencoded"); + expect(searchHeaders.get("referer")).toBe("https://www.startpage.com/"); + expect(searchHeaders.get("sec-fetch-site")).toBe("same-origin"); + expect(searchHeaders.get("sec-fetch-dest")).toBe("document"); + expect(searchHeaders.get("user-agent")).toMatch(/Chrome\/\d+\.0\.0\.0/); + + const form = new URLSearchParams(String(searchInit?.body)); + expect(form.get("query")).toBe("browser headers & parsing"); + expect(form.get("sc")).toBe(SC_TOKEN); + expect(form.get("with_date")).toBe("w"); + expect(form.get("cat")).toBe("home"); + expect(form.get("segment")).toBe("startpage.udog"); + expect(form.get("abp")).toBe("0"); + expect(form.get("lui")).toBe("english"); + + expect(response.provider).toBe("startpage"); + expect(response.sources).toEqual([ + { title: "Result", url: "https://example.com/result", snippet: "Search snippet" }, + ]); + }); + + it("parses result blocks, skips junk rows, deduplicates targets, and clamps to the requested count", async () => { + const html = resultsPage( + resultHtml("https://example.com/a", "First & best result", "A useful\n\tsnippet"), + resultHtml("https://example.com/a", "Duplicate of first", "duplicate"), + `

      No title anchor: instant-answer widget

      `, + resultHtml("https://www.startpage.com/en/privacy", "Internal Startpage link", "never a result"), + resultHtml("javascript:void(0)", "Bad scheme", "never a result"), + resultHtml("https://example.com/b", "Second result"), + resultHtml("https://example.com/c", "Third result", "clamped away"), + ); + const captured: CapturedRequest[] = []; + const fetchMock = dispatchFetch( + { + "/": () => new Response(homepageHtml(SC_TOKEN), { status: 200 }), + "/sp/search": () => new Response(html, { status: 200 }), + }, + captured, + ); + + const response = await searchStartpage({ ...makeParams("mixed markup", fetchMock), numSearchResults: 2 }); + + expect(response.sources).toEqual([ + { title: "First & best result", url: "https://example.com/a", snippet: "A useful snippet" }, + { title: "Second result", url: "https://example.com/b", snippet: undefined }, + ]); + }); + + it("falls back to a direct GET with query params when the homepage yields no sc token", async () => { + const captured: CapturedRequest[] = []; + const fetchMock = dispatchFetch( + { + "/": () => new Response("redesigned homepage without the form", { status: 200 }), + "/sp/search": () => + new Response(resultsPage(resultHtml("https://example.com/fallback", "Fallback", "via GET")), { + status: 200, + }), + }, + captured, + ); + + const response = await searchStartpage({ ...makeParams("fallback query", fetchMock), recency: "month" }); + + expect(captured).toHaveLength(2); + const searchUrl = new URL(captured[1].url); + expect(searchUrl.origin + searchUrl.pathname).toBe("https://www.startpage.com/sp/search"); + expect(searchUrl.searchParams.get("query")).toBe("fallback query"); + expect(searchUrl.searchParams.get("with_date")).toBe("m"); + expect(captured[1].init?.method).toBeUndefined(); + const headers = new Headers(captured[1].init?.headers); + expect(headers.get("referer")).toBe("https://www.startpage.com/"); + expect(headers.get("sec-fetch-site")).toBe("same-origin"); + expect(response.sources).toEqual([ + { title: "Fallback", url: "https://example.com/fallback", snippet: "via GET" }, + ]); + }); + + it("surfaces the CAPTCHA/error shell as a provider-tagged 429", async () => { + const captured: CapturedRequest[] = []; + const fetchMock = dispatchFetch( + { + "/": () => new Response(homepageHtml(SC_TOKEN), { status: 200 }), + "/sp/search": () => new Response(CHALLENGE_HTML, { status: 200 }), + }, + captured, + ); + + try { + await searchStartpage(makeParams("blocked", fetchMock)); + expect.unreachable("Startpage CAPTCHA shell should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "startpage", status: 429 }); + expect((error as SearchProviderError).message).toContain("Startpage"); + expect((error as SearchProviderError).message).toContain("CAPTCHA"); + } + }); + + it("maps non-OK search responses to a status-tagged provider error", async () => { + const captured: CapturedRequest[] = []; + const fetchMock = dispatchFetch( + { + "/": () => new Response(homepageHtml(SC_TOKEN), { status: 200 }), + "/sp/search": () => new Response("upstream exploded", { status: 500 }), + }, + captured, + ); + + try { + await searchStartpage(makeParams("server error", fetchMock)); + expect.unreachable("HTTP 500 should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "startpage", + status: 500, + message: "Startpage HTML error (500)", + }); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-yahoo.test.ts b/packages/coding-agent/test/tools/web-search-yahoo.test.ts new file mode 100644 index 000000000..df0167e14 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-yahoo.test.ts @@ -0,0 +1,188 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import type { SearchParams } from "@oh-my-pi/pi-coding-agent/web/search/providers/base"; +import { searchYahoo } from "@oh-my-pi/pi-coding-agent/web/search/providers/yahoo"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("Yahoo search must not request API keys"); + }, + resolver() { + throw new Error("Yahoo search must not request credential resolvers"); + }, + hasAuth() { + throw new Error("Yahoo search must not check auth"); + }, +} as unknown as AuthStorage; + +function makeParams(query: string, fetch: FetchImpl): SearchParams { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "Yahoo search test prompt", + fetch, + }; +} + +/** Current Yahoo layout: tracker `` wraps a breadcrumb div plus the `

      `. */ +function wrappedResult(target: string, title: string, snippet?: string): string { + const ru = encodeURIComponent(target); + return `
    11. `; +} + +function serp(body: string): string { + return `

      Search Results

      +
        ${body}
      `; +} + +describe("Yahoo web search provider", () => { + it("requests the HTML result page with browser navigation headers, count, and recency", async () => { + let capturedUrl = ""; + let capturedInit: RequestInit | undefined; + const fetchMock: FetchImpl = (input, init) => { + capturedUrl = typeof input === "string" ? input : input.toString(); + capturedInit = init; + return Promise.resolve( + new Response(serp(wrappedResult("https://example.com/result", "Result", "Search snippet")), { + status: 200, + headers: { "Content-Type": "text/html" }, + }), + ); + }; + + const response = await searchYahoo({ + ...makeParams("browser headers & parsing", fetchMock), + numSearchResults: 99, + recency: "week", + }); + + const url = new URL(capturedUrl); + expect(url.origin + url.pathname).toBe("https://search.yahoo.com/search"); + expect(url.searchParams.get("p")).toBe("browser headers & parsing"); + expect(url.searchParams.get("n")).toBe("20"); + expect(url.searchParams.get("btf")).toBe("w"); + expect(capturedInit?.method).toBeUndefined(); + const headers = new Headers(capturedInit?.headers); + expect(headers.get("accept")).toContain("text/html"); + expect(headers.get("user-agent")).toMatch(/Chrome\/\d+\.0\.0\.0/); + expect(headers.get("referer")).toBe("https://search.yahoo.com/"); + expect(headers.get("sec-fetch-dest")).toBe("document"); + expect(headers.get("sec-fetch-mode")).toBe("navigate"); + expect(headers.get("sec-fetch-site")).toBe("same-origin"); + expect(response.sources).toEqual([ + { title: "Result", url: "https://example.com/result", snippet: "Search snippet" }, + ]); + }); + + it("maps day and month recency to btf and silently drops the unsupported year filter", async () => { + const captured: string[] = []; + const fetchMock: FetchImpl = input => { + captured.push(typeof input === "string" ? input : input.toString()); + return Promise.resolve(new Response(serp(""), { status: 200 })); + }; + + await searchYahoo({ ...makeParams("q", fetchMock), recency: "day" }); + await searchYahoo({ ...makeParams("q", fetchMock), recency: "month" }); + await searchYahoo({ ...makeParams("q", fetchMock), recency: "year" }); + await searchYahoo(makeParams("q", fetchMock)); + + expect(new URL(captured[0]).searchParams.get("btf")).toBe("d"); + expect(new URL(captured[1]).searchParams.get("btf")).toBe("m"); + expect(new URL(captured[2]).searchParams.get("btf")).toBeNull(); + expect(new URL(captured[2]).searchParams.get("p")).toBe("q"); + expect(new URL(captured[3]).searchParams.get("btf")).toBeNull(); + }); + + it("unwraps /RU= tracker links, handles legacy plain hrefs, deduplicates, and skips junk rows", async () => { + const target = "https://example.com/page?a=1&b=2"; + const html = serp( + [ + wrappedResult( + "https://bun.sh/", + "Bun — A fast all-in-one JavaScript runtime", + 'Jan 3, 2010 · Bundle, install, and run JavaScript & TypeScript.', + ), + // Legacy layout: anchor nested inside the h3, plain unwrapped href. + `
    12. Legacy snippet

    13. `, + // Same target again: must deduplicate. + wrappedResult(target, "Duplicate target", "duplicate"), + // Tracker link without a recoverable /RU= segment: skipped. + `
    14. `, + // Internal navigation resolves to search.yahoo.com: skipped. + `
    15. `, + // Module header h3 outside any .algo block: never considered. + `
    16. Videos

    17. `, + ].join("\n"), + ); + const fetchMock: FetchImpl = () => Promise.resolve(new Response(html, { status: 200 })); + + const response = await searchYahoo({ ...makeParams("mixed markup", fetchMock), numSearchResults: 10 }); + + expect(response.provider).toBe("yahoo"); + expect(response.sources).toEqual([ + { + title: "Bun — A fast all-in-one JavaScript runtime", + url: "https://bun.sh/", + snippet: "Jan 3, 2010 · Bundle, install, and run JavaScript & TypeScript.", + }, + { + title: "Legacy result", + url: target, + snippet: "Legacy snippet", + }, + ]); + }); + + it("clamps the parsed results to the requested count", async () => { + const html = serp( + Array.from({ length: 5 }, (_, i) => + wrappedResult(`https://example.com/${i}`, `Result ${i}`, `snippet ${i}`), + ).join("\n"), + ); + const fetchMock: FetchImpl = () => Promise.resolve(new Response(html, { status: 200 })); + + const response = await searchYahoo({ ...makeParams("clamp", fetchMock), numSearchResults: 2 }); + + expect(response.sources.map(s => s.url)).toEqual(["https://example.com/0", "https://example.com/1"]); + }); + + it("surfaces the GDPR consent interstitial as a provider-tagged 429", async () => { + const consent = ``; + const fetchMock: FetchImpl = () => Promise.resolve(new Response(consent, { status: 200 })); + + try { + await searchYahoo(makeParams("blocked", fetchMock)); + expect.unreachable("Yahoo consent interstitial should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "yahoo", status: 429 }); + expect((error as SearchProviderError).message).toContain("consent"); + } + }); + + it("maps non-OK HTTP responses to a provider-tagged error with the upstream status", async () => { + const fetchMock: FetchImpl = () => Promise.resolve(new Response("upstream broke", { status: 503 })); + + try { + await searchYahoo(makeParams("unavailable", fetchMock)); + expect.unreachable("HTTP 503 should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "yahoo", + status: 503, + message: "Yahoo HTML error (503)", + }); + } + }); +});