From 64e4f00e3deb76edbba9bb522396c92d45b849ed Mon Sep 17 00:00:00 2001 From: zekdevs <38579990+zekdevs@users.noreply.github.com> Date: Fri, 26 Jun 2026 08:41:20 -0600 Subject: [PATCH 1/6] add xai, ddg, firecrawl, and tinyfish as web_search providers --- README.md | 28 +- docs/tools/web_search.md | 109 +++++--- packages/ai/src/stream.ts | 2 + packages/coding-agent/CHANGELOG.md | 4 + packages/coding-agent/src/cli/args.ts | 2 + .../coding-agent/src/cli/web-search-cli.ts | 2 +- .../coding-agent/src/web/search/provider.ts | 90 +++--- .../src/web/search/providers/duckduckgo.ts | 140 ++++++++++ .../src/web/search/providers/firecrawl.ts | 144 ++++++++++ .../src/web/search/providers/tinyfish.ts | 134 +++++++++ .../src/web/search/providers/xai.ts | 226 +++++++++++++++ packages/coding-agent/src/web/search/types.ts | 8 +- .../test/tools/web-search-duckduckgo.test.ts | 188 +++++++++++++ .../test/tools/web-search-firecrawl.test.ts | 138 ++++++++++ .../test/tools/web-search-tinyfish.test.ts | 127 +++++++++ .../test/tools/web-search-xai.test.ts | 259 ++++++++++++++++++ 16 files changed, 1509 insertions(+), 92 deletions(-) create mode 100644 packages/coding-agent/src/web/search/providers/duckduckgo.ts create mode 100644 packages/coding-agent/src/web/search/providers/firecrawl.ts create mode 100644 packages/coding-agent/src/web/search/providers/tinyfish.ts create mode 100644 packages/coding-agent/src/web/search/providers/xai.ts create mode 100644 packages/coding-agent/test/tools/web-search-duckduckgo.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-firecrawl.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-tinyfish.test.ts create mode 100644 packages/coding-agent/test/tools/web-search-xai.test.ts diff --git a/README.md b/README.md index ab6e6682c..15e7a878e 100644 --- a/README.md +++ b/README.md @@ -151,7 +151,7 @@ _[Watch the capture ↗](https://omp.sh/clips/collab.mp4)_ ### 08 · Read a pdf on arxiv, why not? -web_search chains fourteen ranked providers and hands whatever URLs it finds straight to read. Arxiv PDFs, GitHub pages, Stack Overflow threads come back as structured markdown with anchors intact — the same tool surface you use on local files. Cite, follow, quote, never lose where you came from. +web_search chains eighteen ranked providers and hands whatever URLs it finds straight to read. Arxiv PDFs, GitHub pages, Stack Overflow threads come back as structured markdown with anchors intact — the same tool surface you use on local files. Cite, follow, quote, never lose where you came from. ![omp TUI: web_search returns 10 ranked Perplexity sources for inference-time compute scaling, the agent picks an arxiv paper, calls read https://arxiv.org/pdf/2604.10739v1, and summarizes the paper's headline result with real numbers.](https://omp.sh/clips/web-poster.webp) @@ -309,31 +309,35 @@ Ollama `local` · Ollama Cloud · LM Studio `local` · llama.cpp `local` · vLLM Full provider & routing reference at [omp.sh/docs/providers](https://omp.sh/docs/providers). -## Fourteen backends. _One tool the agent already knows_. +## Eighteen backends. _One tool the agent already knows_. -`web_search` is built in, not bolted on. `auto` walks a fourteen-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive. +`web_search` is built in, not bolted on. `auto` walks an eighteen-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive. ### Search providers -Fourteen backends. Pin one, or let `auto` walk the chain in order. +Eighteen backends. Pin one, or let `auto` walk the chain in order. | provider | auth | | ------------ | ---------------------- | | `auto` | chain | -| `exa` | `EXA_API_KEY` (or mcp) | -| `brave` | `BRAVE_API_KEY` | -| `jina` | `JINA_API_KEY` | -| `kimi` | `MOONSHOT_API_KEY` | -| `zai` | `ZAI_API_KEY` | -| `anthropic` | oauth | | `perplexity` | `PERPLEXITY_API_KEY` | | `gemini` | oauth | +| `anthropic` | oauth | | `codex` | oauth | -| `tavily` | `TAVILY_API_KEY` | -| `parallel` | `PARALLEL_API_KEY` | +| `xai` | `XAI_API_KEY` | +| `zai` | `ZAI_API_KEY` | +| `exa` | `EXA_API_KEY` (or mcp) | +| `tinyfish` | `TINYFISH_API_KEY` | +| `jina` | `JINA_API_KEY` | | `kagi` | `KAGI_API_KEY` | +| `tavily` | `TAVILY_API_KEY` | +| `firecrawl` | `FIRECRAWL_API_KEY` | +| `brave` | `BRAVE_API_KEY` | +| `kimi` | `MOONSHOT_API_KEY` | +| `parallel` | `PARALLEL_API_KEY` | | `synthetic` | `SYNTHETIC_API_KEY` | | `searxng` | self-hosted | +| `duckduckgo` | no key | ### Specialised handlers diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 26cd5f3cb..602824544 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -14,7 +14,9 @@ - `packages/coding-agent/src/web/search/providers/anthropic.ts` — Claude web-search provider. - `packages/coding-agent/src/web/search/providers/brave.ts` — Brave Search API adapter. - `packages/coding-agent/src/web/search/providers/codex.ts` — OpenAI Codex SSE adapter. + - `packages/coding-agent/src/web/search/providers/duckduckgo.ts` — DuckDuckGo Instant Answer API adapter. - `packages/coding-agent/src/web/search/providers/exa.ts` — Exa API or MCP adapter. + - `packages/coding-agent/src/web/search/providers/firecrawl.ts` — Firecrawl search adapter. - `packages/coding-agent/src/web/search/providers/gemini.ts` — Gemini grounding SSE adapter. - `packages/coding-agent/src/web/search/providers/jina.ts` — Jina Reader search adapter. - `packages/coding-agent/src/web/search/providers/kagi.ts` — Kagi provider wrapper. @@ -24,6 +26,8 @@ - `packages/coding-agent/src/web/search/providers/searxng.ts` — self-hosted SearXNG adapter. - `packages/coding-agent/src/web/search/providers/synthetic.ts` — Synthetic search adapter. - `packages/coding-agent/src/web/search/providers/tavily.ts` — Tavily search adapter. + - `packages/coding-agent/src/web/search/providers/tinyfish.ts` — TinyFish search adapter. + - `packages/coding-agent/src/web/search/providers/xai.ts` — xAI Responses web-search adapter. - `packages/coding-agent/src/web/search/providers/zai.ts` — Z.AI remote MCP adapter. - `packages/coding-agent/src/web/parallel.ts` — Parallel search/extract HTTP client. - `packages/coding-agent/src/web/kagi.ts` — Kagi HTTP client. @@ -34,11 +38,11 @@ | Field | Type | Required | Description | | --- | --- | --- | --- | | `query` | `string` | Yes | Search query, passed to providers unchanged. | -| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, and Kagi. | +| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. | | `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. | -| `max_tokens` | `number` | No | Passed through as `maxOutputTokens` / `max_tokens` only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. | -| `temperature` | `number` | No | Passed through only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. | -| `num_search_results` | `number` | No | Requested upstream search breadth. For most providers this is the same count used for returned sources. Perplexity is the only adapter that keeps it distinct from `limit`. | +| `max_tokens` | `number` | No | Passed through as provider token caps (`maxOutputTokens`, `max_tokens`, or xAI `max_output_tokens`) only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | +| `temperature` | `number` | No | Passed through only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | +| `num_search_results` | `number` | No | Requested upstream search breadth. Most providers use this as returned source count. Perplexity keeps it distinct from `limit`; xAI does not send a source-count parameter to Responses API. | ## Outputs The tool returns a single text content block plus structured `details`. @@ -73,7 +77,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - if `params.provider` is set and not `"auto"`, it loads that provider with `getSearchProvider()`; if `isExplicitlyAvailable()` returns true, the list is `[that provider]`, otherwise it falls back to `resolveProviderChain(authStorage, "auto")`. - otherwise it calls `resolveProviderChain()` with the module-global preferred provider from `packages/coding-agent/src/web/search/provider.ts`. 3. `resolveProviderChain()` lazily loads each provider module on demand and returns only available providers. If a preferred provider is set, it is tried first (gated by `isExplicitlyAvailable()`), then the static `SEARCH_PROVIDER_ORDER` excluding that provider, each gated by `isAvailable()`. Providers in the excluded set (`setExcludedSearchProviders()`) are skipped entirely, including as the preferred candidate. -4. If no providers are available, `executeSearch()` returns `Error: No web search provider configured.` with `details.response.provider = "none"`. +4. If no providers are available (for example, after excluding DuckDuckGo and lacking configured keyed/OAuth providers), `executeSearch()` returns `Error: No web search provider configured.` with `details.response.provider = "none"`. 5. For each provider in order, `executeSearch()` calls `provider.search()` with: - `query`, - `limit`, `recency`, `temperature`, `maxOutputTokens`, `numSearchResults`, @@ -91,36 +95,20 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - **Forced provider**: internal callers may pass `provider`; unavailable forced providers fall back to the auto chain instead of hard-failing (`packages/coding-agent/src/web/search/index.ts`). This field is not in the model-facing schema. - **Preferred provider**: `setPreferredSearchProvider()` sets a module-global default used by `resolveProviderChain()`. `packages/coding-agent/src/sdk.ts` and `packages/coding-agent/src/modes/controllers/selector-controller.ts` wire this from settings. - **Excluded providers**: `setExcludedSearchProviders()` records providers `resolveProviderChain()` must never return, including as fallbacks. Wired from the `providers.webSearchExclude` setting (`providers.webSearch` drives the preferred provider) in `packages/coding-agent/src/sdk.ts`, `packages/coding-agent/src/modes/interactive-mode.ts`, and `packages/coding-agent/src/modes/controllers/selector-controller.ts`. - - **Auto chain order**: `perplexity`, `gemini`, `anthropic`, `codex`, `zai`, `exa`, `jina`, `kagi`, `tavily`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). + - **Auto chain order** (18 providers): `perplexity`, `gemini`, `anthropic`, `codex`, `xai`, `zai`, `exa`, `tinyfish`, `jina`, `kagi`, `tavily`, `firecrawl`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng`, `duckduckgo` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). - **Provider adapters** - - **Tavily** — `packages/coding-agent/src/web/search/providers/tavily.ts` - - Availability: API key from env or `agent.db` via `findCredential()`. - - Querying: POST `https://api.tavily.com/search`. - - `recency` maps to Tavily `time_range`; code explicitly keeps `topic` at default general scope instead of narrowing to news. - - `limit` / `num_search_results`: adapter uses `params.numSearchResults ?? params.limit`, clamped to `5..20` with default `5`. - - Output: `answer`, `sources`, `requestId`, `authMode: "api_key"`. - **Perplexity** — `packages/coding-agent/src/web/search/providers/perplexity.ts` - Availability: auth precedence is `PERPLEXITY_COOKIES` -> OAuth token in `agent.db` -> `PERPLEXITY_API_KEY` / `PPLX_API_KEY` -> anonymous ask-endpoint fallback. `isAvailable()` gates the auto chain on credentials, but `isExplicitlyAvailable()` is always true, so explicit selection works unauthenticated. - OAuth/cookie/anonymous mode: POSTs to `https://www.perplexity.ai/rest/sse/perplexity_ask`, consumes SSE, merges partial events, extracts answer and source URLs, sets `authMode: "oauth"` (`"anonymous"` for the unauthenticated fallback). - API-key mode: POSTs to `https://api.perplexity.ai/chat/completions` with `model: "sonar-pro"`, `search_mode: "web"`, `num_search_results`, optional `search_recency_filter`, `max_tokens`, `temperature`. - `num_search_results` controls upstream API breadth only in API-key mode. `limit` is preserved separately as `num_results` and slices returned `sources` after parsing in both auth modes. - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode`. - - **Brave** — `packages/coding-agent/src/web/search/providers/brave.ts` - - Availability: `BRAVE_API_KEY` only. - - Querying: GET `https://api.search.brave.com/res/v1/web/search` with `count`, `extra_snippets=true`, and `freshness=pd|pw|pm|py` for `recency`. - - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. - - Output: `sources`, `requestId`. - - **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts` - - Availability: `JINA_API_KEY` only. - - Querying: GET-like fetch to `https://s.jina.ai/` with bearer auth. - - Ignores `recency`, `max_tokens`, and `temperature`. - - `limit` / `num_search_results`: adapter slices sources to `params.numSearchResults ?? params.limit` when provided; otherwise returns all payload items. - - Output: `sources` only. - - **Kimi** — `packages/coding-agent/src/web/search/providers/kimi.ts` - - Availability: `MOONSHOT_SEARCH_API_KEY`, `KIMI_SEARCH_API_KEY`, `MOONSHOT_API_KEY`, or `agent.db` credentials for `moonshot` / `kimi-code`. - - Querying: POST to `MOONSHOT_SEARCH_BASE_URL` / `KIMI_SEARCH_BASE_URL` / default `https://api.kimi.com/coding/v1/search` with `text_query`, `limit`, `enable_page_crawling`, `timeout_seconds: 30`. - - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. - - Output: `sources`, `requestId`. + - **Gemini** — `packages/coding-agent/src/web/search/providers/gemini.ts` + - Availability: OAuth credentials in `agent.db` for `google-gemini-cli` or `google-antigravity`. + - Querying: SSE `streamGenerateContent` call with Google Search grounding enabled. Antigravity auth tries two fallback endpoints and retries `401/403/400 invalid auth` once after token refresh; `429/5xx` retry with exponential backoff and server-provided retry delay, capped by a `5 * 60 * 1000` ms rate-limit budget. + - `max_tokens` and `temperature` pass through as `generationConfig.maxOutputTokens` / `generationConfig.temperature`. + - `limit` and `num_search_results` are collapsed together before dispatch. + - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage`, `model`. - **Anthropic** — `packages/coding-agent/src/web/search/providers/anthropic.ts` - Availability: `ANTHROPIC_SEARCH_API_KEY` env var, otherwise `authStorage.hasAuth("anthropic")`; search credentials come from `authStorage.getApiKey("anthropic")` when no search-specific key is set. - Env overrides specific to search (do not affect chat completions): @@ -131,18 +119,17 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - `max_tokens` and `temperature` pass through. - `limit` and `num_search_results` are collapsed together before dispatch: `num_results = params.numSearchResults ?? params.limit`. - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage.searchRequests`, `model`, `requestId`. - - **Gemini** — `packages/coding-agent/src/web/search/providers/gemini.ts` - - Availability: OAuth credentials in `agent.db` for `google-gemini-cli` or `google-antigravity`. - - Querying: SSE `streamGenerateContent` call with Google Search grounding enabled. Antigravity auth tries two fallback endpoints and retries `401/403/400 invalid auth` once after token refresh; `429/5xx` retry with exponential backoff and server-provided retry delay, capped by a `5 * 60 * 1000` ms rate-limit budget. - - `max_tokens` and `temperature` pass through as `generationConfig.maxOutputTokens` / `generationConfig.temperature`. - - `limit` and `num_search_results` are collapsed together before dispatch. - - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage`, `model`. - **Codex** — `packages/coding-agent/src/web/search/providers/codex.ts` - Availability: OAuth credential for `openai-codex` in `agent.db` (`hasOAuth()`; expiry is not checked here — refresh is lazy in `searchCodex`). - Querying: SSE POST to `https://chatgpt.com/backend-api/codex/responses` with `tool_choice: { type: "web_search" }` and `search_context_size: "high"` by default. - Ignores `recency`, `max_tokens`, and `temperature` in this tool path. - `limit` and `num_search_results` are collapsed together before dispatch. - Output may include `answer`, `sources`, `usage`, `model`, `requestId`. If the streamed response has no `url_citation` annotations, the adapter falls back to scraping markdown links and bare URLs from the answer text. + - **xAI** — `packages/coding-agent/src/web/search/providers/xai.ts` + - Availability: `XAI_API_KEY` or `agent.db` credential for `xai`. + - Querying: POST `https://api.x.ai/v1/responses` with model `grok-4.3` and `tools: [{ type: "web_search" }]`. + - `max_tokens` and `temperature` pass through; `limit`, `num_search_results`, and `recency` are not sent. + - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode: "api_key"`. - **Z.AI** — `packages/coding-agent/src/web/search/providers/zai.ts` - Availability: env or `agent.db` credential for `zai`. - Querying: JSON-RPC `tools/call` against `https://api.z.ai/api/mcp/web_search_prime/mcp` for remote MCP tool `web_search_prime`. @@ -154,17 +141,47 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Querying: POST `https://api.exa.ai/search` with the resolved Exa API key, otherwise JSON-RPC `tools/call` against `https://mcp.exa.ai/mcp` for remote MCP tool `web_search_exa`. - `limit` and `num_search_results` are collapsed together before dispatch. - Output: synthesized `answer` from up to 3 result summaries, `sources`, `requestId`. + - **TinyFish** — `packages/coding-agent/src/web/search/providers/tinyfish.ts` + - Availability: `TINYFISH_API_KEY` or `agent.db` credential for `tinyfish`. + - Querying: GET `https://api.search.tinyfish.ai` with `X-API-Key`; `recency` maps to `recency_minutes`. + - `limit` / `num_search_results`: collapsed and clamped to `1..20`, default `10`; output `sources`, `authMode: "api_key"`. + - **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts` + - Availability: `JINA_API_KEY` only. + - Querying: GET-like fetch to `https://s.jina.ai/` with bearer auth. + - Ignores `recency`, `max_tokens`, and `temperature`. + - `limit` / `num_search_results`: adapter slices sources to `params.numSearchResults ?? params.limit` when provided; otherwise returns all payload items. + - Output: `sources` only. + - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` + - Availability: env or `agent.db` credential for `kagi`. + - Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer ` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`). + - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. + - Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`). + - **Tavily** — `packages/coding-agent/src/web/search/providers/tavily.ts` + - Availability: API key from env or `agent.db` via `findCredential()`. + - Querying: POST `https://api.tavily.com/search`. + - `recency` maps to Tavily `time_range`; code explicitly keeps `topic` at default general scope instead of narrowing to news. + - `limit` / `num_search_results`: adapter uses `params.numSearchResults ?? params.limit`, clamped to `5..20` with default `5`. + - Output: `answer`, `sources`, `requestId`, `authMode: "api_key"`. + - **Firecrawl** — `packages/coding-agent/src/web/search/providers/firecrawl.ts` + - Availability: `FIRECRAWL_API_KEY` or `agent.db` credential for `firecrawl`. + - Querying: POST `https://api.firecrawl.dev/v2/search` with `sources: [{ type: "web" }]`; `recency` maps to Google-style `tbs`. + - `limit` / `num_search_results`: collapsed and clamped to `1..100`, default `10`; output `sources`, `requestId`, `authMode: "api_key"`. + - **Brave** — `packages/coding-agent/src/web/search/providers/brave.ts` + - Availability: `BRAVE_API_KEY` only. + - Querying: GET `https://api.search.brave.com/res/v1/web/search` with `count`, `extra_snippets=true`, and `freshness=pd|pw|pm|py` for `recency`. + - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. + - Output: `sources`, `requestId`. + - **Kimi** — `packages/coding-agent/src/web/search/providers/kimi.ts` + - Availability: `MOONSHOT_SEARCH_API_KEY`, `KIMI_SEARCH_API_KEY`, `MOONSHOT_API_KEY`, or `agent.db` credentials for `moonshot` / `kimi-code`. + - Querying: POST to `MOONSHOT_SEARCH_BASE_URL` / `KIMI_SEARCH_BASE_URL` / default `https://api.kimi.com/coding/v1/search` with `text_query`, `limit`, `enable_page_crawling`, `timeout_seconds: 30`. + - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. + - Output: `sources`, `requestId`. - **Parallel** — `packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts` - Availability: env or `agent.db` credential for `parallel`. - Querying: POST `https://api.parallel.ai/v1beta/search` with `objective=query`, `search_queries=[query]`, `mode:"fast"`, `max_chars_per_result: 10000`, beta header `search-extract-2025-10-10`. - There is no provider fan-out here despite the name; the current adapter always sends a one-element `search_queries` array. - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. - Output: `sources`, `requestId`. - - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` - - Availability: env or `agent.db` credential for `kagi`. - - Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer ` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`). - - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. - - Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`). - **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts` - Availability: env or `agent.db` credential for `synthetic`. - Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`. @@ -178,6 +195,10 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - `recency` maps to `time_range`; `week` is downgraded to `month` because SearXNG does not support week. - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..20`, default `10`. - Output: `sources`, `relatedQuestions` from `suggestions`. + - **DuckDuckGo** — `packages/coding-agent/src/web/search/providers/duckduckgo.ts` + - Availability: always available; no API key. + - Querying: GET official Instant Answer API `https://api.duckduckgo.com/` with JSON/no-HTML flags; no scraped HTML. + - `limit` / `num_search_results`: collapsed and clamped to `1..20`, default `10`; output may include `answer` and `sources` from abstracts/results/topics. ## Side Effects - Network @@ -193,11 +214,14 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Many provider adapters accept `AbortSignal`; `WebSearchTool.execute()` passes the tool call signal into `executeSearch()`, which forwards it as `params.signal` to providers and rethrows cancellation during fallback. ## Limits & Caps -- Provider auto-order length: 14 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). +- Provider auto-order length: 18 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). - `formatForLLM()` truncates source snippets and citation text to 240 chars (`packages/coding-agent/src/web/search/index.ts`). - `formatForLLM()` emits at most 3 search queries, each truncated to 120 chars (`packages/coding-agent/src/web/search/index.ts`). - Brave result count: default `10`, max `20` (`DEFAULT_NUM_RESULTS`, `MAX_NUM_RESULTS` in `packages/coding-agent/src/web/search/providers/brave.ts`). +- TinyFish result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/tinyfish.ts`). +- DuckDuckGo result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/duckduckgo.ts`). - Tavily result count: default `5`, max `20` (`packages/coding-agent/src/web/search/providers/tavily.ts`). +- Firecrawl result count: default `10`, max `100` (`packages/coding-agent/src/web/search/providers/firecrawl.ts`). - Kimi result count: default `10`, max `20`; request timeout field fixed to `30` seconds (`packages/coding-agent/src/web/search/providers/kimi.ts`). - Parallel result count: default `10`, max `40`; per-result excerpt cap `10_000` chars (`packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts`). - Kagi result count: default `10`, max `40` (`packages/coding-agent/src/web/search/providers/kagi.ts`). @@ -222,7 +246,8 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec ## Notes - The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. - `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. -- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity is the only implementation that preserves both concepts. -- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, and Kagi; the model-facing prompt does not name specific providers. +- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI currently ignores both. +- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; the model-facing prompt does not name specific providers. - `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain. +- DuckDuckGo is intentionally last in the auto chain because it is always available without credentials. - Exa uses `authStorage.getApiKey("exa")`, then `EXA_API_KEY`, then unauthenticated `https://mcp.exa.ai/mcp` fallback. diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 576bd6116..67fab9948 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -166,6 +166,8 @@ const LEGACY_ENV_KEYS: Record = { exa: "EXA_API_KEY", jina: "JINA_API_KEY", brave: "BRAVE_API_KEY", + tinyfish: "TINYFISH_API_KEY", + firecrawl: "FIRECRAWL_API_KEY", }; /** diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1334b25e0..4afb641fb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added TinyFish, DuckDuckGo, xAI, and Firecrawl web_search providers. + ### Fixed - Fixed Kimi-family models defaulting to hashline edit mode; they now fall back to `replace` unless `edit.modelVariants`, `PI_EDIT_VARIANT`, or `PI_STRICT_EDIT_MODE` explicitly opts into hashline. diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 364bb94ac..b0b8d0321 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -310,6 +310,8 @@ export function getExtraHelpText(): string { PERPLEXITY_API_KEY - Perplexity web search API key (optional; anonymous fallback) PERPLEXITY_COOKIES - Perplexity web search (session cookie) TAVILY_API_KEY - Tavily web search + TINYFISH_API_KEY - TinyFish web search + FIRECRAWL_API_KEY - Firecrawl web search ANTHROPIC_SEARCH_API_KEY - Anthropic web search (override; isolates search from main ANTHROPIC_API_KEY) ANTHROPIC_SEARCH_BASE_URL - Anthropic web search base URL (override; pairs with ANTHROPIC_SEARCH_API_KEY) diff --git a/packages/coding-agent/src/cli/web-search-cli.ts b/packages/coding-agent/src/cli/web-search-cli.ts index b7ba9b0ba..a4c6f8cc2 100644 --- a/packages/coding-agent/src/cli/web-search-cli.ts +++ b/packages/coding-agent/src/cli/web-search-cli.ts @@ -120,7 +120,7 @@ ${chalk.bold("Arguments:")} ${chalk.bold("Options:")} --provider Provider: ${PROVIDERS.join(", ")} - --recency Recency filter (Brave/Perplexity): ${RECENCY_OPTIONS.join(", ")} + --recency Recency filter (when supported): ${RECENCY_OPTIONS.join(", ")} -l, --limit Max results to return --compact Render condensed output -h, --help Show this help diff --git a/packages/coding-agent/src/web/search/provider.ts b/packages/coding-agent/src/web/search/provider.ts index b74b6a00a..ee1e40674 100644 --- a/packages/coding-agent/src/web/search/provider.ts +++ b/packages/coding-agent/src/web/search/provider.ts @@ -24,66 +24,81 @@ interface ProviderMeta { /** Lazy factories. Each `load()` dynamic-imports its provider module on first call. */ const PROVIDER_META: Record = { - exa: { - id: "exa", - label: SEARCH_PROVIDER_LABELS.exa, - load: async () => new (await import("./providers/exa")).ExaProvider(), - }, - brave: { - id: "brave", - label: SEARCH_PROVIDER_LABELS.brave, - load: async () => new (await import("./providers/brave")).BraveProvider(), - }, - jina: { - id: "jina", - label: SEARCH_PROVIDER_LABELS.jina, - load: async () => new (await import("./providers/jina")).JinaProvider(), - }, perplexity: { id: "perplexity", label: SEARCH_PROVIDER_LABELS.perplexity, load: async () => new (await import("./providers/perplexity")).PerplexityProvider(), }, - kimi: { - id: "kimi", - label: SEARCH_PROVIDER_LABELS.kimi, - load: async () => new (await import("./providers/kimi")).KimiProvider(), - }, - zai: { - id: "zai", - label: SEARCH_PROVIDER_LABELS.zai, - load: async () => new (await import("./providers/zai")).ZaiProvider(), - }, - anthropic: { - id: "anthropic", - label: SEARCH_PROVIDER_LABELS.anthropic, - load: async () => new (await import("./providers/anthropic")).AnthropicProvider(), - }, gemini: { id: "gemini", label: SEARCH_PROVIDER_LABELS.gemini, load: async () => new (await import("./providers/gemini")).GeminiProvider(), }, + anthropic: { + id: "anthropic", + label: SEARCH_PROVIDER_LABELS.anthropic, + load: async () => new (await import("./providers/anthropic")).AnthropicProvider(), + }, codex: { id: "codex", label: SEARCH_PROVIDER_LABELS.codex, load: async () => new (await import("./providers/codex")).CodexProvider(), }, + xai: { + id: "xai", + label: SEARCH_PROVIDER_LABELS.xai, + load: async () => new (await import("./providers/xai")).XAIProvider(), + }, + zai: { + id: "zai", + label: SEARCH_PROVIDER_LABELS.zai, + load: async () => new (await import("./providers/zai")).ZaiProvider(), + }, + exa: { + id: "exa", + label: SEARCH_PROVIDER_LABELS.exa, + load: async () => new (await import("./providers/exa")).ExaProvider(), + }, + tinyfish: { + id: "tinyfish", + label: SEARCH_PROVIDER_LABELS.tinyfish, + load: async () => new (await import("./providers/tinyfish")).TinyFishProvider(), + }, + jina: { + id: "jina", + label: SEARCH_PROVIDER_LABELS.jina, + load: async () => new (await import("./providers/jina")).JinaProvider(), + }, + kagi: { + id: "kagi", + label: SEARCH_PROVIDER_LABELS.kagi, + load: async () => new (await import("./providers/kagi")).KagiProvider(), + }, tavily: { id: "tavily", label: SEARCH_PROVIDER_LABELS.tavily, load: async () => new (await import("./providers/tavily")).TavilyProvider(), }, + firecrawl: { + id: "firecrawl", + label: SEARCH_PROVIDER_LABELS.firecrawl, + load: async () => new (await import("./providers/firecrawl")).FirecrawlProvider(), + }, + brave: { + id: "brave", + label: SEARCH_PROVIDER_LABELS.brave, + load: async () => new (await import("./providers/brave")).BraveProvider(), + }, + kimi: { + id: "kimi", + label: SEARCH_PROVIDER_LABELS.kimi, + load: async () => new (await import("./providers/kimi")).KimiProvider(), + }, parallel: { id: "parallel", label: SEARCH_PROVIDER_LABELS.parallel, load: async () => new (await import("./providers/parallel")).ParallelProvider(), }, - kagi: { - id: "kagi", - label: SEARCH_PROVIDER_LABELS.kagi, - load: async () => new (await import("./providers/kagi")).KagiProvider(), - }, synthetic: { id: "synthetic", label: SEARCH_PROVIDER_LABELS.synthetic, @@ -94,6 +109,11 @@ const PROVIDER_META: Record = { label: SEARCH_PROVIDER_LABELS.searxng, load: async () => new (await import("./providers/searxng")).SearXNGProvider(), }, + duckduckgo: { + id: "duckduckgo", + label: SEARCH_PROVIDER_LABELS.duckduckgo, + load: async () => new (await import("./providers/duckduckgo")).DuckDuckGoProvider(), + }, }; const instanceCache = new Map(); diff --git a/packages/coding-agent/src/web/search/providers/duckduckgo.ts b/packages/coding-agent/src/web/search/providers/duckduckgo.ts new file mode 100644 index 000000000..88b2b7afc --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/duckduckgo.ts @@ -0,0 +1,140 @@ +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const DUCKDUCKGO_SEARCH_URL = "https://api.duckduckgo.com/"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; + +interface DuckDuckGoTopic { + FirstURL?: string | null; + Text?: string | null; + Topics?: DuckDuckGoTopic[] | null; +} + +interface DuckDuckGoResponse { + AbstractText?: string | null; + AbstractURL?: string | null; + AbstractSource?: string | null; + Answer?: string | null; + Definition?: string | null; + Heading?: string | null; + Results?: DuckDuckGoTopic[] | null; + RelatedTopics?: DuckDuckGoTopic[] | null; +} + +function cleanText(value: string | null | undefined): string | undefined { + const cleaned = value + ?.replace(/<[^>]*>/g, " ") + .replace(/ /gi, " ") + .replace(/&/gi, "&") + .replace(/</gi, "<") + .replace(/>/gi, ">") + .replace(/"/gi, '"') + .replace(/'/gi, "'") + .replace(/\s+/g, " ") + .trim(); + return cleaned ? cleaned : undefined; +} + +function addSource(sources: SearchSource[], source: SearchSource): void { + if (!source.url || sources.some(existing => existing.url === source.url)) return; + sources.push(source); +} + +function addTopicSource(sources: SearchSource[], topic: DuckDuckGoTopic): void { + const url = topic.FirstURL?.trim(); + if (!url) return; + const text = cleanText(topic.Text); + addSource(sources, { + title: text ?? url, + url, + snippet: text, + }); +} + +function collectTopicSources(sources: SearchSource[], topics: readonly DuckDuckGoTopic[] | null | undefined): void { + if (!topics) return; + for (const topic of topics) { + addTopicSource(sources, topic); + collectTopicSources(sources, topic.Topics); + } +} + +async function callDuckDuckGoSearch(params: SearchParams): Promise { + const queryString = [ + ["q", params.query], + ["format", "json"], + ["no_redirect", "1"], + ["no_html", "1"], + ["skip_disambig", "1"], + ["t", "oh-my-pi"], + ] + .map(([key, value]) => `${encodeURIComponent(key)}=${encodeURIComponent(value)}`) + .join("&"); + const response = await (params.fetch ?? fetch)(`${DUCKDUCKGO_SEARCH_URL}?${queryString}`, { + method: "GET", + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("duckduckgo", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "duckduckgo", + `DuckDuckGo API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as DuckDuckGoResponse; +} + +/** Execute DuckDuckGo Instant Answer API search. */ +export async function searchDuckDuckGo(params: SearchParams): Promise { + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const data = await callDuckDuckGoSearch(params); + const answer = cleanText(data.AbstractText) ?? cleanText(data.Answer) ?? cleanText(data.Definition); + const sources: SearchSource[] = []; + + const abstractUrl = data.AbstractURL?.trim(); + if (abstractUrl) { + addSource(sources, { + title: cleanText(data.AbstractSource) ?? cleanText(data.Heading) ?? abstractUrl, + url: abstractUrl, + snippet: cleanText(data.AbstractText), + }); + } + + collectTopicSources(sources, data.Results); + collectTopicSources(sources, data.RelatedTopics); + + return { + provider: "duckduckgo", + answer, + sources: sources.slice(0, numResults), + }; +} + +/** Search provider for DuckDuckGo Instant Answer API. */ +export class DuckDuckGoProvider extends SearchProvider { + readonly id = "duckduckgo"; + readonly label = "DuckDuckGo"; + + isAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchDuckDuckGo(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/firecrawl.ts b/packages/coding-agent/src/web/search/providers/firecrawl.ts new file mode 100644 index 000000000..3d7a02916 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/firecrawl.ts @@ -0,0 +1,144 @@ +/** + * Firecrawl Web Search Provider + * + * Calls Firecrawl's search API and maps web results into the unified + * SearchResponse shape used by the web search tool. + */ +import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const FIRECRAWL_SEARCH_URL = "https://api.firecrawl.dev/v2/search"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 100; + +const RECENCY_TBS: Record, string> = { + day: "qdr:d", + week: "qdr:w", + month: "qdr:m", + year: "qdr:y", +}; + +export interface FirecrawlSearchParams { + query: string; + num_results?: number; + recency?: SearchParams["recency"]; + signal?: AbortSignal; + fetch?: FetchImpl; +} + +interface FirecrawlWebResult { + title?: string | null; + url?: string | null; + description?: string | null; + markdown?: string | null; +} + +interface FirecrawlSearchResponse { + id?: string | null; + data?: { + web?: FirecrawlWebResult[] | null; + } | null; +} + +/** Resolve Firecrawl API key through the shared auth storage pipeline. */ +export function findApiKey( + authStorage: AuthStorage, + sessionId?: string, + signal?: AbortSignal, +): Promise { + return authStorage.getApiKey("firecrawl", sessionId, { signal }); +} + +function buildRequestBody(params: FirecrawlSearchParams): Record { + const body: Record = { + query: params.query, + limit: clampNumResults(params.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS), + sources: [{ type: "web" }], + }; + if (params.recency) { + body.tbs = RECENCY_TBS[params.recency]; + } + return body; +} + +async function callFirecrawlSearch(apiKey: string, params: FirecrawlSearchParams): Promise { + const response = await (params.fetch ?? fetch)(FIRECRAWL_SEARCH_URL, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${apiKey}`, + }, + body: JSON.stringify(buildRequestBody(params)), + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("firecrawl", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "firecrawl", + `Firecrawl API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as FirecrawlSearchResponse; +} + +/** Execute Firecrawl web search. */ +export async function searchFirecrawl(params: SearchParams): Promise { + const firecrawlParams: FirecrawlSearchParams = { + query: params.query, + num_results: params.numSearchResults ?? params.limit, + recency: params.recency, + signal: params.signal, + fetch: params.fetch, + }; + const keyOrResolver: ApiKey = params.authStorage.resolver("firecrawl", { + sessionId: params.sessionId, + }); + const numResults = clampNumResults(firecrawlParams.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + + const data = await withAuth(keyOrResolver, key => callFirecrawlSearch(key, firecrawlParams), { + signal: params.signal, + missingKeyMessage: + 'Firecrawl credentials not found. Set FIRECRAWL_API_KEY or configure an API key for provider "firecrawl".', + }); + const sources: SearchSource[] = []; + + for (const result of data.data?.web ?? []) { + if (!result.url) continue; + sources.push({ + title: result.title ?? result.url, + url: result.url, + snippet: result.description ?? result.markdown ?? undefined, + }); + } + + return { + provider: "firecrawl", + sources: sources.slice(0, numResults), + requestId: data.id ?? undefined, + authMode: "api_key", + }; +} + +/** Search provider for Firecrawl web search. */ +export class FirecrawlProvider extends SearchProvider { + readonly id = "firecrawl"; + readonly label = "Firecrawl"; + + isAvailable(authStorage: AuthStorage): boolean { + return authStorage.hasAuth("firecrawl") || !!getEnvApiKey("firecrawl"); + } + + search(params: SearchParams): Promise { + return searchFirecrawl(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/tinyfish.ts b/packages/coding-agent/src/web/search/providers/tinyfish.ts new file mode 100644 index 000000000..06d343b84 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/tinyfish.ts @@ -0,0 +1,134 @@ +/** + * TinyFish Web Search Provider + * + * Calls TinyFish's search API and maps results into the unified + * SearchResponse shape used by the web search tool. + */ +import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const TINYFISH_SEARCH_URL = "https://api.search.tinyfish.ai"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; + +const RECENCY_MINUTES: Record, number> = { + day: 1440, + week: 10080, + month: 43200, + year: 525600, +}; + +export interface TinyFishSearchParams { + query: string; + num_results?: number; + recency?: SearchParams["recency"]; + signal?: AbortSignal; + fetch?: FetchImpl; +} + +interface TinyFishSearchResult { + title?: string | null; + url?: string | null; + snippet?: string | null; + site_name?: string | null; +} + +interface TinyFishSearchResponse { + results?: TinyFishSearchResult[] | null; +} + +/** Resolve TinyFish API key through the shared auth storage pipeline. */ +export function findApiKey( + authStorage: AuthStorage, + sessionId?: string, + signal?: AbortSignal, +): Promise { + return authStorage.getApiKey("tinyfish", sessionId, { signal }); +} + +async function callTinyFishSearch(apiKey: string, params: TinyFishSearchParams): Promise { + const url = new URL(TINYFISH_SEARCH_URL); + url.searchParams.set("query", params.query); + if (params.recency) { + url.searchParams.set("recency_minutes", String(RECENCY_MINUTES[params.recency])); + } + + const response = await (params.fetch ?? fetch)(url, { + method: "GET", + headers: { + Accept: "application/json", + "X-API-Key": apiKey, + }, + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("tinyfish", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "tinyfish", + `TinyFish API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as TinyFishSearchResponse; +} + +/** Execute TinyFish web search. */ +export async function searchTinyFish(params: SearchParams): Promise { + const tinyFishParams: TinyFishSearchParams = { + query: params.query, + num_results: params.numSearchResults ?? params.limit, + recency: params.recency, + signal: params.signal, + fetch: params.fetch, + }; + const keyOrResolver: ApiKey = params.authStorage.resolver("tinyfish", { + sessionId: params.sessionId, + }); + const numResults = clampNumResults(tinyFishParams.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + + const data = await withAuth(keyOrResolver, key => callTinyFishSearch(key, tinyFishParams), { + signal: params.signal, + missingKeyMessage: + 'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".', + }); + const sources: SearchSource[] = []; + + for (const result of data.results ?? []) { + if (!result.url) continue; + sources.push({ + title: result.title ?? result.site_name ?? result.url, + url: result.url, + snippet: result.snippet ?? undefined, + author: result.site_name ?? undefined, + }); + } + + return { + provider: "tinyfish", + sources: sources.slice(0, numResults), + authMode: "api_key", + }; +} + +/** Search provider for TinyFish web search. */ +export class TinyFishProvider extends SearchProvider { + readonly id = "tinyfish"; + readonly label = "TinyFish"; + + isAvailable(authStorage: AuthStorage): boolean { + return authStorage.hasAuth("tinyfish") || !!getEnvApiKey("tinyfish"); + } + + search(params: SearchParams): Promise { + return searchTinyFish(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/xai.ts b/packages/coding-agent/src/web/search/providers/xai.ts new file mode 100644 index 000000000..d5c047130 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/xai.ts @@ -0,0 +1,226 @@ +import { type ApiKey, type AuthStorage, withAuth } from "@oh-my-pi/pi-ai"; +import type { SearchCitation, SearchResponse, SearchSource, SearchUsage } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const XAI_RESPONSES_URL = "https://api.x.ai/v1/responses"; +const XAI_WEB_SEARCH_MODEL = "grok-4.3"; + +interface XAIUrlCitationAnnotation { + type?: string; + url?: string | null; + title?: string | null; + text?: string | null; + cited_text?: string | null; +} + +interface XAIResponseContentPart { + type?: string; + text?: string | null; + output_text?: string | null; + annotations?: XAIUrlCitationAnnotation[] | null; +} + +interface XAIResponseOutputItem { + content?: XAIResponseContentPart[] | null; + annotations?: XAIUrlCitationAnnotation[] | null; +} + +interface XAIResponsesUsage { + input_tokens?: number; + output_tokens?: number; + total_tokens?: number; + inputTokens?: number; + outputTokens?: number; + totalTokens?: number; +} + +interface XAIResponsesResponse { + id?: string; + model?: string; + output_text?: string | null; + output?: XAIResponseOutputItem[] | null; + annotations?: XAIUrlCitationAnnotation[] | null; + citations?: string[] | null; + usage?: XAIResponsesUsage | null; +} + +function buildRequestBody(params: SearchParams): Record { + const body: Record = { + model: XAI_WEB_SEARCH_MODEL, + input: [ + { role: "system", content: params.systemPrompt }, + { role: "user", content: params.query }, + ], + tools: [{ type: "web_search" }], + }; + + if (params.maxOutputTokens !== undefined) { + body.max_output_tokens = params.maxOutputTokens; + } + if (params.temperature !== undefined) { + body.temperature = params.temperature; + } + + return body; +} + +async function callXAIResponses(apiKey: string, params: SearchParams): Promise { + const response = await (params.fetch ?? fetch)(XAI_RESPONSES_URL, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${apiKey}`, + }, + body: JSON.stringify(buildRequestBody(params)), + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("xai", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "xai", + `xAI Responses API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as XAIResponsesResponse; +} + +function addCitationSource( + sources: SearchSource[], + citations: SearchCitation[], + seenUrls: Set, + url: string, + title?: string | null, + citedText?: string | null, +): void { + const trimmedUrl = url.trim(); + if (!trimmedUrl || seenUrls.has(trimmedUrl)) return; + seenUrls.add(trimmedUrl); + const sourceTitle = title?.trim() || trimmedUrl; + const sourceSnippet = citedText?.trim() || undefined; + + sources.push({ + title: sourceTitle, + url: trimmedUrl, + snippet: sourceSnippet, + }); + citations.push({ + title: sourceTitle, + url: trimmedUrl, + citedText: sourceSnippet, + }); +} + +function collectAnnotationSources( + annotations: readonly XAIUrlCitationAnnotation[] | null | undefined, + sources: SearchSource[], + citations: SearchCitation[], + seenUrls: Set, +): void { + if (!annotations) return; + for (const annotation of annotations) { + if (annotation.type !== "url_citation" || !annotation.url) continue; + addCitationSource( + sources, + citations, + seenUrls, + annotation.url, + annotation.title, + annotation.cited_text ?? annotation.text, + ); + } +} + +function parseAnswer(response: XAIResponsesResponse): string | undefined { + const topLevelText = response.output_text?.trim(); + if (topLevelText) return topLevelText; + + const answerParts: string[] = []; + for (const item of response.output ?? []) { + for (const part of item.content ?? []) { + const text = part.output_text ?? part.text; + if ((part.type === "output_text" || part.type === "text") && text?.trim()) { + answerParts.push(text.trim()); + } + } + } + + const answer = answerParts.join("\n").trim(); + return answer ? answer : undefined; +} + +function parseUsage(usage: XAIResponsesUsage | null | undefined): SearchUsage | undefined { + if (!usage) return undefined; + const parsed: SearchUsage = {}; + const inputTokens = usage.input_tokens ?? usage.inputTokens; + const outputTokens = usage.output_tokens ?? usage.outputTokens; + const totalTokens = usage.total_tokens ?? usage.totalTokens; + + if (typeof inputTokens === "number") parsed.inputTokens = inputTokens; + if (typeof outputTokens === "number") parsed.outputTokens = outputTokens; + if (typeof totalTokens === "number") parsed.totalTokens = totalTokens; + + return Object.keys(parsed).length > 0 ? parsed : undefined; +} + +function parseResponse(response: XAIResponsesResponse): SearchResponse { + const sources: SearchSource[] = []; + const citations: SearchCitation[] = []; + const seenUrls = new Set(); + + collectAnnotationSources(response.annotations, sources, citations, seenUrls); + for (const item of response.output ?? []) { + collectAnnotationSources(item.annotations, sources, citations, seenUrls); + for (const part of item.content ?? []) { + collectAnnotationSources(part.annotations, sources, citations, seenUrls); + } + } + for (const url of response.citations ?? []) { + addCitationSource(sources, citations, seenUrls, url); + } + + return { + provider: "xai", + answer: parseAnswer(response), + sources, + citations: citations.length > 0 ? citations : undefined, + usage: parseUsage(response.usage), + model: response.model, + requestId: response.id, + authMode: "api_key", + }; +} + +/** Execute xAI Responses API web search. */ +export async function searchXAI(params: SearchParams): Promise { + const keyOrResolver: ApiKey = params.authStorage.resolver("xai", { + sessionId: params.sessionId, + }); + + const response = await withAuth(keyOrResolver, (key: string) => callXAIResponses(key, params), { + signal: params.signal, + missingKeyMessage: 'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".', + }); + return parseResponse(response); +} + +/** Search provider for xAI web search. */ +export class XAIProvider extends SearchProvider { + readonly id = "xai"; + readonly label = "xAI"; + + isAvailable(authStorage: AuthStorage): boolean { + return authStorage.hasAuth("xai"); + } + + search(params: SearchParams): Promise { + return searchXAI(params); + } +} diff --git a/packages/coding-agent/src/web/search/types.ts b/packages/coding-agent/src/web/search/types.ts index ab15f2ab5..d2654211c 100644 --- a/packages/coding-agent/src/web/search/types.ts +++ b/packages/coding-agent/src/web/search/types.ts @@ -30,16 +30,20 @@ export const SEARCH_PROVIDER_OPTIONS = [ label: "OpenAI", description: "OpenAI's native web_search (uses ChatGPT OAuth via /login openai-codex)", }, + { value: "xai", label: "xAI", description: "Grok web search via xAI Responses API (requires XAI_API_KEY)" }, { value: "zai", label: "Z.AI", description: "Calls Z.AI webSearchPrime MCP" }, { value: "exa", label: "Exa", description: "Uses Exa API when EXA_API_KEY is set; falls back to Exa MCP" }, + { value: "tinyfish", label: "TinyFish", description: "Requires TINYFISH_API_KEY" }, { value: "jina", label: "Jina", description: "Requires JINA_API_KEY" }, { value: "kagi", label: "Kagi", description: "Requires KAGI_API_KEY and Kagi Search API beta access" }, { value: "tavily", label: "Tavily", description: "Requires TAVILY_API_KEY" }, + { value: "firecrawl", label: "Firecrawl", description: "Requires FIRECRAWL_API_KEY" }, { value: "brave", label: "Brave", description: "Requires BRAVE_API_KEY" }, { value: "kimi", label: "Kimi", description: "Requires MOONSHOT_SEARCH_API_KEY or MOONSHOT_API_KEY" }, { value: "parallel", label: "Parallel", description: "Requires PARALLEL_API_KEY" }, { value: "synthetic", label: "Synthetic", description: "Requires SYNTHETIC_API_KEY" }, { value: "searxng", label: "SearXNG", description: "Requires SEARXNG_ENDPOINT or searxng.endpoint" }, + { value: "duckduckgo", label: "DuckDuckGo", description: "Uses DuckDuckGo Instant Answer API (no API key)" }, ] as const; /** Supported web search providers (every option except `auto`). */ @@ -81,7 +85,7 @@ export interface SearchSource { author?: string; } -/** Citation with text reference (anthropic, perplexity) */ +/** Citation with text reference (LLM-mediated providers) */ export interface SearchCitation { url: string; title: string; @@ -101,7 +105,7 @@ export interface SearchUsage { /** Unified response across providers */ export interface SearchResponse { provider: SearchProviderId | "none"; - /** Synthesized answer text (anthropic, perplexity) */ + /** Synthesized answer text (LLM-mediated providers) */ answer?: string; /** Search result sources */ sources: SearchSource[]; diff --git a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts new file mode 100644 index 000000000..e99861b27 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts @@ -0,0 +1,188 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchDuckDuckGo } from "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("DuckDuckGo must not request API keys"); + }, + resolver() { + throw new Error("DuckDuckGo must not request credential resolvers"); + }, + hasAuth() { + throw new Error("DuckDuckGo search must not check auth"); + }, +} as unknown as AuthStorage; + +function makeParams(query: string, fetch: FetchImpl) { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "DuckDuckGo test prompt", + fetch, + } as const; +} + +describe("DuckDuckGo web search provider", () => { + it("calls the official Instant Answer API with unauthenticated JSON query params", async () => { + let capturedUrl: string | null = null; + let capturedInit: RequestInit | undefined; + const fetchMock: FetchImpl = (input, init) => { + capturedUrl = typeof input === "string" ? input : input.toString(); + capturedInit = init; + return Promise.resolve( + new Response(JSON.stringify({ AbstractText: "Duck answer", Results: [] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }), + ); + }; + + await searchDuckDuckGo(makeParams("instant answer", fetchMock)); + + expect(capturedUrl).not.toBeNull(); + const url = new URL(capturedUrl ?? ""); + expect(`${url.origin}${url.pathname}`).toBe("https://api.duckduckgo.com/"); + expect(url.searchParams.get("q")).toBe("instant answer"); + expect(url.searchParams.get("format")).toBe("json"); + expect(url.searchParams.get("no_redirect")).toBe("1"); + expect(url.searchParams.get("no_html")).toBe("1"); + expect(url.searchParams.get("skip_disambig")).toBe("1"); + expect(url.searchParams.get("t")).toBe("oh-my-pi"); + expect(capturedInit?.method).toBe("GET"); + expect(capturedInit?.headers).toBeUndefined(); + }); + + it("uses AbstractText as the answer and flattens abstract, result, and nested related topics within the local limit", async () => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response( + JSON.stringify({ + AbstractText: " DuckDuckGo abstract & answer ", + AbstractURL: " https://example.com/abstract ", + AbstractSource: " Example Abstract Source ", + Heading: "Example Heading", + Results: [ + { + FirstURL: "https://example.com/result", + Text: "Result snippet", + }, + ], + RelatedTopics: [ + { + FirstURL: "https://example.com/related", + Text: "Related topic", + }, + { + Topics: [ + { + FirstURL: "https://example.com/nested", + Text: "Nested related topic", + }, + ], + }, + { + FirstURL: "https://example.com/omitted-by-limit", + Text: "Should be omitted by local limit", + }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ), + ); + + const response = await searchDuckDuckGo({ ...makeParams("duck mapping", fetchMock), numSearchResults: 4 }); + + expect(response).toMatchObject({ + provider: "duckduckgo", + answer: "DuckDuckGo abstract & answer", + sources: [ + { + title: "Example Abstract Source", + url: "https://example.com/abstract", + snippet: "DuckDuckGo abstract & answer", + }, + { + title: "Result snippet", + url: "https://example.com/result", + snippet: "Result snippet", + }, + { + title: "Related topic", + url: "https://example.com/related", + snippet: "Related topic", + }, + { + title: "Nested related topic", + url: "https://example.com/nested", + snippet: "Nested related topic", + }, + ], + }); + expect(response.sources).toHaveLength(4); + expect(response.sources.some(source => source.url === "https://example.com/omitted-by-limit")).toBe(false); + }); + + it("clamps oversized local result limits to DuckDuckGo's provider maximum", async () => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response( + JSON.stringify({ + RelatedTopics: Array.from({ length: 25 }, (_value, index) => ({ + FirstURL: `https://example.com/topic-${index}`, + Text: `Topic ${index}`, + })), + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ), + ); + + const response = await searchDuckDuckGo({ ...makeParams("duck clamp", fetchMock), numSearchResults: 999 }); + + expect(response.sources).toHaveLength(20); + expect(response.sources.at(0)?.url).toBe("https://example.com/topic-0"); + expect(response.sources.at(-1)?.url).toBe("https://example.com/topic-19"); + expect(response.sources.some(source => source.url === "https://example.com/topic-20")).toBe(false); + }); + + it.each([ + ["Answer", { Answer: " Direct answer " }, "Direct answer"], + ["Definition", { Definition: " Definition answer " }, "Definition answer"], + ] as const)("falls back to %s when AbstractText is absent", async (_field, payload, expectedAnswer) => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response(JSON.stringify(payload), { + status: 200, + headers: { "Content-Type": "application/json" }, + }), + ); + + const response = await searchDuckDuckGo(makeParams("fallback answer", fetchMock)); + expect(response).toMatchObject({ + provider: "duckduckgo", + answer: expectedAnswer, + }); + }); + + it("throws a provider-tagged SearchProviderError for HTTP failures", async () => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response("upstream unavailable", { + status: 503, + }), + ); + + try { + await searchDuckDuckGo(makeParams("http failure", fetchMock)); + expect.unreachable("DuckDuckGo HTTP failure should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "duckduckgo", + status: 503, + message: "DuckDuckGo API error (503): upstream unavailable", + }); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-firecrawl.test.ts b/packages/coding-agent/test/tools/web-search-firecrawl.test.ts new file mode 100644 index 000000000..d4b2b34a8 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-firecrawl.test.ts @@ -0,0 +1,138 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchFirecrawl } from "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const TEST_KEY = "test-firecrawl-key"; + +function makeAuthStorage(apiKey: string | undefined): AuthStorage { + return { + resolver(provider: string, options?: { sessionId?: string }) { + expect(provider).toBe("firecrawl"); + expect(options?.sessionId).toBe("session-firecrawl-test"); + return async () => apiKey; + }, + hasAuth(provider: string) { + return provider === "firecrawl" && Boolean(apiKey); + }, + } as unknown as AuthStorage; +} + +function makeParams(query: string, authStorage: AuthStorage = makeAuthStorage(TEST_KEY)) { + return { + query, + authStorage, + systemPrompt: "Firecrawl test prompt", + sessionId: "session-firecrawl-test", + } as const; +} + +function getHeader(headers: RequestInit["headers"] | undefined, name: string): string | null { + if (!headers) return null; + if (headers instanceof Headers) return headers.get(name); + if (Array.isArray(headers)) { + return headers.find(([key]) => key.toLowerCase() === name.toLowerCase())?.[1] ?? null; + } + const record = headers as Record; + return record[name] ?? record[name.toLowerCase()] ?? null; +} + +describe("Firecrawl web search provider", () => { + it("sends the Firecrawl POST request and maps web results", async () => { + const captured: { url?: string; init?: RequestInit; body?: unknown } = {}; + + const fetchMock: FetchImpl = async (input, init) => { + captured.url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + captured.init = init; + captured.body = JSON.parse(String(init?.body ?? "null")) as unknown; + return new Response( + JSON.stringify({ + id: "firecrawl-request-123", + data: { + web: [ + { + title: "Firecrawl result one", + url: "https://example.com/one", + description: "Description snippet", + markdown: "Ignored markdown", + }, + { + title: "Firecrawl result two", + url: "https://example.com/two", + description: null, + markdown: "Markdown fallback snippet", + }, + ], + }, + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + }; + + const response = await searchFirecrawl({ + ...makeParams("firecrawl query"), + numSearchResults: 2, + recency: "month", + fetch: fetchMock, + }); + + expect(captured.url).toBe("https://api.firecrawl.dev/v2/search"); + expect(captured.init?.method).toBe("POST"); + expect(getHeader(captured.init?.headers, "Authorization")).toBe(`Bearer ${TEST_KEY}`); + expect(getHeader(captured.init?.headers, "Content-Type")).toBe("application/json"); + expect(captured.body).toEqual({ + query: "firecrawl query", + limit: 2, + sources: [{ type: "web" }], + tbs: "qdr:m", + }); + expect(response).toEqual({ + provider: "firecrawl", + sources: [ + { + title: "Firecrawl result one", + url: "https://example.com/one", + snippet: "Description snippet", + }, + { + title: "Firecrawl result two", + url: "https://example.com/two", + snippet: "Markdown fallback snippet", + }, + ], + requestId: "firecrawl-request-123", + authMode: "api_key", + }); + }); + + it.each([ + [401, "firecrawl: 401 unauthorized"], + [402, "firecrawl: 402 credits exhausted"], + ] as const)("maps HTTP %d to a SearchProviderError", async (status, message) => { + const fetchMock: FetchImpl = async () => new Response("upstream rejected", { status }); + + try { + await searchFirecrawl({ ...makeParams("bad auth"), fetch: fetchMock }); + expect.unreachable("expected searchFirecrawl to throw"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "firecrawl", status, message }); + } + }); + + it("throws a clear error when Firecrawl credentials are missing", async () => { + const fetchMock: FetchImpl = async () => { + throw new Error("fetch should not be called without credentials"); + }; + + try { + await searchFirecrawl({ ...makeParams("missing creds", makeAuthStorage(undefined)), fetch: fetchMock }); + expect.unreachable("expected searchFirecrawl to throw"); + } catch (error) { + expect(error).toBeInstanceOf(Error); + expect((error as Error).message).toBe( + 'Firecrawl credentials not found. Set FIRECRAWL_API_KEY or configure an API key for provider "firecrawl".', + ); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-tinyfish.test.ts b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts new file mode 100644 index 000000000..c8b735e62 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts @@ -0,0 +1,127 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchTinyFish } from "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const TEST_KEY = "test-tinyfish-key"; + +function makeAuthStorage(apiKey: string | undefined): AuthStorage { + return { + resolver(provider: string, options?: { sessionId?: string }) { + expect(provider).toBe("tinyfish"); + expect(options?.sessionId).toBe("session-tinyfish-test"); + return async () => apiKey; + }, + hasAuth(provider: string) { + return provider === "tinyfish" && Boolean(apiKey); + }, + } as unknown as AuthStorage; +} + +function makeParams(query: string, authStorage: AuthStorage = makeAuthStorage(TEST_KEY)) { + return { + query, + authStorage, + systemPrompt: "TinyFish test prompt", + sessionId: "session-tinyfish-test", + } as const; +} + +function getHeader(headers: RequestInit["headers"] | undefined, name: string): string | null { + if (!headers) return null; + if (headers instanceof Headers) return headers.get(name); + if (Array.isArray(headers)) { + return headers.find(([key]) => key.toLowerCase() === name.toLowerCase())?.[1] ?? null; + } + const record = headers as Record; + return record[name] ?? record[name.toLowerCase()] ?? null; +} + +describe("TinyFish web search provider", () => { + it("sends the TinyFish GET request and locally clamps results", async () => { + const captured: { url?: URL; init?: RequestInit } = {}; + + const fetchMock: FetchImpl = async (input, init) => { + captured.url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.init = init; + return new Response( + JSON.stringify({ + results: [ + { + title: "TinyFish result one", + url: "https://example.com/one", + snippet: "First snippet", + site_name: "Example Site", + }, + { + title: "TinyFish result two", + url: "https://example.com/two", + snippet: "Second snippet", + }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + }; + + const response = await searchTinyFish({ + ...makeParams("fresh fish"), + numSearchResults: 1, + recency: "week", + fetch: fetchMock, + }); + + const capturedUrl = captured.url; + if (!capturedUrl) throw new Error("TinyFish request was not captured"); + const endpoint = `${capturedUrl.origin}${capturedUrl.pathname === "/" ? "" : capturedUrl.pathname}`; + expect(endpoint).toBe("https://api.search.tinyfish.ai"); + expect(captured.init?.method ?? "GET").toBe("GET"); + expect(getHeader(captured.init?.headers, "X-API-Key")).toBe(TEST_KEY); + expect(capturedUrl.searchParams.get("query")).toBe("fresh fish"); + expect(capturedUrl.searchParams.get("recency_minutes")).toBe("10080"); + expect([...capturedUrl.searchParams.keys()].sort()).toEqual(["query", "recency_minutes"]); + expect(response).toEqual({ + provider: "tinyfish", + sources: [ + { + title: "TinyFish result one", + url: "https://example.com/one", + snippet: "First snippet", + author: "Example Site", + }, + ], + authMode: "api_key", + }); + }); + + it.each([ + [401, "tinyfish: 401 unauthorized"], + [402, "tinyfish: 402 credits exhausted"], + ] as const)("maps HTTP %d to a SearchProviderError", async (status, message) => { + const fetchMock: FetchImpl = async () => new Response("upstream rejected", { status }); + + try { + await searchTinyFish({ ...makeParams("bad auth"), fetch: fetchMock }); + expect.unreachable("expected searchTinyFish to throw"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "tinyfish", status, message }); + } + }); + + it("throws a clear error when TinyFish credentials are missing", async () => { + const fetchMock: FetchImpl = async () => { + throw new Error("fetch should not be called without credentials"); + }; + + try { + await searchTinyFish({ ...makeParams("missing creds", makeAuthStorage(undefined)), fetch: fetchMock }); + expect.unreachable("expected searchTinyFish to throw"); + } catch (error) { + expect(error).toBeInstanceOf(Error); + expect((error as Error).message).toBe( + 'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".', + ); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-xai.test.ts b/packages/coding-agent/test/tools/web-search-xai.test.ts new file mode 100644 index 000000000..4b0db7dd3 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-xai.test.ts @@ -0,0 +1,259 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchXAI } from "@oh-my-pi/pi-coding-agent/web/search/providers/xai"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +type CapturedRequest = { + url: string; + method: string | undefined; + headers: RequestInit["headers"]; + body: Record | null; +}; + +function makeAuthStorage(apiKey: string | undefined) { + return { + resolver(provider: string, options?: { sessionId?: string }) { + expect(provider).toBe("xai"); + expect(options?.sessionId).toBe("session-xai-test"); + return async () => apiKey; + }, + hasAuth(provider: string) { + return provider === "xai" && Boolean(apiKey); + }, + } as unknown as AuthStorage; +} + +function makeParams(fetch: FetchImpl, authStorage: AuthStorage = makeAuthStorage("test-xai-key")) { + return { + query: "latest xAI web search", + systemPrompt: "Use web search for current xAI facts.", + authStorage, + fetch, + sessionId: "session-xai-test", + } as const; +} + +function captureFetch(responseBody: Record, status = 200) { + let capturedRequest: CapturedRequest | null = null; + const fetchMock: FetchImpl = (input, init) => { + capturedRequest = { + url: typeof input === "string" ? input : input.toString(), + method: init?.method, + headers: init?.headers, + body: init?.body ? (JSON.parse(String(init.body)) as Record) : null, + }; + return Promise.resolve( + new Response(JSON.stringify(responseBody), { + status, + headers: { "Content-Type": "application/json" }, + }), + ); + }; + return { + fetchMock, + get capturedRequest() { + return capturedRequest; + }, + }; +} + +describe("xAI web search provider", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("POSTs the Responses API with bearer auth and xAI web_search tool payload", async () => { + const capture = captureFetch({ id: "resp_request", model: "grok-4.3", output_text: "xAI answer" }); + + await searchXAI({ + ...makeParams(capture.fetchMock), + maxOutputTokens: 512, + temperature: 0.2, + }); + + expect(capture.capturedRequest).not.toBeNull(); + expect(capture.capturedRequest?.url).toBe("https://api.x.ai/v1/responses"); + expect(capture.capturedRequest?.method).toBe("POST"); + expect(capture.capturedRequest?.headers).toMatchObject({ + "Content-Type": "application/json", + Authorization: "Bearer test-xai-key", + }); + expect(capture.capturedRequest?.body).toMatchObject({ + model: "grok-4.3", + input: [ + { role: "system", content: "Use web search for current xAI facts." }, + { role: "user", content: "latest xAI web search" }, + ], + tools: [{ type: "web_search" }], + max_output_tokens: 512, + temperature: 0.2, + }); + }); + + it("maps output_text, URL citation annotations, top-level citations, id, model, usage, and auth mode", async () => { + const capture = captureFetch({ + id: "resp_xai_123", + model: "grok-4.3", + output_text: "Top-level xAI answer", + annotations: [ + { + type: "url_citation", + url: "https://example.com/top-annotation", + title: "Top Annotation", + text: "Top annotation text", + }, + ], + output: [ + { + type: "message", + annotations: [ + { + type: "url_citation", + url: "https://example.com/item-annotation", + title: "Item Annotation", + cited_text: "Item annotation text", + }, + ], + content: [ + { + type: "output_text", + text: "Ignored because output_text wins", + annotations: [ + { + type: "url_citation", + url: "https://example.com/annotated", + title: "Annotated Source", + cited_text: "Annotated cited text", + }, + ], + }, + ], + }, + ], + citations: ["https://example.com/top-level-citation"], + usage: { + input_tokens: 12, + output_tokens: 8, + total_tokens: 20, + }, + }); + + const response = await searchXAI(makeParams(capture.fetchMock)); + + expect(response).toMatchObject({ + provider: "xai", + answer: "Top-level xAI answer", + requestId: "resp_xai_123", + model: "grok-4.3", + authMode: "api_key", + usage: { + inputTokens: 12, + outputTokens: 8, + totalTokens: 20, + }, + sources: [ + { + title: "Top Annotation", + url: "https://example.com/top-annotation", + snippet: "Top annotation text", + }, + { + title: "Item Annotation", + url: "https://example.com/item-annotation", + snippet: "Item annotation text", + }, + { + title: "Annotated Source", + url: "https://example.com/annotated", + snippet: "Annotated cited text", + }, + { + title: "https://example.com/top-level-citation", + url: "https://example.com/top-level-citation", + }, + ], + citations: [ + { + title: "Top Annotation", + url: "https://example.com/top-annotation", + citedText: "Top annotation text", + }, + { + title: "Item Annotation", + url: "https://example.com/item-annotation", + citedText: "Item annotation text", + }, + { + title: "Annotated Source", + url: "https://example.com/annotated", + citedText: "Annotated cited text", + }, + { + title: "https://example.com/top-level-citation", + url: "https://example.com/top-level-citation", + }, + ], + }); + }); + + it("falls back to output content parts when output_text is absent", async () => { + const capture = captureFetch({ + id: "resp_content_parts", + model: "grok-4.3", + output: [ + { + content: [ + { type: "output_text", text: "First content part" }, + { type: "text", output_text: "Second content part" }, + ], + }, + ], + }); + + const response = await searchXAI(makeParams(capture.fetchMock)); + expect(response).toMatchObject({ + answer: "First content part\nSecond content part", + }); + }); + + it.each([ + [401, "xai: 401 unauthorized"], + [402, "xai: 402 credits exhausted"], + ] as const)("maps HTTP %s failures to SearchProviderError", async (status, message) => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response(JSON.stringify({ error: "request failed" }), { + status, + headers: { "Content-Type": "application/json" }, + }), + ); + + try { + await searchXAI(makeParams(fetchMock)); + expect.unreachable(`xAI HTTP ${status} failure should reject`); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "xai", + status, + message, + }); + } + }); + + it("throws a clear missing-key error before fetch when credentials are unavailable", async () => { + const fetchMock = vi.fn(() => Promise.resolve(new Response("{}", { status: 200 }))) as unknown as FetchImpl; + + try { + await searchXAI(makeParams(fetchMock, makeAuthStorage(undefined))); + expect.unreachable("missing xAI credentials should reject"); + } catch (error) { + expect(error).toBeInstanceOf(Error); + expect(error).toHaveProperty( + "message", + 'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".', + ); + } + expect(fetchMock).not.toHaveBeenCalled(); + }); +}); From 1b530c6a2d563952f90bd1880c9584bfba600789 Mon Sep 17 00:00:00 2001 From: zekdevs <38579990+zekdevs@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:23:24 -0600 Subject: [PATCH 2/6] move to new xai api and fix tinyfish review comment --- docs/tools/web_search.md | 14 +-- .../src/web/search/providers/xai.ts | 30 +++-- .../test/tools/web-search-tinyfish.test.ts | 104 ++++++++++++------ .../test/tools/web-search-xai.test.ts | 73 +++++++++++- 4 files changed, 165 insertions(+), 56 deletions(-) diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 602824544..ab65837a6 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -38,11 +38,11 @@ | Field | Type | Required | Description | | --- | --- | --- | --- | | `query` | `string` | Yes | Search query, passed to providers unchanged. | -| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. | -| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. | +| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. | +| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. xAI ignores it because the documented Agent Tools `web_search` API does not expose result-count controls. | | `max_tokens` | `number` | No | Passed through as provider token caps (`maxOutputTokens`, `max_tokens`, or xAI `max_output_tokens`) only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | | `temperature` | `number` | No | Passed through only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | -| `num_search_results` | `number` | No | Requested upstream search breadth. Most providers use this as returned source count. Perplexity keeps it distinct from `limit`; xAI does not send a source-count parameter to Responses API. | +| `num_search_results` | `number` | No | Requested upstream search breadth. Most providers use this as returned source count. Perplexity keeps it distinct from `limit`; xAI ignores it because the documented Agent Tools `web_search` API does not expose result-count controls. | ## Outputs The tool returns a single text content block plus structured `details`. @@ -127,8 +127,8 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Output may include `answer`, `sources`, `usage`, `model`, `requestId`. If the streamed response has no `url_citation` annotations, the adapter falls back to scraping markdown links and bare URLs from the answer text. - **xAI** — `packages/coding-agent/src/web/search/providers/xai.ts` - Availability: `XAI_API_KEY` or `agent.db` credential for `xai`. - - Querying: POST `https://api.x.ai/v1/responses` with model `grok-4.3` and `tools: [{ type: "web_search" }]`. - - `max_tokens` and `temperature` pass through; `limit`, `num_search_results`, and `recency` are not sent. + - Querying: POST `https://api.x.ai/v1/responses` with model `grok-4.3` and `tools: [{ type: "web_search" }]` using the `/v1/responses` Agent Tools API. + - `max_tokens` and `temperature` pass through. `limit`, `num_search_results`, and `recency` are ignored because the documented Agent Tools `web_search` API does not expose equivalent result-count or date controls. - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode: "api_key"`. - **Z.AI** — `packages/coding-agent/src/web/search/providers/zai.ts` - Availability: env or `agent.db` credential for `zai`. @@ -246,8 +246,8 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec ## Notes - The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. - `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. -- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI currently ignores both. -- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; the model-facing prompt does not name specific providers. +- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI ignores both because the documented Agent Tools `web_search` API does not expose result-count controls. +- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. The model-facing prompt does not name specific providers. - `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain. - DuckDuckGo is intentionally last in the auto chain because it is always available without credentials. - Exa uses `authStorage.getApiKey("exa")`, then `EXA_API_KEY`, then unauthenticated `https://mcp.exa.ai/mcp` fallback. diff --git a/packages/coding-agent/src/web/search/providers/xai.ts b/packages/coding-agent/src/web/search/providers/xai.ts index d5c047130..4447df4ea 100644 --- a/packages/coding-agent/src/web/search/providers/xai.ts +++ b/packages/coding-agent/src/web/search/providers/xai.ts @@ -67,26 +67,34 @@ function buildRequestBody(params: SearchParams): Record { return body; } -async function callXAIResponses(apiKey: string, params: SearchParams): Promise { - const response = await (params.fetch ?? fetch)(XAI_RESPONSES_URL, { +async function postXAIResponses( + apiKey: string, + params: SearchParams, + body: Record, +): Promise { + return (params.fetch ?? fetch)(XAI_RESPONSES_URL, { method: "POST", headers: { "Content-Type": "application/json", Authorization: `Bearer ${apiKey}`, }, - body: JSON.stringify(buildRequestBody(params)), + body: JSON.stringify(body), signal: withHardTimeout(params.signal), }); +} + +function throwXAIResponsesError(status: number, errorText: string): never { + const classified = classifyProviderHttpError("xai", status, errorText); + if (classified) throw classified; + throw new SearchProviderError("xai", `xAI Responses API error (${status}): ${errorText}`, status); +} + +async function callXAIResponses(apiKey: string, params: SearchParams): Promise { + const requestBody = buildRequestBody(params); + const response = await postXAIResponses(apiKey, params, requestBody); if (!response.ok) { - const errorText = await response.text(); - const classified = classifyProviderHttpError("xai", response.status, errorText); - if (classified) throw classified; - throw new SearchProviderError( - "xai", - `xAI Responses API error (${response.status}): ${errorText}`, - response.status, - ); + throwXAIResponsesError(response.status, await response.text()); } return (await response.json()) as XAIResponsesResponse; diff --git a/packages/coding-agent/test/tools/web-search-tinyfish.test.ts b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts index c8b735e62..6ac8f9e7e 100644 --- a/packages/coding-agent/test/tools/web-search-tinyfish.test.ts +++ b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts @@ -4,6 +4,7 @@ import { searchTinyFish } from "@oh-my-pi/pi-coding-agent/web/search/providers/t import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; const TEST_KEY = "test-tinyfish-key"; +const UNSUPPORTED_TINYFISH_COUNT_PARAMS = ["limit", "num_results", "count", "size", "max_results"] as const; function makeAuthStorage(apiKey: string | undefined): AuthStorage { return { @@ -37,36 +38,35 @@ function getHeader(headers: RequestInit["headers"] | undefined, name: string): s return record[name] ?? record[name.toLowerCase()] ?? null; } +function expectOnlyDocumentedTinyFishParams(url: URL, expectedParams: readonly string[]): void { + expect([...url.searchParams.keys()].sort()).toEqual([...expectedParams].sort()); + for (const unsupportedParam of UNSUPPORTED_TINYFISH_COUNT_PARAMS) { + expect(url.searchParams.has(unsupportedParam)).toBe(false); + } +} + describe("TinyFish web search provider", () => { - it("sends the TinyFish GET request and locally clamps results", async () => { + it("documents TinyFish's absent result-count parameter and applies numSearchResults locally", async () => { const captured: { url?: URL; init?: RequestInit } = {}; + const upstreamResults = Array.from({ length: 13 }, (_, index) => ({ + title: `TinyFish result ${index}`, + url: `https://example.com/${index}`, + snippet: `Snippet ${index}`, + site_name: index === 0 ? "Example Site" : undefined, + })); const fetchMock: FetchImpl = async (input, init) => { captured.url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); captured.init = init; - return new Response( - JSON.stringify({ - results: [ - { - title: "TinyFish result one", - url: "https://example.com/one", - snippet: "First snippet", - site_name: "Example Site", - }, - { - title: "TinyFish result two", - url: "https://example.com/two", - snippet: "Second snippet", - }, - ], - }), - { status: 200, headers: { "Content-Type": "application/json" } }, - ); + return new Response(JSON.stringify({ results: upstreamResults }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); }; const response = await searchTinyFish({ ...makeParams("fresh fish"), - numSearchResults: 1, + numSearchResults: 12, recency: "week", fetch: fetchMock, }); @@ -79,19 +79,59 @@ describe("TinyFish web search provider", () => { expect(getHeader(captured.init?.headers, "X-API-Key")).toBe(TEST_KEY); expect(capturedUrl.searchParams.get("query")).toBe("fresh fish"); expect(capturedUrl.searchParams.get("recency_minutes")).toBe("10080"); - expect([...capturedUrl.searchParams.keys()].sort()).toEqual(["query", "recency_minutes"]); - expect(response).toEqual({ - provider: "tinyfish", - sources: [ - { - title: "TinyFish result one", - url: "https://example.com/one", - snippet: "First snippet", - author: "Example Site", - }, - ], - authMode: "api_key", + + // TinyFish Search docs expose no result-count parameter; unified counts are applied after the response. + expectOnlyDocumentedTinyFishParams(capturedUrl, ["query", "recency_minutes"]); + + expect(response.provider).toBe("tinyfish"); + expect(response.authMode).toBe("api_key"); + expect(response.sources).toHaveLength(12); + expect(response.sources[0]).toEqual({ + title: "TinyFish result 0", + url: "https://example.com/0", + snippet: "Snippet 0", + author: "Example Site", }); + expect(response.sources.at(-1)).toEqual({ + title: "TinyFish result 11", + url: "https://example.com/11", + snippet: "Snippet 11", + author: undefined, + }); + expect(response.sources.some(source => source.url === "https://example.com/12")).toBe(false); + }); + + it("does not serialize unsupported count-like params for the unified limit option", async () => { + const captured: { url?: URL } = {}; + const upstreamResults = Array.from({ length: 12 }, (_, index) => ({ + title: `TinyFish limit result ${index}`, + url: `https://example.com/limit-${index}`, + snippet: `Limit snippet ${index}`, + })); + + const fetchMock: FetchImpl = async input => { + captured.url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + return new Response(JSON.stringify({ results: upstreamResults }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + const response = await searchTinyFish({ + ...makeParams("limit fish"), + limit: 11, + fetch: fetchMock, + }); + + const capturedUrl = captured.url; + if (!capturedUrl) throw new Error("TinyFish request was not captured"); + expect(capturedUrl.searchParams.get("query")).toBe("limit fish"); + // The unified limit option must not invent an upstream TinyFish count parameter. + expectOnlyDocumentedTinyFishParams(capturedUrl, ["query"]); + + expect(response.sources).toHaveLength(11); + expect(response.sources.at(-1)?.url).toBe("https://example.com/limit-10"); + expect(response.sources.some(source => source.url === "https://example.com/limit-11")).toBe(false); }); it.each([ diff --git a/packages/coding-agent/test/tools/web-search-xai.test.ts b/packages/coding-agent/test/tools/web-search-xai.test.ts index 4b0db7dd3..8c76896d5 100644 --- a/packages/coding-agent/test/tools/web-search-xai.test.ts +++ b/packages/coding-agent/test/tools/web-search-xai.test.ts @@ -33,17 +33,17 @@ function makeParams(fetch: FetchImpl, authStorage: AuthStorage = makeAuthStorage } as const; } -function captureFetch(responseBody: Record, status = 200) { - let capturedRequest: CapturedRequest | null = null; +function captureFetch(responseBody: Record | string, status = 200) { + const capturedRequests: CapturedRequest[] = []; const fetchMock: FetchImpl = (input, init) => { - capturedRequest = { + capturedRequests.push({ url: typeof input === "string" ? input : input.toString(), method: init?.method, headers: init?.headers, body: init?.body ? (JSON.parse(String(init.body)) as Record) : null, - }; + }); return Promise.resolve( - new Response(JSON.stringify(responseBody), { + new Response(typeof responseBody === "string" ? responseBody : JSON.stringify(responseBody), { status, headers: { "Content-Type": "application/json" }, }), @@ -51,8 +51,9 @@ function captureFetch(responseBody: Record, status = 200) { }; return { fetchMock, + capturedRequests, get capturedRequest() { - return capturedRequest; + return capturedRequests.at(-1) ?? null; }, }; } @@ -88,6 +89,66 @@ describe("xAI web search provider", () => { max_output_tokens: 512, temperature: 0.2, }); + expect(capture.capturedRequest?.body?.tools).toEqual([{ type: "web_search" }]); + expect(capture.capturedRequest?.body).not.toHaveProperty("search_parameters"); + }); + + it("omits search_parameters for minimal web_search requests", async () => { + const capture = captureFetch({ id: "resp_minimal", model: "grok-4.3", output_text: "minimal xAI answer" }); + + await searchXAI(makeParams(capture.fetchMock)); + + expect(capture.capturedRequest).not.toBeNull(); + const body = capture.capturedRequest?.body; + expect(body?.tools).toEqual([{ type: "web_search" }]); + expect(body).not.toHaveProperty("search_parameters"); + }); + + it.each([ + ["limit", { limit: 6 }], + ["numSearchResults", { numSearchResults: 7 }], + ["limit and numSearchResults", { limit: 2, numSearchResults: 50 }], + ["recency", { recency: "week" }], + ["limit, numSearchResults, and recency", { limit: 0, numSearchResults: 30, recency: "day" }], + ] as const)("does not map %s to xAI search_parameters", async (_caseName, searchParams) => { + const capture = captureFetch({ id: "resp_agent_tools", model: "grok-4.3", output_text: "xAI answer" }); + + await searchXAI({ + ...makeParams(capture.fetchMock), + ...searchParams, + }); + + expect(capture.capturedRequest).not.toBeNull(); + const body = capture.capturedRequest?.body; + expect(body?.tools).toEqual([{ type: "web_search" }]); + expect(body).not.toHaveProperty("search_parameters"); + expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]); + }); + + it("rejects deprecated live-search 410 responses without retrying", async () => { + const capture = captureFetch("Live search is deprecated. Please use the Agent Tools API.", 410); + + try { + await searchXAI({ + ...makeParams(capture.fetchMock), + limit: 2, + numSearchResults: 5, + recency: "week", + }); + expect.unreachable("xAI HTTP 410 deprecation failure should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "xai", + status: 410, + message: "xAI Responses API error (410): Live search is deprecated. Please use the Agent Tools API.", + }); + } + + expect(capture.capturedRequests).toHaveLength(1); + const body = capture.capturedRequests[0]?.body; + expect(body?.tools).toEqual([{ type: "web_search" }]); + expect(body).not.toHaveProperty("search_parameters"); }); it("maps output_text, URL citation annotations, top-level citations, id, model, usage, and auth mode", async () => { From 3106f76d11bdb8e7f2cd6f66e95a8c89c0998c7d Mon Sep 17 00:00:00 2001 From: zekdevs <38579990+zekdevs@users.noreply.github.com> Date: Fri, 26 Jun 2026 09:54:32 -0600 Subject: [PATCH 3/6] cap xai response locally as upstream api has no limit support --- docs/tools/web_search.md | 8 +- .../src/web/search/providers/xai.ts | 22 +++- .../test/tools/web-search-xai.test.ts | 112 ++++++++++++++++++ 3 files changed, 134 insertions(+), 8 deletions(-) diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index ab65837a6..22dcb1672 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -39,10 +39,10 @@ | --- | --- | --- | --- | | `query` | `string` | Yes | Search query, passed to providers unchanged. | | `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. | -| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. xAI ignores it because the documented Agent Tools `web_search` API does not expose result-count controls. | +| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. xAI does not send it upstream; it caps returned sources/citations after parsing unless `num_search_results` is provided. | | `max_tokens` | `number` | No | Passed through as provider token caps (`maxOutputTokens`, `max_tokens`, or xAI `max_output_tokens`) only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | | `temperature` | `number` | No | Passed through only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | -| `num_search_results` | `number` | No | Requested upstream search breadth. Most providers use this as returned source count. Perplexity keeps it distinct from `limit`; xAI ignores it because the documented Agent Tools `web_search` API does not expose result-count controls. | +| `num_search_results` | `number` | No | Requested search breadth or local result cap. Most providers send it upstream. xAI does not send it upstream; it caps returned sources/citations after parsing and takes precedence over `limit`. | ## Outputs The tool returns a single text content block plus structured `details`. @@ -128,7 +128,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - **xAI** — `packages/coding-agent/src/web/search/providers/xai.ts` - Availability: `XAI_API_KEY` or `agent.db` credential for `xai`. - Querying: POST `https://api.x.ai/v1/responses` with model `grok-4.3` and `tools: [{ type: "web_search" }]` using the `/v1/responses` Agent Tools API. - - `max_tokens` and `temperature` pass through. `limit`, `num_search_results`, and `recency` are ignored because the documented Agent Tools `web_search` API does not expose equivalent result-count or date controls. + - `max_tokens` and `temperature` pass through. `recency` is ignored because the documented Agent Tools `web_search` API does not expose date controls. `limit` and `num_search_results` are not sent upstream; because xAI citations may include every encountered URL, the adapter locally caps returned `sources` and `citations` after parsing (`numSearchResults ?? limit`). Missing cap returns all sources/citations. - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode: "api_key"`. - **Z.AI** — `packages/coding-agent/src/web/search/providers/zai.ts` - Availability: env or `agent.db` credential for `zai`. @@ -246,7 +246,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec ## Notes - The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. - `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. -- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI ignores both because the documented Agent Tools `web_search` API does not expose result-count controls. +- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI applies that same precedence locally after parsing to cap returned sources/citations without sending a result-count control upstream. - `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. The model-facing prompt does not name specific providers. - `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain. - DuckDuckGo is intentionally last in the auto chain because it is always available without credentials. diff --git a/packages/coding-agent/src/web/search/providers/xai.ts b/packages/coding-agent/src/web/search/providers/xai.ts index 4447df4ea..89156c8d0 100644 --- a/packages/coding-agent/src/web/search/providers/xai.ts +++ b/packages/coding-agent/src/web/search/providers/xai.ts @@ -178,7 +178,20 @@ function parseUsage(usage: XAIResponsesUsage | null | undefined): SearchUsage | return Object.keys(parsed).length > 0 ? parsed : undefined; } -function parseResponse(response: XAIResponsesResponse): SearchResponse { +function applyResultCap( + sources: SearchSource[], + citations: SearchCitation[], + requestedCap: number | undefined, +): { sources: SearchSource[]; citations: SearchCitation[] } { + if (requestedCap === undefined) return { sources, citations }; + + return { + sources: sources.slice(0, requestedCap), + citations: citations.slice(0, requestedCap), + }; +} + +function parseResponse(response: XAIResponsesResponse, requestedCap?: number): SearchResponse { const sources: SearchSource[] = []; const citations: SearchCitation[] = []; const seenUrls = new Set(); @@ -193,12 +206,13 @@ function parseResponse(response: XAIResponsesResponse): SearchResponse { for (const url of response.citations ?? []) { addCitationSource(sources, citations, seenUrls, url); } + const limited = applyResultCap(sources, citations, requestedCap); return { provider: "xai", answer: parseAnswer(response), - sources, - citations: citations.length > 0 ? citations : undefined, + sources: limited.sources, + citations: limited.citations.length > 0 ? limited.citations : undefined, usage: parseUsage(response.usage), model: response.model, requestId: response.id, @@ -216,7 +230,7 @@ export async function searchXAI(params: SearchParams): Promise { signal: params.signal, missingKeyMessage: 'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".', }); - return parseResponse(response); + return parseResponse(response, params.numSearchResults ?? params.limit); } /** Search provider for xAI web search. */ diff --git a/packages/coding-agent/test/tools/web-search-xai.test.ts b/packages/coding-agent/test/tools/web-search-xai.test.ts index 8c76896d5..bbc01c19f 100644 --- a/packages/coding-agent/test/tools/web-search-xai.test.ts +++ b/packages/coding-agent/test/tools/web-search-xai.test.ts @@ -257,6 +257,118 @@ describe("xAI web search provider", () => { }); }); + it("caps parsed sources and citations locally without changing Agent Tools request shape", async () => { + const capture = captureFetch({ + id: "resp_local_cap", + model: "grok-4.3", + output_text: "Capped xAI answer", + annotations: [ + { + type: "url_citation", + url: "https://example.com/annotation-1", + title: "Annotation 1", + text: "Annotation 1 text", + }, + ], + output: [ + { + annotations: [ + { + type: "url_citation", + url: "https://example.com/annotation-2", + title: "Annotation 2", + cited_text: "Annotation 2 text", + }, + ], + content: [ + { + type: "output_text", + text: "Ignored because output_text wins", + annotations: [ + { + type: "url_citation", + url: "https://example.com/annotation-3", + title: "Annotation 3", + cited_text: "Annotation 3 text", + }, + ], + }, + ], + }, + ], + citations: ["https://example.com/top-level-4", "https://example.com/top-level-5"], + }); + + const response = await searchXAI({ + ...makeParams(capture.fetchMock), + limit: 4, + }); + + expect(response.sources).toHaveLength(4); + expect(response.citations).toHaveLength(4); + expect(response.sources.map(source => source.url)).toEqual([ + "https://example.com/annotation-1", + "https://example.com/annotation-2", + "https://example.com/annotation-3", + "https://example.com/top-level-4", + ]); + expect(response.citations?.map(citation => citation.url)).toEqual([ + "https://example.com/annotation-1", + "https://example.com/annotation-2", + "https://example.com/annotation-3", + "https://example.com/top-level-4", + ]); + expect(capture.capturedRequest).not.toBeNull(); + const body = capture.capturedRequest?.body; + expect(body?.tools).toEqual([{ type: "web_search" }]); + expect(body).not.toHaveProperty("search_parameters"); + expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]); + }); + + it("uses numSearchResults before limit for the local xAI output cap", async () => { + const capture = captureFetch({ + id: "resp_num_search_results_cap", + model: "grok-4.3", + output_text: "numSearchResults capped xAI answer", + annotations: [ + { + type: "url_citation", + url: "https://example.com/precedence-1", + title: "Precedence 1", + }, + ], + citations: [ + "https://example.com/precedence-2", + "https://example.com/precedence-3", + "https://example.com/precedence-4", + ], + }); + + const response = await searchXAI({ + ...makeParams(capture.fetchMock), + limit: 1, + numSearchResults: 3, + }); + expect(response.sources).toHaveLength(3); + expect(response.citations).toHaveLength(3); + + expect(response.sources.map(source => source.url)).toEqual([ + "https://example.com/precedence-1", + "https://example.com/precedence-2", + "https://example.com/precedence-3", + ]); + expect(response.citations?.map(citation => citation.url)).toEqual([ + "https://example.com/precedence-1", + "https://example.com/precedence-2", + "https://example.com/precedence-3", + ]); + expect(capture.capturedRequest).not.toBeNull(); + const body = capture.capturedRequest?.body; + expect(body?.tools).toEqual([{ type: "web_search" }]); + expect(body).not.toHaveProperty("search_parameters"); + expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]); + }); + it("falls back to output content parts when output_text is absent", async () => { const capture = captureFetch({ id: "resp_content_parts", From 2023b45d2c6e47b76ee2bea098e3f3a827c98c1b Mon Sep 17 00:00:00 2001 From: zekdevs <38579990+zekdevs@users.noreply.github.com> Date: Fri, 26 Jun 2026 10:11:31 -0600 Subject: [PATCH 4/6] add cap to xai sources --- docs/tools/web_search.md | 9 ++-- .../src/web/search/providers/xai.ts | 18 ++++--- .../test/tools/web-search-xai.test.ts | 53 +++++++++++++++++++ 3 files changed, 68 insertions(+), 12 deletions(-) diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 22dcb1672..7a33cc757 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -39,10 +39,10 @@ | --- | --- | --- | --- | | `query` | `string` | Yes | Search query, passed to providers unchanged. | | `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. | -| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. xAI does not send it upstream; it caps returned sources/citations after parsing unless `num_search_results` is provided. | +| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. For xAI, it is not sent upstream; it is a local post-parse cap on returned sources/citations, defaulting to `10` when omitted/invalid/zero and capped at `30`. | | `max_tokens` | `number` | No | Passed through as provider token caps (`maxOutputTokens`, `max_tokens`, or xAI `max_output_tokens`) only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | | `temperature` | `number` | No | Passed through only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | -| `num_search_results` | `number` | No | Requested search breadth or local result cap. Most providers send it upstream. xAI does not send it upstream; it caps returned sources/citations after parsing and takes precedence over `limit`. | +| `num_search_results` | `number` | No | Requested search breadth or local result cap. Most providers send it upstream. xAI does not send it upstream; it is a local cap on returned sources/citations after parsing, takes precedence over `limit`, defaults to `10` when omitted/invalid/zero, and is capped at `30`. | ## Outputs The tool returns a single text content block plus structured `details`. @@ -128,7 +128,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - **xAI** — `packages/coding-agent/src/web/search/providers/xai.ts` - Availability: `XAI_API_KEY` or `agent.db` credential for `xai`. - Querying: POST `https://api.x.ai/v1/responses` with model `grok-4.3` and `tools: [{ type: "web_search" }]` using the `/v1/responses` Agent Tools API. - - `max_tokens` and `temperature` pass through. `recency` is ignored because the documented Agent Tools `web_search` API does not expose date controls. `limit` and `num_search_results` are not sent upstream; because xAI citations may include every encountered URL, the adapter locally caps returned `sources` and `citations` after parsing (`numSearchResults ?? limit`). Missing cap returns all sources/citations. + - `max_tokens` and `temperature` pass through. `recency` is ignored because the documented Agent Tools `web_search` API does not expose date controls. `limit` and `num_search_results` are not sent upstream; because xAI citations may include every encountered URL, the adapter locally caps returned `sources` and `citations` after parsing. The local cap uses `num_search_results` before `limit`, defaults to `10` when omitted/invalid/zero, and is capped at `30`. - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode: "api_key"`. - **Z.AI** — `packages/coding-agent/src/web/search/providers/zai.ts` - Availability: env or `agent.db` credential for `zai`. @@ -226,6 +226,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Parallel result count: default `10`, max `40`; per-result excerpt cap `10_000` chars (`packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts`). - Kagi result count: default `10`, max `40` (`packages/coding-agent/src/web/search/providers/kagi.ts`). - SearXNG result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/searxng.ts`). +- xAI local sources/citations cap: `num_search_results` before `limit`, omitted/invalid/zero => default `10`, max `30`; neither value is sent upstream (`packages/coding-agent/src/web/search/providers/xai.ts`). - Perplexity API-key mode defaults: `max_tokens = 8192`, `temperature = 0.2`, `num_search_results = 20` (`packages/coding-agent/src/web/search/providers/perplexity.ts`). - Anthropic defaults: model `claude-haiku-4-5`, `DEFAULT_MAX_TOKENS = 4096` when the provider omits `max_tokens` (`packages/coding-agent/src/web/search/providers/anthropic.ts`). - Gemini retries: up to `3` retries per endpoint, base delay `1000` ms, rate-limit delay budget `5 * 60 * 1000` ms (`packages/coding-agent/src/web/search/providers/gemini.ts`). @@ -246,7 +247,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec ## Notes - The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. - `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. -- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI applies that same precedence locally after parsing to cap returned sources/citations without sending a result-count control upstream. +- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI applies that same precedence locally after parsing to cap returned sources/citations without sending a result-count control upstream (`10` default, `30` max). - `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. The model-facing prompt does not name specific providers. - `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain. - DuckDuckGo is intentionally last in the auto chain because it is always available without credentials. diff --git a/packages/coding-agent/src/web/search/providers/xai.ts b/packages/coding-agent/src/web/search/providers/xai.ts index 89156c8d0..aafa53c56 100644 --- a/packages/coding-agent/src/web/search/providers/xai.ts +++ b/packages/coding-agent/src/web/search/providers/xai.ts @@ -1,12 +1,15 @@ import { type ApiKey, type AuthStorage, withAuth } from "@oh-my-pi/pi-ai"; import type { SearchCitation, SearchResponse, SearchSource, SearchUsage } from "../../../web/search/types"; import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; import type { SearchParams } from "./base"; import { SearchProvider } from "./base"; import { classifyProviderHttpError, withHardTimeout } from "./utils"; const XAI_RESPONSES_URL = "https://api.x.ai/v1/responses"; const XAI_WEB_SEARCH_MODEL = "grok-4.3"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 30; interface XAIUrlCitationAnnotation { type?: string; @@ -181,17 +184,15 @@ function parseUsage(usage: XAIResponsesUsage | null | undefined): SearchUsage | function applyResultCap( sources: SearchSource[], citations: SearchCitation[], - requestedCap: number | undefined, + resultCap: number, ): { sources: SearchSource[]; citations: SearchCitation[] } { - if (requestedCap === undefined) return { sources, citations }; - return { - sources: sources.slice(0, requestedCap), - citations: citations.slice(0, requestedCap), + sources: sources.slice(0, resultCap), + citations: citations.slice(0, resultCap), }; } -function parseResponse(response: XAIResponsesResponse, requestedCap?: number): SearchResponse { +function parseResponse(response: XAIResponsesResponse, resultCap: number): SearchResponse { const sources: SearchSource[] = []; const citations: SearchCitation[] = []; const seenUrls = new Set(); @@ -206,7 +207,7 @@ function parseResponse(response: XAIResponsesResponse, requestedCap?: number): S for (const url of response.citations ?? []) { addCitationSource(sources, citations, seenUrls, url); } - const limited = applyResultCap(sources, citations, requestedCap); + const limited = applyResultCap(sources, citations, resultCap); return { provider: "xai", @@ -226,11 +227,12 @@ export async function searchXAI(params: SearchParams): Promise { sessionId: params.sessionId, }); + const resultCap = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); const response = await withAuth(keyOrResolver, (key: string) => callXAIResponses(key, params), { signal: params.signal, missingKeyMessage: 'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".', }); - return parseResponse(response, params.numSearchResults ?? params.limit); + return parseResponse(response, resultCap); } /** Search provider for xAI web search. */ diff --git a/packages/coding-agent/test/tools/web-search-xai.test.ts b/packages/coding-agent/test/tools/web-search-xai.test.ts index bbc01c19f..27a6d3578 100644 --- a/packages/coding-agent/test/tools/web-search-xai.test.ts +++ b/packages/coding-agent/test/tools/web-search-xai.test.ts @@ -58,6 +58,10 @@ function captureFetch(responseBody: Record | string, status = 2 }; } +function citationUrls(prefix: string, count: number): string[] { + return Array.from({ length: count }, (_, index) => `https://example.com/${prefix}-${index + 1}`); +} + describe("xAI web search provider", () => { afterEach(() => { vi.restoreAllMocks(); @@ -257,6 +261,55 @@ describe("xAI web search provider", () => { }); }); + it("defaults xAI local cap to 10 sources and citations when no count is requested", async () => { + const urls = citationUrls("default-cap", 12); + const capture = captureFetch({ + id: "resp_default_cap", + model: "grok-4.3", + output_text: "Default capped xAI answer", + citations: urls, + }); + + const response = await searchXAI(makeParams(capture.fetchMock)); + const expectedUrls = urls.slice(0, 10); + + expect(response.sources).toHaveLength(10); + expect(response.citations).toHaveLength(10); + expect(response.sources.map(source => source.url)).toEqual(expectedUrls); + expect(response.citations?.map(citation => citation.url)).toEqual(expectedUrls); + expect(capture.capturedRequest).not.toBeNull(); + const body = capture.capturedRequest?.body; + expect(body?.tools).toEqual([{ type: "web_search" }]); + expect(body).not.toHaveProperty("search_parameters"); + expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]); + }); + + it("clamps oversized xAI local cap requests to 30 sources and citations", async () => { + const urls = citationUrls("max-cap", 35); + const capture = captureFetch({ + id: "resp_max_cap", + model: "grok-4.3", + output_text: "Max capped xAI answer", + citations: urls, + }); + + const response = await searchXAI({ + ...makeParams(capture.fetchMock), + numSearchResults: 99, + }); + const expectedUrls = urls.slice(0, 30); + + expect(response.sources).toHaveLength(30); + expect(response.citations).toHaveLength(30); + expect(response.sources.map(source => source.url)).toEqual(expectedUrls); + expect(response.citations?.map(citation => citation.url)).toEqual(expectedUrls); + expect(capture.capturedRequest).not.toBeNull(); + const body = capture.capturedRequest?.body; + expect(body?.tools).toEqual([{ type: "web_search" }]); + expect(body).not.toHaveProperty("search_parameters"); + expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]); + }); + it("caps parsed sources and citations locally without changing Agent Tools request shape", async () => { const capture = captureFetch({ id: "resp_local_cap", From 656a8e5c40abc27395fe2eca8e771d1d6c1f301f Mon Sep 17 00:00:00 2001 From: zekdevs <38579990+zekdevs@users.noreply.github.com> Date: Fri, 26 Jun 2026 10:31:50 -0600 Subject: [PATCH 5/6] add provider subpaths to legacy pi bundle --- .../plugins/legacy-pi-bundled-keys.ts | 4 ++++ .../plugins/legacy-pi-bundled-registry.ts | 12 ++++++++++++ .../legacy-pi-bundled-subpath-overrides.test.ts | 15 +++++++++++++++ 3 files changed, 31 insertions(+) diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-keys.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-keys.ts index c2fbb7779..9a2ecfc14 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-keys.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-keys.ts @@ -955,7 +955,9 @@ export const BUNDLED_PI_REGISTRY_KEYS: ReadonlySet = new Set([ "@oh-my-pi/pi-coding-agent/web/search/providers/base", "@oh-my-pi/pi-coding-agent/web/search/providers/brave", "@oh-my-pi/pi-coding-agent/web/search/providers/codex", + "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo", "@oh-my-pi/pi-coding-agent/web/search/providers/exa", + "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl", "@oh-my-pi/pi-coding-agent/web/search/providers/gemini", "@oh-my-pi/pi-coding-agent/web/search/providers/jina", "@oh-my-pi/pi-coding-agent/web/search/providers/kagi", @@ -966,7 +968,9 @@ export const BUNDLED_PI_REGISTRY_KEYS: ReadonlySet = new Set([ "@oh-my-pi/pi-coding-agent/web/search/providers/searxng", "@oh-my-pi/pi-coding-agent/web/search/providers/synthetic", "@oh-my-pi/pi-coding-agent/web/search/providers/tavily", + "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish", "@oh-my-pi/pi-coding-agent/web/search/providers/utils", + "@oh-my-pi/pi-coding-agent/web/search/providers/xai", "@oh-my-pi/pi-coding-agent/web/search/providers/zai", "@oh-my-pi/pi-natives", "@oh-my-pi/pi-tui", diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-registry.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-registry.ts index d00dcadd3..49b472a89 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-registry.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-bundled-registry.ts @@ -957,7 +957,9 @@ import * as bundledPiCodingAgentWebSearchProvidersAnthropic from "@oh-my-pi/pi-c import * as bundledPiCodingAgentWebSearchProvidersBase from "@oh-my-pi/pi-coding-agent/web/search/providers/base"; import * as bundledPiCodingAgentWebSearchProvidersBrave from "@oh-my-pi/pi-coding-agent/web/search/providers/brave"; import * as bundledPiCodingAgentWebSearchProvidersCodex from "@oh-my-pi/pi-coding-agent/web/search/providers/codex"; +import * as bundledPiCodingAgentWebSearchProvidersDuckduckgo from "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo"; import * as bundledPiCodingAgentWebSearchProvidersExa from "@oh-my-pi/pi-coding-agent/web/search/providers/exa"; +import * as bundledPiCodingAgentWebSearchProvidersFirecrawl from "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl"; import * as bundledPiCodingAgentWebSearchProvidersGemini from "@oh-my-pi/pi-coding-agent/web/search/providers/gemini"; import * as bundledPiCodingAgentWebSearchProvidersJina from "@oh-my-pi/pi-coding-agent/web/search/providers/jina"; import * as bundledPiCodingAgentWebSearchProvidersKagi from "@oh-my-pi/pi-coding-agent/web/search/providers/kagi"; @@ -968,7 +970,9 @@ import * as bundledPiCodingAgentWebSearchProvidersPerplexityAuth from "@oh-my-pi import * as bundledPiCodingAgentWebSearchProvidersSearxng from "@oh-my-pi/pi-coding-agent/web/search/providers/searxng"; import * as bundledPiCodingAgentWebSearchProvidersSynthetic from "@oh-my-pi/pi-coding-agent/web/search/providers/synthetic"; import * as bundledPiCodingAgentWebSearchProvidersTavily from "@oh-my-pi/pi-coding-agent/web/search/providers/tavily"; +import * as bundledPiCodingAgentWebSearchProvidersTinyfish from "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish"; import * as bundledPiCodingAgentWebSearchProvidersUtils from "@oh-my-pi/pi-coding-agent/web/search/providers/utils"; +import * as bundledPiCodingAgentWebSearchProvidersXai from "@oh-my-pi/pi-coding-agent/web/search/providers/xai"; import * as bundledPiCodingAgentWebSearchProvidersZai from "@oh-my-pi/pi-coding-agent/web/search/providers/zai"; import * as bundledPiCodingAgentWebSearchRender from "@oh-my-pi/pi-coding-agent/web/search/render"; import * as bundledPiCodingAgentWebSearchTypes from "@oh-my-pi/pi-coding-agent/web/search/types"; @@ -3299,8 +3303,12 @@ export const BUNDLED_PI_REGISTRY: Readonly>, "@oh-my-pi/pi-coding-agent/web/search/providers/codex": bundledPiCodingAgentWebSearchProvidersCodex as unknown as Readonly>, + "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo": + bundledPiCodingAgentWebSearchProvidersDuckduckgo as unknown as Readonly>, "@oh-my-pi/pi-coding-agent/web/search/providers/exa": bundledPiCodingAgentWebSearchProvidersExa as unknown as Readonly>, + "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl": + bundledPiCodingAgentWebSearchProvidersFirecrawl as unknown as Readonly>, "@oh-my-pi/pi-coding-agent/web/search/providers/gemini": bundledPiCodingAgentWebSearchProvidersGemini as unknown as Readonly>, "@oh-my-pi/pi-coding-agent/web/search/providers/jina": @@ -3321,8 +3329,12 @@ export const BUNDLED_PI_REGISTRY: Readonly>, "@oh-my-pi/pi-coding-agent/web/search/providers/tavily": bundledPiCodingAgentWebSearchProvidersTavily as unknown as Readonly>, + "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish": + bundledPiCodingAgentWebSearchProvidersTinyfish as unknown as Readonly>, "@oh-my-pi/pi-coding-agent/web/search/providers/utils": bundledPiCodingAgentWebSearchProvidersUtils as unknown as Readonly>, + "@oh-my-pi/pi-coding-agent/web/search/providers/xai": + bundledPiCodingAgentWebSearchProvidersXai as unknown as Readonly>, "@oh-my-pi/pi-coding-agent/web/search/providers/zai": bundledPiCodingAgentWebSearchProvidersZai as unknown as Readonly>, "@oh-my-pi/pi-natives": bundledPiNatives as unknown as Readonly>, diff --git a/packages/coding-agent/test/extensibility/legacy-pi-bundled-subpath-overrides.test.ts b/packages/coding-agent/test/extensibility/legacy-pi-bundled-subpath-overrides.test.ts index c1d9b3fa2..27ad9733f 100644 --- a/packages/coding-agent/test/extensibility/legacy-pi-bundled-subpath-overrides.test.ts +++ b/packages/coding-agent/test/extensibility/legacy-pi-bundled-subpath-overrides.test.ts @@ -34,6 +34,21 @@ describe("legacy pi compat compiled-mode subpath overrides (issue #3442)", () => expect(BUNDLED_PI_REGISTRY_KEYS.has("@oh-my-pi/pi-ai/oauth/openai-codex")).toBe(true); }); + it("expands web search provider wildcard exports for compiled plugin imports", () => { + const overrides = __buildLegacyPiPackageRootOverrides(true); + const providerKeys = [ + "@oh-my-pi/pi-coding-agent/web/search/providers/xai", + "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish", + "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl", + "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo", + ] as const; + + for (const key of providerKeys) { + expect(BUNDLED_PI_REGISTRY_KEYS.has(key)).toBe(true); + expect(overrides[key]).toBe(`omp-legacy-pi-bundled:${key}`); + } + }); + it("does not enumerate root catch-all wildcards (./* / ./*.js)", () => { // Root `./*` / `./*.js` patterns would static-import top-level files // like the package's own `cli.ts` and explode the bundle through the From 4939bdd591c79e2b1b7781e50bfe7981a41960bd Mon Sep 17 00:00:00 2001 From: zekdevs <38579990+zekdevs@users.noreply.github.com> Date: Fri, 26 Jun 2026 10:57:17 -0600 Subject: [PATCH 6/6] fix tinyfish results fetch --- docs/tools/web_search.md | 12 +- .../src/web/search/providers/tinyfish.ts | 72 ++++-- .../test/tools/web-search-tinyfish.test.ts | 229 ++++++++++++++---- 3 files changed, 240 insertions(+), 73 deletions(-) diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 7a33cc757..311d45afa 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -39,10 +39,10 @@ | --- | --- | --- | --- | | `query` | `string` | Yes | Search query, passed to providers unchanged. | | `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. | -| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. For xAI, it is not sent upstream; it is a local post-parse cap on returned sources/citations, defaulting to `10` when omitted/invalid/zero and capped at `30`. | +| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. TinyFish and xAI are local-cap exceptions: TinyFish uses it only for page fetching and slicing; xAI caps parsed sources/citations locally, defaulting to `10` and max `30`. | | `max_tokens` | `number` | No | Passed through as provider token caps (`maxOutputTokens`, `max_tokens`, or xAI `max_output_tokens`) only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | | `temperature` | `number` | No | Passed through only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | -| `num_search_results` | `number` | No | Requested search breadth or local result cap. Most providers send it upstream. xAI does not send it upstream; it is a local cap on returned sources/citations after parsing, takes precedence over `limit`, defaults to `10` when omitted/invalid/zero, and is capped at `30`. | +| `num_search_results` | `number` | No | Requested search breadth or local result cap. Most providers send it upstream. TinyFish and xAI do not; TinyFish clamps to `1..20` with default `10` and uses it for paginated fetches before slicing, while xAI caps parsed sources/citations locally with default `10` and max `30`. | ## Outputs The tool returns a single text content block plus structured `details`. @@ -143,8 +143,8 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Output: synthesized `answer` from up to 3 result summaries, `sources`, `requestId`. - **TinyFish** — `packages/coding-agent/src/web/search/providers/tinyfish.ts` - Availability: `TINYFISH_API_KEY` or `agent.db` credential for `tinyfish`. - - Querying: GET `https://api.search.tinyfish.ai` with `X-API-Key`; `recency` maps to `recency_minutes`. - - `limit` / `num_search_results`: collapsed and clamped to `1..20`, default `10`; output `sources`, `authMode: "api_key"`. + - Querying: GET `https://api.search.tinyfish.ai` with `X-API-Key` and `query`; `recency` maps to `recency_minutes`. + - `limit` / `num_search_results`: collapsed as `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. TinyFish has no count parameter and returns at most 10 results per page; for counts above the first page, the adapter fetches documented `page` values (`0`, then `1` when needed) before slicing locally. Output `sources`, `authMode: "api_key"`. - **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts` - Availability: `JINA_API_KEY` only. - Querying: GET-like fetch to `https://s.jina.ai/` with bearer auth. @@ -218,7 +218,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - `formatForLLM()` truncates source snippets and citation text to 240 chars (`packages/coding-agent/src/web/search/index.ts`). - `formatForLLM()` emits at most 3 search queries, each truncated to 120 chars (`packages/coding-agent/src/web/search/index.ts`). - Brave result count: default `10`, max `20` (`DEFAULT_NUM_RESULTS`, `MAX_NUM_RESULTS` in `packages/coding-agent/src/web/search/providers/brave.ts`). -- TinyFish result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/tinyfish.ts`). +- TinyFish local result count: default `10`, max `20`; the API has no count parameter and returns at most 10 results per page, so the adapter fetches documented pages (`page=0`, then `page=1` when needed) and slices locally (`packages/coding-agent/src/web/search/providers/tinyfish.ts`). - DuckDuckGo result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/duckduckgo.ts`). - Tavily result count: default `5`, max `20` (`packages/coding-agent/src/web/search/providers/tavily.ts`). - Firecrawl result count: default `10`, max `100` (`packages/coding-agent/src/web/search/providers/firecrawl.ts`). @@ -247,7 +247,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec ## Notes - The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. - `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. -- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI applies that same precedence locally after parsing to cap returned sources/citations without sending a result-count control upstream (`10` default, `30` max). +- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts. TinyFish uses that collapsed value only as a local cap and to decide whether to fetch page `1`; it does not serialize a count parameter. xAI applies that same precedence locally after parsing to cap returned sources/citations without sending a result-count control upstream (`10` default, `30` max). - `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. The model-facing prompt does not name specific providers. - `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain. - DuckDuckGo is intentionally last in the auto chain because it is always available without credentials. diff --git a/packages/coding-agent/src/web/search/providers/tinyfish.ts b/packages/coding-agent/src/web/search/providers/tinyfish.ts index 06d343b84..8f95101fa 100644 --- a/packages/coding-agent/src/web/search/providers/tinyfish.ts +++ b/packages/coding-agent/src/web/search/providers/tinyfish.ts @@ -27,6 +27,7 @@ export interface TinyFishSearchParams { query: string; num_results?: number; recency?: SearchParams["recency"]; + page?: number; signal?: AbortSignal; fetch?: FetchImpl; } @@ -39,6 +40,8 @@ interface TinyFishSearchResult { } interface TinyFishSearchResponse { + total_results?: number | null; + page?: number | null; results?: TinyFishSearchResult[] | null; } @@ -57,6 +60,9 @@ async function callTinyFishSearch(apiKey: string, params: TinyFishSearchParams): if (params.recency) { url.searchParams.set("recency_minutes", String(RECENCY_MINUTES[params.recency])); } + if (params.page !== undefined) { + url.searchParams.set("page", String(params.page)); + } const response = await (params.fetch ?? fetch)(url, { method: "GET", @@ -81,28 +87,8 @@ async function callTinyFishSearch(apiKey: string, params: TinyFishSearchParams): return (await response.json()) as TinyFishSearchResponse; } -/** Execute TinyFish web search. */ -export async function searchTinyFish(params: SearchParams): Promise { - const tinyFishParams: TinyFishSearchParams = { - query: params.query, - num_results: params.numSearchResults ?? params.limit, - recency: params.recency, - signal: params.signal, - fetch: params.fetch, - }; - const keyOrResolver: ApiKey = params.authStorage.resolver("tinyfish", { - sessionId: params.sessionId, - }); - const numResults = clampNumResults(tinyFishParams.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); - - const data = await withAuth(keyOrResolver, key => callTinyFishSearch(key, tinyFishParams), { - signal: params.signal, - missingKeyMessage: - 'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".', - }); - const sources: SearchSource[] = []; - - for (const result of data.results ?? []) { +function appendTinyFishSources(sources: SearchSource[], results: readonly TinyFishSearchResult[]): void { + for (const result of results) { if (!result.url) continue; sources.push({ title: result.title ?? result.site_name ?? result.url, @@ -111,10 +97,50 @@ export async function searchTinyFish(params: SearchParams): Promise { + const tinyFishParams: TinyFishSearchParams = { + query: params.query, + recency: params.recency, + signal: params.signal, + fetch: params.fetch, + }; + const keyOrResolver: ApiKey = params.authStorage.resolver("tinyfish", { + sessionId: params.sessionId, + }); + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + + const sources = await withAuth( + keyOrResolver, + async key => { + const collected: SearchSource[] = []; + const firstPage = await callTinyFishSearch(key, { ...tinyFishParams, page: 0 }); + const firstPageResults = firstPage.results ?? []; + appendTinyFishSources(collected, firstPageResults); + + if ( + numResults > DEFAULT_NUM_RESULTS && + collected.length < numResults && + firstPageResults.length >= DEFAULT_NUM_RESULTS + ) { + const secondPage = await callTinyFishSearch(key, { ...tinyFishParams, page: 1 }); + appendTinyFishSources(collected, secondPage.results ?? []); + } + + return collected.slice(0, numResults); + }, + { + signal: params.signal, + missingKeyMessage: + 'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".', + }, + ); return { provider: "tinyfish", - sources: sources.slice(0, numResults), + sources, authMode: "api_key", }; } diff --git a/packages/coding-agent/test/tools/web-search-tinyfish.test.ts b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts index 6ac8f9e7e..95cae5261 100644 --- a/packages/coding-agent/test/tools/web-search-tinyfish.test.ts +++ b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts @@ -38,6 +38,29 @@ function getHeader(headers: RequestInit["headers"] | undefined, name: string): s return record[name] ?? record[name.toLowerCase()] ?? null; } +interface TinyFishMockResult { + title: string; + url: string | null; + snippet: string; + site_name?: string; +} + +function tinyFishResults(prefix: string, count: number, start = 0): TinyFishMockResult[] { + return Array.from({ length: count }, (_, offset) => { + const index = start + offset; + return { + title: `${prefix} result ${index}`, + url: `https://example.com/${prefix}-${index}`, + snippet: `${prefix} snippet ${index}`, + site_name: index === 0 ? "Example Site" : undefined, + }; + }); +} + +function tinyFishPage(results: TinyFishMockResult[], page = 0, totalResults = results.length) { + return { results, total_results: totalResults, page }; +} + function expectOnlyDocumentedTinyFishParams(url: URL, expectedParams: readonly string[]): void { expect([...url.searchParams.keys()].sort()).toEqual([...expectedParams].sort()); for (const unsupportedParam of UNSUPPORTED_TINYFISH_COUNT_PARAMS) { @@ -46,19 +69,18 @@ function expectOnlyDocumentedTinyFishParams(url: URL, expectedParams: readonly s } describe("TinyFish web search provider", () => { - it("documents TinyFish's absent result-count parameter and applies numSearchResults locally", async () => { - const captured: { url?: URL; init?: RequestInit } = {}; - const upstreamResults = Array.from({ length: 13 }, (_, index) => ({ - title: `TinyFish result ${index}`, - url: `https://example.com/${index}`, - snippet: `Snippet ${index}`, - site_name: index === 0 ? "Example Site" : undefined, - })); + it("documents TinyFish's absent result-count parameter and applies numSearchResults across pages", async () => { + const captured: { url: URL; init?: RequestInit }[] = []; + const pages = new Map([ + ["0", tinyFishResults("tinyfish", 10)], + ["1", tinyFishResults("tinyfish", 10, 10)], + ]); const fetchMock: FetchImpl = async (input, init) => { - captured.url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); - captured.init = init; - return new Response(JSON.stringify({ results: upstreamResults }), { + const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.push({ url, init }); + const page = Number(url.searchParams.get("page") ?? 0); + return new Response(JSON.stringify(tinyFishPage(pages.get(String(page)) ?? [], page, 20)), { status: 200, headers: { "Content-Type": "application/json" }, }); @@ -71,47 +93,53 @@ describe("TinyFish web search provider", () => { fetch: fetchMock, }); - const capturedUrl = captured.url; - if (!capturedUrl) throw new Error("TinyFish request was not captured"); - const endpoint = `${capturedUrl.origin}${capturedUrl.pathname === "/" ? "" : capturedUrl.pathname}`; + expect(captured).toHaveLength(2); + const [firstRequest, secondRequest] = captured; + const endpoint = `${firstRequest.url.origin}${firstRequest.url.pathname === "/" ? "" : firstRequest.url.pathname}`; expect(endpoint).toBe("https://api.search.tinyfish.ai"); - expect(captured.init?.method ?? "GET").toBe("GET"); - expect(getHeader(captured.init?.headers, "X-API-Key")).toBe(TEST_KEY); - expect(capturedUrl.searchParams.get("query")).toBe("fresh fish"); - expect(capturedUrl.searchParams.get("recency_minutes")).toBe("10080"); + expect(firstRequest.init?.method ?? "GET").toBe("GET"); + expect(getHeader(firstRequest.init?.headers, "X-API-Key")).toBe(TEST_KEY); + expect(firstRequest.url.searchParams.get("query")).toBe("fresh fish"); + expect(firstRequest.url.searchParams.get("recency_minutes")).toBe("10080"); + expect(firstRequest.url.searchParams.get("page")).toBe("0"); + expect(secondRequest.url.searchParams.get("query")).toBe("fresh fish"); + expect(secondRequest.url.searchParams.get("recency_minutes")).toBe("10080"); + expect(secondRequest.url.searchParams.get("page")).toBe("1"); - // TinyFish Search docs expose no result-count parameter; unified counts are applied after the response. - expectOnlyDocumentedTinyFishParams(capturedUrl, ["query", "recency_minutes"]); + // TinyFish Search docs expose no result-count parameter; unified counts are applied after paginated responses. + expectOnlyDocumentedTinyFishParams(firstRequest.url, ["query", "recency_minutes", "page"]); + expectOnlyDocumentedTinyFishParams(secondRequest.url, ["query", "recency_minutes", "page"]); expect(response.provider).toBe("tinyfish"); expect(response.authMode).toBe("api_key"); expect(response.sources).toHaveLength(12); expect(response.sources[0]).toEqual({ - title: "TinyFish result 0", - url: "https://example.com/0", - snippet: "Snippet 0", + title: "tinyfish result 0", + url: "https://example.com/tinyfish-0", + snippet: "tinyfish snippet 0", author: "Example Site", }); expect(response.sources.at(-1)).toEqual({ - title: "TinyFish result 11", - url: "https://example.com/11", - snippet: "Snippet 11", + title: "tinyfish result 11", + url: "https://example.com/tinyfish-11", + snippet: "tinyfish snippet 11", author: undefined, }); - expect(response.sources.some(source => source.url === "https://example.com/12")).toBe(false); + expect(response.sources.some(source => source.url === "https://example.com/tinyfish-12")).toBe(false); }); - it("does not serialize unsupported count-like params for the unified limit option", async () => { - const captured: { url?: URL } = {}; - const upstreamResults = Array.from({ length: 12 }, (_, index) => ({ - title: `TinyFish limit result ${index}`, - url: `https://example.com/limit-${index}`, - snippet: `Limit snippet ${index}`, - })); + it("requests two documented TinyFish pages for limit 20 without unsupported count-like params", async () => { + const captured: URL[] = []; + const pages = new Map([ + ["0", tinyFishResults("limit", 10)], + ["1", tinyFishResults("limit", 10, 10)], + ]); const fetchMock: FetchImpl = async input => { - captured.url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); - return new Response(JSON.stringify({ results: upstreamResults }), { + const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.push(url); + const page = Number(url.searchParams.get("page") ?? 0); + return new Response(JSON.stringify(tinyFishPage(pages.get(String(page)) ?? [], page, 20)), { status: 200, headers: { "Content-Type": "application/json" }, }); @@ -119,19 +147,132 @@ describe("TinyFish web search provider", () => { const response = await searchTinyFish({ ...makeParams("limit fish"), - limit: 11, + limit: 20, + recency: "day", fetch: fetchMock, }); - const capturedUrl = captured.url; - if (!capturedUrl) throw new Error("TinyFish request was not captured"); - expect(capturedUrl.searchParams.get("query")).toBe("limit fish"); - // The unified limit option must not invent an upstream TinyFish count parameter. - expectOnlyDocumentedTinyFishParams(capturedUrl, ["query"]); + expect(captured).toHaveLength(2); + expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0", "1"]); + for (const url of captured) { + expect(url.searchParams.get("query")).toBe("limit fish"); + expect(url.searchParams.get("recency_minutes")).toBe("1440"); + expectOnlyDocumentedTinyFishParams(url, ["query", "recency_minutes", "page"]); + } + expect(response.sources).toHaveLength(20); + expect(response.sources.at(-1)?.url).toBe("https://example.com/limit-19"); + }); + + it("requests page 1 when page 0 has 10 raw results but fewer usable sources", async () => { + const captured: URL[] = []; + const firstPageResults = tinyFishResults("raw-page", 10); + firstPageResults[0] = { ...firstPageResults[0], url: null }; + const pages = new Map([ + ["0", firstPageResults], + ["1", tinyFishResults("raw-page", 10, 10)], + ]); + + const fetchMock: FetchImpl = async input => { + const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.push(url); + const page = Number(url.searchParams.get("page") ?? 0); + return new Response(JSON.stringify(tinyFishPage(pages.get(String(page)) ?? [], page, 20)), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + const response = await searchTinyFish({ ...makeParams("raw page fish"), limit: 11, fetch: fetchMock }); + + expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0", "1"]); expect(response.sources).toHaveLength(11); - expect(response.sources.at(-1)?.url).toBe("https://example.com/limit-10"); - expect(response.sources.some(source => source.url === "https://example.com/limit-11")).toBe(false); + expect(response.sources[0]?.url).toBe("https://example.com/raw-page-1"); + expect(response.sources.at(-1)?.url).toBe("https://example.com/raw-page-11"); + }); + + it("stops early for limit 20 when page 0 returns fewer than 10 raw results", async () => { + const captured: URL[] = []; + const fetchMock: FetchImpl = async input => { + const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.push(url); + return new Response(JSON.stringify(tinyFishPage(tinyFishResults("short-page", 9), 0, 9)), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + const response = await searchTinyFish({ ...makeParams("short page fish"), limit: 20, fetch: fetchMock }); + + expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0"]); + expect(response.sources).toHaveLength(9); + expect(response.sources.at(-1)?.url).toBe("https://example.com/short-page-8"); + }); + + it("does not request a second page for the default 10-result page", async () => { + const captured: URL[] = []; + const fetchMock: FetchImpl = async input => { + const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.push(url); + return new Response(JSON.stringify(tinyFishPage(tinyFishResults("default", 10), 0, 10)), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + const response = await searchTinyFish({ ...makeParams("default fish"), fetch: fetchMock }); + + expect(captured).toHaveLength(1); + expect(captured[0].searchParams.get("query")).toBe("default fish"); + expect(captured[0].searchParams.get("page")).toBe("0"); + expectOnlyDocumentedTinyFishParams(captured[0], ["query", "page"]); + expect(response.sources).toHaveLength(10); + }); + + it("does not request a second page when the local limit is 10 or below", async () => { + const captured: URL[] = []; + const fetchMock: FetchImpl = async input => { + const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.push(url); + return new Response(JSON.stringify(tinyFishPage(tinyFishResults("small-limit", 10), 0, 10)), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + const response = await searchTinyFish({ ...makeParams("small limit fish"), limit: 7, fetch: fetchMock }); + + expect(captured).toHaveLength(1); + expect(captured[0].searchParams.get("query")).toBe("small limit fish"); + expect(captured[0].searchParams.get("page")).toBe("0"); + expectOnlyDocumentedTinyFishParams(captured[0], ["query", "page"]); + expect(response.sources).toHaveLength(7); + expect(response.sources.at(-1)?.url).toBe("https://example.com/small-limit-6"); + }); + + it("propagates second-page HTTP errors", async () => { + const captured: URL[] = []; + const fetchMock: FetchImpl = async input => { + const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.push(url); + if (url.searchParams.get("page") === "1") { + return new Response("upstream rejected page 1", { status: 402 }); + } + + return new Response(JSON.stringify(tinyFishPage(tinyFishResults("page-error", 10), 0, 20)), { + status: 200, + headers: { "Content-Type": "application/json" }, + }); + }; + + try { + await searchTinyFish({ ...makeParams("page error fish"), limit: 20, fetch: fetchMock }); + expect.unreachable("expected searchTinyFish to throw"); + } catch (error) { + expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0", "1"]); + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "tinyfish", status: 402, message: "tinyfish: 402 credits exhausted" }); + } }); it.each([