Merge PR #3572: feat: add xai/ddg/firecrawl/tinyfish web_search providers (@zekdevs)

This commit is contained in:
can1357
2026-06-27 01:45:39 +02:00
19 changed files with 1996 additions and 93 deletions
+16 -12
View File
@@ -151,7 +151,7 @@ _[Watch the capture ↗](https://omp.sh/clips/collab.mp4)_
### 08 · Read a pdf on arxiv, why not?
web_search chains fourteen ranked providers and hands whatever URLs it finds straight to read. Arxiv PDFs, GitHub pages, Stack Overflow threads come back as structured markdown with anchors intact — the same tool surface you use on local files. Cite, follow, quote, never lose where you came from.
web_search chains eighteen ranked providers and hands whatever URLs it finds straight to read. Arxiv PDFs, GitHub pages, Stack Overflow threads come back as structured markdown with anchors intact — the same tool surface you use on local files. Cite, follow, quote, never lose where you came from.
![omp TUI: web_search returns 10 ranked Perplexity sources for inference-time compute scaling, the agent picks an arxiv paper, calls read https://arxiv.org/pdf/2604.10739v1, and summarizes the paper's headline result with real numbers.](https://omp.sh/clips/web-poster.webp)
@@ -309,31 +309,35 @@ Ollama `local` · Ollama Cloud · LM Studio `local` · llama.cpp `local` · vLLM
Full provider & routing reference at [omp.sh/docs/providers](https://omp.sh/docs/providers).
## Fourteen backends. _One tool the agent already knows_.
## Eighteen backends. _One tool the agent already knows_.
`web_search` is built in, not bolted on. `auto` walks a fourteen-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive.
`web_search` is built in, not bolted on. `auto` walks an eighteen-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive.
### Search providers
Fourteen backends. Pin one, or let `auto` walk the chain in order.
Eighteen backends. Pin one, or let `auto` walk the chain in order.
| provider | auth |
| ------------ | ---------------------- |
| `auto` | chain |
| `exa` | `EXA_API_KEY` (or mcp) |
| `brave` | `BRAVE_API_KEY` |
| `jina` | `JINA_API_KEY` |
| `kimi` | `MOONSHOT_API_KEY` |
| `zai` | `ZAI_API_KEY` |
| `anthropic` | oauth |
| `perplexity` | `PERPLEXITY_API_KEY` |
| `gemini` | oauth |
| `anthropic` | oauth |
| `codex` | oauth |
| `tavily` | `TAVILY_API_KEY` |
| `parallel` | `PARALLEL_API_KEY` |
| `xai` | `XAI_API_KEY` |
| `zai` | `ZAI_API_KEY` |
| `exa` | `EXA_API_KEY` (or mcp) |
| `tinyfish` | `TINYFISH_API_KEY` |
| `jina` | `JINA_API_KEY` |
| `kagi` | `KAGI_API_KEY` |
| `tavily` | `TAVILY_API_KEY` |
| `firecrawl` | `FIRECRAWL_API_KEY` |
| `brave` | `BRAVE_API_KEY` |
| `kimi` | `MOONSHOT_API_KEY` |
| `parallel` | `PARALLEL_API_KEY` |
| `synthetic` | `SYNTHETIC_API_KEY` |
| `searxng` | self-hosted |
| `duckduckgo` | no key |
### Specialised handlers
+69 -43
View File
@@ -14,7 +14,9 @@
- `packages/coding-agent/src/web/search/providers/anthropic.ts` — Claude web-search provider.
- `packages/coding-agent/src/web/search/providers/brave.ts` — Brave Search API adapter.
- `packages/coding-agent/src/web/search/providers/codex.ts` — OpenAI Codex SSE adapter.
- `packages/coding-agent/src/web/search/providers/duckduckgo.ts` — DuckDuckGo Instant Answer API adapter.
- `packages/coding-agent/src/web/search/providers/exa.ts` — Exa API or MCP adapter.
- `packages/coding-agent/src/web/search/providers/firecrawl.ts` — Firecrawl search adapter.
- `packages/coding-agent/src/web/search/providers/gemini.ts` — Gemini grounding SSE adapter.
- `packages/coding-agent/src/web/search/providers/jina.ts` — Jina Reader search adapter.
- `packages/coding-agent/src/web/search/providers/kagi.ts` — Kagi provider wrapper.
@@ -24,6 +26,8 @@
- `packages/coding-agent/src/web/search/providers/searxng.ts` — self-hosted SearXNG adapter.
- `packages/coding-agent/src/web/search/providers/synthetic.ts` — Synthetic search adapter.
- `packages/coding-agent/src/web/search/providers/tavily.ts` — Tavily search adapter.
- `packages/coding-agent/src/web/search/providers/tinyfish.ts` — TinyFish search adapter.
- `packages/coding-agent/src/web/search/providers/xai.ts` — xAI Responses web-search adapter.
- `packages/coding-agent/src/web/search/providers/zai.ts` — Z.AI remote MCP adapter.
- `packages/coding-agent/src/web/parallel.ts` — Parallel search/extract HTTP client.
- `packages/coding-agent/src/web/kagi.ts` — Kagi HTTP client.
@@ -34,11 +38,11 @@
| Field | Type | Required | Description |
| --- | --- | --- | --- |
| `query` | `string` | Yes | Search query, passed to providers unchanged. |
| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, and Kagi. |
| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. |
| `max_tokens` | `number` | No | Passed through as `maxOutputTokens` / `max_tokens` only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. |
| `temperature` | `number` | No | Passed through only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. |
| `num_search_results` | `number` | No | Requested upstream search breadth. For most providers this is the same count used for returned sources. Perplexity is the only adapter that keeps it distinct from `limit`. |
| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. |
| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. TinyFish and xAI are local-cap exceptions: TinyFish uses it only for page fetching and slicing; xAI caps parsed sources/citations locally, defaulting to `10` and max `30`. |
| `max_tokens` | `number` | No | Passed through as provider token caps (`maxOutputTokens`, `max_tokens`, or xAI `max_output_tokens`) only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. |
| `temperature` | `number` | No | Passed through only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. |
| `num_search_results` | `number` | No | Requested search breadth or local result cap. Most providers send it upstream. TinyFish and xAI do not; TinyFish clamps to `1..20` with default `10` and uses it for paginated fetches before slicing, while xAI caps parsed sources/citations locally with default `10` and max `30`. |
## Outputs
The tool returns a single text content block plus structured `details`.
@@ -73,7 +77,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
- if `params.provider` is set and not `"auto"`, it loads that provider with `getSearchProvider()`; if `isExplicitlyAvailable()` returns true, the list is `[that provider]`, otherwise it falls back to `resolveProviderChain(authStorage, "auto")`.
- otherwise it calls `resolveProviderChain()` with the module-global preferred provider from `packages/coding-agent/src/web/search/provider.ts`.
3. `resolveProviderChain()` lazily loads each provider module on demand and returns only available providers. If a preferred provider is set, it is tried first (gated by `isExplicitlyAvailable()`), then the static `SEARCH_PROVIDER_ORDER` excluding that provider, each gated by `isAvailable()`. Providers in the excluded set (`setExcludedSearchProviders()`) are skipped entirely, including as the preferred candidate.
4. If no providers are available, `executeSearch()` returns `Error: No web search provider configured.` with `details.response.provider = "none"`.
4. If no providers are available (for example, after excluding DuckDuckGo and lacking configured keyed/OAuth providers), `executeSearch()` returns `Error: No web search provider configured.` with `details.response.provider = "none"`.
5. For each provider in order, `executeSearch()` calls `provider.search()` with:
- `query`,
- `limit`, `recency`, `temperature`, `maxOutputTokens`, `numSearchResults`,
@@ -91,36 +95,20 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
- **Forced provider**: internal callers may pass `provider`; unavailable forced providers fall back to the auto chain instead of hard-failing (`packages/coding-agent/src/web/search/index.ts`). This field is not in the model-facing schema.
- **Preferred provider**: `setPreferredSearchProvider()` sets a module-global default used by `resolveProviderChain()`. `packages/coding-agent/src/sdk.ts` and `packages/coding-agent/src/modes/controllers/selector-controller.ts` wire this from settings.
- **Excluded providers**: `setExcludedSearchProviders()` records providers `resolveProviderChain()` must never return, including as fallbacks. Wired from the `providers.webSearchExclude` setting (`providers.webSearch` drives the preferred provider) in `packages/coding-agent/src/sdk.ts`, `packages/coding-agent/src/modes/interactive-mode.ts`, and `packages/coding-agent/src/modes/controllers/selector-controller.ts`.
- **Auto chain order**: `perplexity`, `gemini`, `anthropic`, `codex`, `zai`, `exa`, `jina`, `kagi`, `tavily`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`).
- **Auto chain order** (18 providers): `perplexity`, `gemini`, `anthropic`, `codex`, `xai`, `zai`, `exa`, `tinyfish`, `jina`, `kagi`, `tavily`, `firecrawl`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng`, `duckduckgo` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`).
- **Provider adapters**
- **Tavily** — `packages/coding-agent/src/web/search/providers/tavily.ts`
- Availability: API key from env or `agent.db` via `findCredential()`.
- Querying: POST `https://api.tavily.com/search`.
- `recency` maps to Tavily `time_range`; code explicitly keeps `topic` at default general scope instead of narrowing to news.
- `limit` / `num_search_results`: adapter uses `params.numSearchResults ?? params.limit`, clamped to `5..20` with default `5`.
- Output: `answer`, `sources`, `requestId`, `authMode: "api_key"`.
- **Perplexity** — `packages/coding-agent/src/web/search/providers/perplexity.ts`
- Availability: auth precedence is `PERPLEXITY_COOKIES` -> OAuth token in `agent.db` -> `PERPLEXITY_API_KEY` / `PPLX_API_KEY` -> anonymous ask-endpoint fallback. `isAvailable()` gates the auto chain on credentials, but `isExplicitlyAvailable()` is always true, so explicit selection works unauthenticated.
- OAuth/cookie/anonymous mode: POSTs to `https://www.perplexity.ai/rest/sse/perplexity_ask`, consumes SSE, merges partial events, extracts answer and source URLs, sets `authMode: "oauth"` (`"anonymous"` for the unauthenticated fallback).
- API-key mode: POSTs to `https://api.perplexity.ai/chat/completions` with `model: "sonar-pro"`, `search_mode: "web"`, `num_search_results`, optional `search_recency_filter`, `max_tokens`, `temperature`.
- `num_search_results` controls upstream API breadth only in API-key mode. `limit` is preserved separately as `num_results` and slices returned `sources` after parsing in both auth modes.
- Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode`.
- **Brave** — `packages/coding-agent/src/web/search/providers/brave.ts`
- Availability: `BRAVE_API_KEY` only.
- Querying: GET `https://api.search.brave.com/res/v1/web/search` with `count`, `extra_snippets=true`, and `freshness=pd|pw|pm|py` for `recency`.
- `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`.
- Output: `sources`, `requestId`.
- **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts`
- Availability: `JINA_API_KEY` only.
- Querying: GET-like fetch to `https://s.jina.ai/<encoded query>` with bearer auth.
- Ignores `recency`, `max_tokens`, and `temperature`.
- `limit` / `num_search_results`: adapter slices sources to `params.numSearchResults ?? params.limit` when provided; otherwise returns all payload items.
- Output: `sources` only.
- **Kimi** — `packages/coding-agent/src/web/search/providers/kimi.ts`
- Availability: `MOONSHOT_SEARCH_API_KEY`, `KIMI_SEARCH_API_KEY`, `MOONSHOT_API_KEY`, or `agent.db` credentials for `moonshot` / `kimi-code`.
- Querying: POST to `MOONSHOT_SEARCH_BASE_URL` / `KIMI_SEARCH_BASE_URL` / default `https://api.kimi.com/coding/v1/search` with `text_query`, `limit`, `enable_page_crawling`, `timeout_seconds: 30`.
- `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`.
- Output: `sources`, `requestId`.
- **Gemini** — `packages/coding-agent/src/web/search/providers/gemini.ts`
- Availability: OAuth credentials in `agent.db` for `google-gemini-cli` or `google-antigravity`.
- Querying: SSE `streamGenerateContent` call with Google Search grounding enabled. Antigravity auth tries two fallback endpoints and retries `401/403/400 invalid auth` once after token refresh; `429/5xx` retry with exponential backoff and server-provided retry delay, capped by a `5 * 60 * 1000` ms rate-limit budget.
- `max_tokens` and `temperature` pass through as `generationConfig.maxOutputTokens` / `generationConfig.temperature`.
- `limit` and `num_search_results` are collapsed together before dispatch.
- Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage`, `model`.
- **Anthropic** — `packages/coding-agent/src/web/search/providers/anthropic.ts`
- Availability: `ANTHROPIC_SEARCH_API_KEY` env var, otherwise `authStorage.hasAuth("anthropic")`; search credentials come from `authStorage.getApiKey("anthropic")` when no search-specific key is set.
- Env overrides specific to search (do not affect chat completions):
@@ -131,18 +119,17 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
- `max_tokens` and `temperature` pass through.
- `limit` and `num_search_results` are collapsed together before dispatch: `num_results = params.numSearchResults ?? params.limit`.
- Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage.searchRequests`, `model`, `requestId`.
- **Gemini** — `packages/coding-agent/src/web/search/providers/gemini.ts`
- Availability: OAuth credentials in `agent.db` for `google-gemini-cli` or `google-antigravity`.
- Querying: SSE `streamGenerateContent` call with Google Search grounding enabled. Antigravity auth tries two fallback endpoints and retries `401/403/400 invalid auth` once after token refresh; `429/5xx` retry with exponential backoff and server-provided retry delay, capped by a `5 * 60 * 1000` ms rate-limit budget.
- `max_tokens` and `temperature` pass through as `generationConfig.maxOutputTokens` / `generationConfig.temperature`.
- `limit` and `num_search_results` are collapsed together before dispatch.
- Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage`, `model`.
- **Codex** — `packages/coding-agent/src/web/search/providers/codex.ts`
- Availability: OAuth credential for `openai-codex` in `agent.db` (`hasOAuth()`; expiry is not checked here — refresh is lazy in `searchCodex`).
- Querying: SSE POST to `https://chatgpt.com/backend-api/codex/responses` with `tool_choice: { type: "web_search" }` and `search_context_size: "high"` by default.
- Ignores `recency`, `max_tokens`, and `temperature` in this tool path.
- `limit` and `num_search_results` are collapsed together before dispatch.
- Output may include `answer`, `sources`, `usage`, `model`, `requestId`. If the streamed response has no `url_citation` annotations, the adapter falls back to scraping markdown links and bare URLs from the answer text.
- **xAI** — `packages/coding-agent/src/web/search/providers/xai.ts`
- Availability: `XAI_API_KEY` or `agent.db` credential for `xai`.
- Querying: POST `https://api.x.ai/v1/responses` with model `grok-4.3` and `tools: [{ type: "web_search" }]` using the `/v1/responses` Agent Tools API.
- `max_tokens` and `temperature` pass through. `recency` is ignored because the documented Agent Tools `web_search` API does not expose date controls. `limit` and `num_search_results` are not sent upstream; because xAI citations may include every encountered URL, the adapter locally caps returned `sources` and `citations` after parsing. The local cap uses `num_search_results` before `limit`, defaults to `10` when omitted/invalid/zero, and is capped at `30`.
- Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode: "api_key"`.
- **Z.AI** — `packages/coding-agent/src/web/search/providers/zai.ts`
- Availability: env or `agent.db` credential for `zai`.
- Querying: JSON-RPC `tools/call` against `https://api.z.ai/api/mcp/web_search_prime/mcp` for remote MCP tool `web_search_prime`.
@@ -154,17 +141,47 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
- Querying: POST `https://api.exa.ai/search` with the resolved Exa API key, otherwise JSON-RPC `tools/call` against `https://mcp.exa.ai/mcp` for remote MCP tool `web_search_exa`.
- `limit` and `num_search_results` are collapsed together before dispatch.
- Output: synthesized `answer` from up to 3 result summaries, `sources`, `requestId`.
- **TinyFish** — `packages/coding-agent/src/web/search/providers/tinyfish.ts`
- Availability: `TINYFISH_API_KEY` or `agent.db` credential for `tinyfish`.
- Querying: GET `https://api.search.tinyfish.ai` with `X-API-Key` and `query`; `recency` maps to `recency_minutes`.
- `limit` / `num_search_results`: collapsed as `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. TinyFish has no count parameter and returns at most 10 results per page; for counts above the first page, the adapter fetches documented `page` values (`0`, then `1` when needed) before slicing locally. Output `sources`, `authMode: "api_key"`.
- **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts`
- Availability: `JINA_API_KEY` only.
- Querying: GET-like fetch to `https://s.jina.ai/<encoded query>` with bearer auth.
- Ignores `recency`, `max_tokens`, and `temperature`.
- `limit` / `num_search_results`: adapter slices sources to `params.numSearchResults ?? params.limit` when provided; otherwise returns all payload items.
- Output: `sources` only.
- **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts`
- Availability: env or `agent.db` credential for `kagi`.
- Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer <key>` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`).
- `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`.
- Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`).
- **Tavily** — `packages/coding-agent/src/web/search/providers/tavily.ts`
- Availability: API key from env or `agent.db` via `findCredential()`.
- Querying: POST `https://api.tavily.com/search`.
- `recency` maps to Tavily `time_range`; code explicitly keeps `topic` at default general scope instead of narrowing to news.
- `limit` / `num_search_results`: adapter uses `params.numSearchResults ?? params.limit`, clamped to `5..20` with default `5`.
- Output: `answer`, `sources`, `requestId`, `authMode: "api_key"`.
- **Firecrawl** — `packages/coding-agent/src/web/search/providers/firecrawl.ts`
- Availability: `FIRECRAWL_API_KEY` or `agent.db` credential for `firecrawl`.
- Querying: POST `https://api.firecrawl.dev/v2/search` with `sources: [{ type: "web" }]`; `recency` maps to Google-style `tbs`.
- `limit` / `num_search_results`: collapsed and clamped to `1..100`, default `10`; output `sources`, `requestId`, `authMode: "api_key"`.
- **Brave** — `packages/coding-agent/src/web/search/providers/brave.ts`
- Availability: `BRAVE_API_KEY` only.
- Querying: GET `https://api.search.brave.com/res/v1/web/search` with `count`, `extra_snippets=true`, and `freshness=pd|pw|pm|py` for `recency`.
- `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`.
- Output: `sources`, `requestId`.
- **Kimi** — `packages/coding-agent/src/web/search/providers/kimi.ts`
- Availability: `MOONSHOT_SEARCH_API_KEY`, `KIMI_SEARCH_API_KEY`, `MOONSHOT_API_KEY`, or `agent.db` credentials for `moonshot` / `kimi-code`.
- Querying: POST to `MOONSHOT_SEARCH_BASE_URL` / `KIMI_SEARCH_BASE_URL` / default `https://api.kimi.com/coding/v1/search` with `text_query`, `limit`, `enable_page_crawling`, `timeout_seconds: 30`.
- `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`.
- Output: `sources`, `requestId`.
- **Parallel** — `packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts`
- Availability: env or `agent.db` credential for `parallel`.
- Querying: POST `https://api.parallel.ai/v1beta/search` with `objective=query`, `search_queries=[query]`, `mode:"fast"`, `max_chars_per_result: 10000`, beta header `search-extract-2025-10-10`.
- There is no provider fan-out here despite the name; the current adapter always sends a one-element `search_queries` array.
- `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`.
- Output: `sources`, `requestId`.
- **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts`
- Availability: env or `agent.db` credential for `kagi`.
- Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer <key>` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`).
- `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`.
- Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`).
- **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts`
- Availability: env or `agent.db` credential for `synthetic`.
- Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`.
@@ -178,6 +195,10 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
- `recency` maps to `time_range`; `week` is downgraded to `month` because SearXNG does not support week.
- `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..20`, default `10`.
- Output: `sources`, `relatedQuestions` from `suggestions`.
- **DuckDuckGo** — `packages/coding-agent/src/web/search/providers/duckduckgo.ts`
- Availability: always available; no API key.
- Querying: GET official Instant Answer API `https://api.duckduckgo.com/` with JSON/no-HTML flags; no scraped HTML.
- `limit` / `num_search_results`: collapsed and clamped to `1..20`, default `10`; output may include `answer` and `sources` from abstracts/results/topics.
## Side Effects
- Network
@@ -193,15 +214,19 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
- Many provider adapters accept `AbortSignal`; `WebSearchTool.execute()` passes the tool call signal into `executeSearch()`, which forwards it as `params.signal` to providers and rethrows cancellation during fallback.
## Limits & Caps
- Provider auto-order length: 14 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`).
- Provider auto-order length: 18 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`).
- `formatForLLM()` truncates source snippets and citation text to 240 chars (`packages/coding-agent/src/web/search/index.ts`).
- `formatForLLM()` emits at most 3 search queries, each truncated to 120 chars (`packages/coding-agent/src/web/search/index.ts`).
- Brave result count: default `10`, max `20` (`DEFAULT_NUM_RESULTS`, `MAX_NUM_RESULTS` in `packages/coding-agent/src/web/search/providers/brave.ts`).
- TinyFish local result count: default `10`, max `20`; the API has no count parameter and returns at most 10 results per page, so the adapter fetches documented pages (`page=0`, then `page=1` when needed) and slices locally (`packages/coding-agent/src/web/search/providers/tinyfish.ts`).
- DuckDuckGo result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/duckduckgo.ts`).
- Tavily result count: default `5`, max `20` (`packages/coding-agent/src/web/search/providers/tavily.ts`).
- Firecrawl result count: default `10`, max `100` (`packages/coding-agent/src/web/search/providers/firecrawl.ts`).
- Kimi result count: default `10`, max `20`; request timeout field fixed to `30` seconds (`packages/coding-agent/src/web/search/providers/kimi.ts`).
- Parallel result count: default `10`, max `40`; per-result excerpt cap `10_000` chars (`packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts`).
- Kagi result count: default `10`, max `40` (`packages/coding-agent/src/web/search/providers/kagi.ts`).
- SearXNG result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/searxng.ts`).
- xAI local sources/citations cap: `num_search_results` before `limit`, omitted/invalid/zero => default `10`, max `30`; neither value is sent upstream (`packages/coding-agent/src/web/search/providers/xai.ts`).
- Perplexity API-key mode defaults: `max_tokens = 8192`, `temperature = 0.2`, `num_search_results = 20` (`packages/coding-agent/src/web/search/providers/perplexity.ts`).
- Anthropic defaults: model `claude-haiku-4-5`, `DEFAULT_MAX_TOKENS = 4096` when the provider omits `max_tokens` (`packages/coding-agent/src/web/search/providers/anthropic.ts`).
- Gemini retries: up to `3` retries per endpoint, base delay `1000` ms, rate-limit delay budget `5 * 60 * 1000` ms (`packages/coding-agent/src/web/search/providers/gemini.ts`).
@@ -222,7 +247,8 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
## Notes
- The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`.
- `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports.
- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity is the only implementation that preserves both concepts.
- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, and Kagi; the model-facing prompt does not name specific providers.
- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts. TinyFish uses that collapsed value only as a local cap and to decide whether to fetch page `1`; it does not serialize a count parameter. xAI applies that same precedence locally after parsing to cap returned sources/citations without sending a result-count control upstream (`10` default, `30` max).
- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; xAI ignores it because the documented Agent Tools `web_search` API does not expose date controls. The model-facing prompt does not name specific providers.
- `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain.
- DuckDuckGo is intentionally last in the auto chain because it is always available without credentials.
- Exa uses `authStorage.getApiKey("exa")`, then `EXA_API_KEY`, then unauthenticated `https://mcp.exa.ai/mcp` fallback.
+2
View File
@@ -619,6 +619,8 @@ const LEGACY_ENV_KEYS: Record<string, KeyResolver> = {
exa: "EXA_API_KEY",
jina: "JINA_API_KEY",
brave: "BRAVE_API_KEY",
tinyfish: "TINYFISH_API_KEY",
firecrawl: "FIRECRAWL_API_KEY",
};
/**
+1
View File
@@ -63,6 +63,7 @@
### Added
- Added TinyFish, DuckDuckGo, xAI, and Firecrawl web_search providers.
- Added `compaction.midTurnEnabled` for mid-turn threshold auto-compaction before the next tool-loop provider request. ([#3525](https://github.com/can1357/oh-my-pi/issues/3525))
- Added `grep -q`/`--quiet`/`--silent` and `-x`/`--line-regexp` to the in-process `grep` builtin used by the bash tool. `-q` suppresses all stdout and exits 0 on the first match (short-circuiting, with match status taking precedence over read errors per GNU); `-x` anchors each pattern to whole lines. Unblocks shell conditionals such as `grep -qx "$applet" <(strings bin)`.
- Added plan-mode guidance (hashline edit mode only) steering the agent to revise the plan file section-by-section with `SWAP.BLK`/`DEL.BLK`/`INS.BLK.POST` anchored on markdown headings — a heading resolves its whole section (through nested deeper headings), so the agent can rewrite, drop, or append sections without rewriting the file.
+2
View File
@@ -311,6 +311,8 @@ export function getExtraHelpText(): string {
PERPLEXITY_API_KEY - Perplexity web search API key (optional; anonymous fallback)
PERPLEXITY_COOKIES - Perplexity web search (session cookie)
TAVILY_API_KEY - Tavily web search
TINYFISH_API_KEY - TinyFish web search
FIRECRAWL_API_KEY - Firecrawl web search
ANTHROPIC_SEARCH_API_KEY - Anthropic web search (override; isolates search from main ANTHROPIC_API_KEY)
ANTHROPIC_SEARCH_BASE_URL - Anthropic web search base URL (override; pairs with ANTHROPIC_SEARCH_API_KEY)
@@ -120,7 +120,7 @@ ${chalk.bold("Arguments:")}
${chalk.bold("Options:")}
--provider <name> Provider: ${PROVIDERS.join(", ")}
--recency <value> Recency filter (Brave/Perplexity): ${RECENCY_OPTIONS.join(", ")}
--recency <value> Recency filter (when supported): ${RECENCY_OPTIONS.join(", ")}
-l, --limit <n> Max results to return
--compact Render condensed output
-h, --help Show this help
@@ -956,7 +956,9 @@ export const BUNDLED_PI_REGISTRY_KEYS: ReadonlySet<string> = new Set([
"@oh-my-pi/pi-coding-agent/web/search/providers/base",
"@oh-my-pi/pi-coding-agent/web/search/providers/brave",
"@oh-my-pi/pi-coding-agent/web/search/providers/codex",
"@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo",
"@oh-my-pi/pi-coding-agent/web/search/providers/exa",
"@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl",
"@oh-my-pi/pi-coding-agent/web/search/providers/gemini",
"@oh-my-pi/pi-coding-agent/web/search/providers/jina",
"@oh-my-pi/pi-coding-agent/web/search/providers/kagi",
@@ -967,7 +969,9 @@ export const BUNDLED_PI_REGISTRY_KEYS: ReadonlySet<string> = new Set([
"@oh-my-pi/pi-coding-agent/web/search/providers/searxng",
"@oh-my-pi/pi-coding-agent/web/search/providers/synthetic",
"@oh-my-pi/pi-coding-agent/web/search/providers/tavily",
"@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish",
"@oh-my-pi/pi-coding-agent/web/search/providers/utils",
"@oh-my-pi/pi-coding-agent/web/search/providers/xai",
"@oh-my-pi/pi-coding-agent/web/search/providers/zai",
"@oh-my-pi/pi-natives",
"@oh-my-pi/pi-tui",
@@ -958,7 +958,9 @@ import * as bundledPiCodingAgentWebSearchProvidersAnthropic from "@oh-my-pi/pi-c
import * as bundledPiCodingAgentWebSearchProvidersBase from "@oh-my-pi/pi-coding-agent/web/search/providers/base";
import * as bundledPiCodingAgentWebSearchProvidersBrave from "@oh-my-pi/pi-coding-agent/web/search/providers/brave";
import * as bundledPiCodingAgentWebSearchProvidersCodex from "@oh-my-pi/pi-coding-agent/web/search/providers/codex";
import * as bundledPiCodingAgentWebSearchProvidersDuckduckgo from "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo";
import * as bundledPiCodingAgentWebSearchProvidersExa from "@oh-my-pi/pi-coding-agent/web/search/providers/exa";
import * as bundledPiCodingAgentWebSearchProvidersFirecrawl from "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl";
import * as bundledPiCodingAgentWebSearchProvidersGemini from "@oh-my-pi/pi-coding-agent/web/search/providers/gemini";
import * as bundledPiCodingAgentWebSearchProvidersJina from "@oh-my-pi/pi-coding-agent/web/search/providers/jina";
import * as bundledPiCodingAgentWebSearchProvidersKagi from "@oh-my-pi/pi-coding-agent/web/search/providers/kagi";
@@ -969,7 +971,9 @@ import * as bundledPiCodingAgentWebSearchProvidersPerplexityAuth from "@oh-my-pi
import * as bundledPiCodingAgentWebSearchProvidersSearxng from "@oh-my-pi/pi-coding-agent/web/search/providers/searxng";
import * as bundledPiCodingAgentWebSearchProvidersSynthetic from "@oh-my-pi/pi-coding-agent/web/search/providers/synthetic";
import * as bundledPiCodingAgentWebSearchProvidersTavily from "@oh-my-pi/pi-coding-agent/web/search/providers/tavily";
import * as bundledPiCodingAgentWebSearchProvidersTinyfish from "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish";
import * as bundledPiCodingAgentWebSearchProvidersUtils from "@oh-my-pi/pi-coding-agent/web/search/providers/utils";
import * as bundledPiCodingAgentWebSearchProvidersXai from "@oh-my-pi/pi-coding-agent/web/search/providers/xai";
import * as bundledPiCodingAgentWebSearchProvidersZai from "@oh-my-pi/pi-coding-agent/web/search/providers/zai";
import * as bundledPiCodingAgentWebSearchRender from "@oh-my-pi/pi-coding-agent/web/search/render";
import * as bundledPiCodingAgentWebSearchTypes from "@oh-my-pi/pi-coding-agent/web/search/types";
@@ -3302,8 +3306,12 @@ export const BUNDLED_PI_REGISTRY: Readonly<Record<string, Readonly<Record<string
bundledPiCodingAgentWebSearchProvidersBrave as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/codex":
bundledPiCodingAgentWebSearchProvidersCodex as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo":
bundledPiCodingAgentWebSearchProvidersDuckduckgo as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/exa":
bundledPiCodingAgentWebSearchProvidersExa as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl":
bundledPiCodingAgentWebSearchProvidersFirecrawl as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/gemini":
bundledPiCodingAgentWebSearchProvidersGemini as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/jina":
@@ -3324,8 +3332,12 @@ export const BUNDLED_PI_REGISTRY: Readonly<Record<string, Readonly<Record<string
bundledPiCodingAgentWebSearchProvidersSynthetic as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/tavily":
bundledPiCodingAgentWebSearchProvidersTavily as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish":
bundledPiCodingAgentWebSearchProvidersTinyfish as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/utils":
bundledPiCodingAgentWebSearchProvidersUtils as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/xai":
bundledPiCodingAgentWebSearchProvidersXai as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-coding-agent/web/search/providers/zai":
bundledPiCodingAgentWebSearchProvidersZai as unknown as Readonly<Record<string, unknown>>,
"@oh-my-pi/pi-natives": bundledPiNatives as unknown as Readonly<Record<string, unknown>>,
@@ -24,66 +24,81 @@ interface ProviderMeta {
/** Lazy factories. Each `load()` dynamic-imports its provider module on first call. */
const PROVIDER_META: Record<SearchProviderId, ProviderMeta> = {
exa: {
id: "exa",
label: SEARCH_PROVIDER_LABELS.exa,
load: async () => new (await import("./providers/exa")).ExaProvider(),
},
brave: {
id: "brave",
label: SEARCH_PROVIDER_LABELS.brave,
load: async () => new (await import("./providers/brave")).BraveProvider(),
},
jina: {
id: "jina",
label: SEARCH_PROVIDER_LABELS.jina,
load: async () => new (await import("./providers/jina")).JinaProvider(),
},
perplexity: {
id: "perplexity",
label: SEARCH_PROVIDER_LABELS.perplexity,
load: async () => new (await import("./providers/perplexity")).PerplexityProvider(),
},
kimi: {
id: "kimi",
label: SEARCH_PROVIDER_LABELS.kimi,
load: async () => new (await import("./providers/kimi")).KimiProvider(),
},
zai: {
id: "zai",
label: SEARCH_PROVIDER_LABELS.zai,
load: async () => new (await import("./providers/zai")).ZaiProvider(),
},
anthropic: {
id: "anthropic",
label: SEARCH_PROVIDER_LABELS.anthropic,
load: async () => new (await import("./providers/anthropic")).AnthropicProvider(),
},
gemini: {
id: "gemini",
label: SEARCH_PROVIDER_LABELS.gemini,
load: async () => new (await import("./providers/gemini")).GeminiProvider(),
},
anthropic: {
id: "anthropic",
label: SEARCH_PROVIDER_LABELS.anthropic,
load: async () => new (await import("./providers/anthropic")).AnthropicProvider(),
},
codex: {
id: "codex",
label: SEARCH_PROVIDER_LABELS.codex,
load: async () => new (await import("./providers/codex")).CodexProvider(),
},
xai: {
id: "xai",
label: SEARCH_PROVIDER_LABELS.xai,
load: async () => new (await import("./providers/xai")).XAIProvider(),
},
zai: {
id: "zai",
label: SEARCH_PROVIDER_LABELS.zai,
load: async () => new (await import("./providers/zai")).ZaiProvider(),
},
exa: {
id: "exa",
label: SEARCH_PROVIDER_LABELS.exa,
load: async () => new (await import("./providers/exa")).ExaProvider(),
},
tinyfish: {
id: "tinyfish",
label: SEARCH_PROVIDER_LABELS.tinyfish,
load: async () => new (await import("./providers/tinyfish")).TinyFishProvider(),
},
jina: {
id: "jina",
label: SEARCH_PROVIDER_LABELS.jina,
load: async () => new (await import("./providers/jina")).JinaProvider(),
},
kagi: {
id: "kagi",
label: SEARCH_PROVIDER_LABELS.kagi,
load: async () => new (await import("./providers/kagi")).KagiProvider(),
},
tavily: {
id: "tavily",
label: SEARCH_PROVIDER_LABELS.tavily,
load: async () => new (await import("./providers/tavily")).TavilyProvider(),
},
firecrawl: {
id: "firecrawl",
label: SEARCH_PROVIDER_LABELS.firecrawl,
load: async () => new (await import("./providers/firecrawl")).FirecrawlProvider(),
},
brave: {
id: "brave",
label: SEARCH_PROVIDER_LABELS.brave,
load: async () => new (await import("./providers/brave")).BraveProvider(),
},
kimi: {
id: "kimi",
label: SEARCH_PROVIDER_LABELS.kimi,
load: async () => new (await import("./providers/kimi")).KimiProvider(),
},
parallel: {
id: "parallel",
label: SEARCH_PROVIDER_LABELS.parallel,
load: async () => new (await import("./providers/parallel")).ParallelProvider(),
},
kagi: {
id: "kagi",
label: SEARCH_PROVIDER_LABELS.kagi,
load: async () => new (await import("./providers/kagi")).KagiProvider(),
},
synthetic: {
id: "synthetic",
label: SEARCH_PROVIDER_LABELS.synthetic,
@@ -94,6 +109,11 @@ const PROVIDER_META: Record<SearchProviderId, ProviderMeta> = {
label: SEARCH_PROVIDER_LABELS.searxng,
load: async () => new (await import("./providers/searxng")).SearXNGProvider(),
},
duckduckgo: {
id: "duckduckgo",
label: SEARCH_PROVIDER_LABELS.duckduckgo,
load: async () => new (await import("./providers/duckduckgo")).DuckDuckGoProvider(),
},
};
const instanceCache = new Map<SearchProviderId, SearchProvider>();
@@ -0,0 +1,140 @@
import type { AuthStorage } from "@oh-my-pi/pi-ai";
import type { SearchResponse, SearchSource } from "../../../web/search/types";
import { SearchProviderError } from "../../../web/search/types";
import { clampNumResults } from "../utils";
import type { SearchParams } from "./base";
import { SearchProvider } from "./base";
import { classifyProviderHttpError, withHardTimeout } from "./utils";
const DUCKDUCKGO_SEARCH_URL = "https://api.duckduckgo.com/";
const DEFAULT_NUM_RESULTS = 10;
const MAX_NUM_RESULTS = 20;
interface DuckDuckGoTopic {
FirstURL?: string | null;
Text?: string | null;
Topics?: DuckDuckGoTopic[] | null;
}
interface DuckDuckGoResponse {
AbstractText?: string | null;
AbstractURL?: string | null;
AbstractSource?: string | null;
Answer?: string | null;
Definition?: string | null;
Heading?: string | null;
Results?: DuckDuckGoTopic[] | null;
RelatedTopics?: DuckDuckGoTopic[] | null;
}
function cleanText(value: string | null | undefined): string | undefined {
const cleaned = value
?.replace(/<[^>]*>/g, " ")
.replace(/&nbsp;/gi, " ")
.replace(/&amp;/gi, "&")
.replace(/&lt;/gi, "<")
.replace(/&gt;/gi, ">")
.replace(/&quot;/gi, '"')
.replace(/&#39;/gi, "'")
.replace(/\s+/g, " ")
.trim();
return cleaned ? cleaned : undefined;
}
function addSource(sources: SearchSource[], source: SearchSource): void {
if (!source.url || sources.some(existing => existing.url === source.url)) return;
sources.push(source);
}
function addTopicSource(sources: SearchSource[], topic: DuckDuckGoTopic): void {
const url = topic.FirstURL?.trim();
if (!url) return;
const text = cleanText(topic.Text);
addSource(sources, {
title: text ?? url,
url,
snippet: text,
});
}
function collectTopicSources(sources: SearchSource[], topics: readonly DuckDuckGoTopic[] | null | undefined): void {
if (!topics) return;
for (const topic of topics) {
addTopicSource(sources, topic);
collectTopicSources(sources, topic.Topics);
}
}
async function callDuckDuckGoSearch(params: SearchParams): Promise<DuckDuckGoResponse> {
const queryString = [
["q", params.query],
["format", "json"],
["no_redirect", "1"],
["no_html", "1"],
["skip_disambig", "1"],
["t", "oh-my-pi"],
]
.map(([key, value]) => `${encodeURIComponent(key)}=${encodeURIComponent(value)}`)
.join("&");
const response = await (params.fetch ?? fetch)(`${DUCKDUCKGO_SEARCH_URL}?${queryString}`, {
method: "GET",
signal: withHardTimeout(params.signal),
});
if (!response.ok) {
const errorText = await response.text();
const classified = classifyProviderHttpError("duckduckgo", response.status, errorText);
if (classified) throw classified;
throw new SearchProviderError(
"duckduckgo",
`DuckDuckGo API error (${response.status}): ${errorText}`,
response.status,
);
}
return (await response.json()) as DuckDuckGoResponse;
}
/** Execute DuckDuckGo Instant Answer API search. */
export async function searchDuckDuckGo(params: SearchParams): Promise<SearchResponse> {
const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS);
const data = await callDuckDuckGoSearch(params);
const answer = cleanText(data.AbstractText) ?? cleanText(data.Answer) ?? cleanText(data.Definition);
const sources: SearchSource[] = [];
const abstractUrl = data.AbstractURL?.trim();
if (abstractUrl) {
addSource(sources, {
title: cleanText(data.AbstractSource) ?? cleanText(data.Heading) ?? abstractUrl,
url: abstractUrl,
snippet: cleanText(data.AbstractText),
});
}
collectTopicSources(sources, data.Results);
collectTopicSources(sources, data.RelatedTopics);
return {
provider: "duckduckgo",
answer,
sources: sources.slice(0, numResults),
};
}
/** Search provider for DuckDuckGo Instant Answer API. */
export class DuckDuckGoProvider extends SearchProvider {
readonly id = "duckduckgo";
readonly label = "DuckDuckGo";
isAvailable(_authStorage: AuthStorage): boolean {
return true;
}
isExplicitlyAvailable(_authStorage: AuthStorage): boolean {
return true;
}
search(params: SearchParams): Promise<SearchResponse> {
return searchDuckDuckGo(params);
}
}
@@ -0,0 +1,144 @@
/**
* Firecrawl Web Search Provider
*
* Calls Firecrawl's search API and maps web results into the unified
* SearchResponse shape used by the web search tool.
*/
import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai";
import type { SearchResponse, SearchSource } from "../../../web/search/types";
import { SearchProviderError } from "../../../web/search/types";
import { clampNumResults } from "../utils";
import type { SearchParams } from "./base";
import { SearchProvider } from "./base";
import { classifyProviderHttpError, withHardTimeout } from "./utils";
const FIRECRAWL_SEARCH_URL = "https://api.firecrawl.dev/v2/search";
const DEFAULT_NUM_RESULTS = 10;
const MAX_NUM_RESULTS = 100;
const RECENCY_TBS: Record<NonNullable<SearchParams["recency"]>, string> = {
day: "qdr:d",
week: "qdr:w",
month: "qdr:m",
year: "qdr:y",
};
export interface FirecrawlSearchParams {
query: string;
num_results?: number;
recency?: SearchParams["recency"];
signal?: AbortSignal;
fetch?: FetchImpl;
}
interface FirecrawlWebResult {
title?: string | null;
url?: string | null;
description?: string | null;
markdown?: string | null;
}
interface FirecrawlSearchResponse {
id?: string | null;
data?: {
web?: FirecrawlWebResult[] | null;
} | null;
}
/** Resolve Firecrawl API key through the shared auth storage pipeline. */
export function findApiKey(
authStorage: AuthStorage,
sessionId?: string,
signal?: AbortSignal,
): Promise<string | undefined> {
return authStorage.getApiKey("firecrawl", sessionId, { signal });
}
function buildRequestBody(params: FirecrawlSearchParams): Record<string, unknown> {
const body: Record<string, unknown> = {
query: params.query,
limit: clampNumResults(params.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS),
sources: [{ type: "web" }],
};
if (params.recency) {
body.tbs = RECENCY_TBS[params.recency];
}
return body;
}
async function callFirecrawlSearch(apiKey: string, params: FirecrawlSearchParams): Promise<FirecrawlSearchResponse> {
const response = await (params.fetch ?? fetch)(FIRECRAWL_SEARCH_URL, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${apiKey}`,
},
body: JSON.stringify(buildRequestBody(params)),
signal: withHardTimeout(params.signal),
});
if (!response.ok) {
const errorText = await response.text();
const classified = classifyProviderHttpError("firecrawl", response.status, errorText);
if (classified) throw classified;
throw new SearchProviderError(
"firecrawl",
`Firecrawl API error (${response.status}): ${errorText}`,
response.status,
);
}
return (await response.json()) as FirecrawlSearchResponse;
}
/** Execute Firecrawl web search. */
export async function searchFirecrawl(params: SearchParams): Promise<SearchResponse> {
const firecrawlParams: FirecrawlSearchParams = {
query: params.query,
num_results: params.numSearchResults ?? params.limit,
recency: params.recency,
signal: params.signal,
fetch: params.fetch,
};
const keyOrResolver: ApiKey = params.authStorage.resolver("firecrawl", {
sessionId: params.sessionId,
});
const numResults = clampNumResults(firecrawlParams.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS);
const data = await withAuth(keyOrResolver, key => callFirecrawlSearch(key, firecrawlParams), {
signal: params.signal,
missingKeyMessage:
'Firecrawl credentials not found. Set FIRECRAWL_API_KEY or configure an API key for provider "firecrawl".',
});
const sources: SearchSource[] = [];
for (const result of data.data?.web ?? []) {
if (!result.url) continue;
sources.push({
title: result.title ?? result.url,
url: result.url,
snippet: result.description ?? result.markdown ?? undefined,
});
}
return {
provider: "firecrawl",
sources: sources.slice(0, numResults),
requestId: data.id ?? undefined,
authMode: "api_key",
};
}
/** Search provider for Firecrawl web search. */
export class FirecrawlProvider extends SearchProvider {
readonly id = "firecrawl";
readonly label = "Firecrawl";
isAvailable(authStorage: AuthStorage): boolean {
return authStorage.hasAuth("firecrawl") || !!getEnvApiKey("firecrawl");
}
search(params: SearchParams): Promise<SearchResponse> {
return searchFirecrawl(params);
}
}
@@ -0,0 +1,160 @@
/**
* TinyFish Web Search Provider
*
* Calls TinyFish's search API and maps results into the unified
* SearchResponse shape used by the web search tool.
*/
import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai";
import type { SearchResponse, SearchSource } from "../../../web/search/types";
import { SearchProviderError } from "../../../web/search/types";
import { clampNumResults } from "../utils";
import type { SearchParams } from "./base";
import { SearchProvider } from "./base";
import { classifyProviderHttpError, withHardTimeout } from "./utils";
const TINYFISH_SEARCH_URL = "https://api.search.tinyfish.ai";
const DEFAULT_NUM_RESULTS = 10;
const MAX_NUM_RESULTS = 20;
const RECENCY_MINUTES: Record<NonNullable<SearchParams["recency"]>, number> = {
day: 1440,
week: 10080,
month: 43200,
year: 525600,
};
export interface TinyFishSearchParams {
query: string;
num_results?: number;
recency?: SearchParams["recency"];
page?: number;
signal?: AbortSignal;
fetch?: FetchImpl;
}
interface TinyFishSearchResult {
title?: string | null;
url?: string | null;
snippet?: string | null;
site_name?: string | null;
}
interface TinyFishSearchResponse {
total_results?: number | null;
page?: number | null;
results?: TinyFishSearchResult[] | null;
}
/** Resolve TinyFish API key through the shared auth storage pipeline. */
export function findApiKey(
authStorage: AuthStorage,
sessionId?: string,
signal?: AbortSignal,
): Promise<string | undefined> {
return authStorage.getApiKey("tinyfish", sessionId, { signal });
}
async function callTinyFishSearch(apiKey: string, params: TinyFishSearchParams): Promise<TinyFishSearchResponse> {
const url = new URL(TINYFISH_SEARCH_URL);
url.searchParams.set("query", params.query);
if (params.recency) {
url.searchParams.set("recency_minutes", String(RECENCY_MINUTES[params.recency]));
}
if (params.page !== undefined) {
url.searchParams.set("page", String(params.page));
}
const response = await (params.fetch ?? fetch)(url, {
method: "GET",
headers: {
Accept: "application/json",
"X-API-Key": apiKey,
},
signal: withHardTimeout(params.signal),
});
if (!response.ok) {
const errorText = await response.text();
const classified = classifyProviderHttpError("tinyfish", response.status, errorText);
if (classified) throw classified;
throw new SearchProviderError(
"tinyfish",
`TinyFish API error (${response.status}): ${errorText}`,
response.status,
);
}
return (await response.json()) as TinyFishSearchResponse;
}
function appendTinyFishSources(sources: SearchSource[], results: readonly TinyFishSearchResult[]): void {
for (const result of results) {
if (!result.url) continue;
sources.push({
title: result.title ?? result.site_name ?? result.url,
url: result.url,
snippet: result.snippet ?? undefined,
author: result.site_name ?? undefined,
});
}
}
/** Execute TinyFish web search. */
export async function searchTinyFish(params: SearchParams): Promise<SearchResponse> {
const tinyFishParams: TinyFishSearchParams = {
query: params.query,
recency: params.recency,
signal: params.signal,
fetch: params.fetch,
};
const keyOrResolver: ApiKey = params.authStorage.resolver("tinyfish", {
sessionId: params.sessionId,
});
const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS);
const sources = await withAuth(
keyOrResolver,
async key => {
const collected: SearchSource[] = [];
const firstPage = await callTinyFishSearch(key, { ...tinyFishParams, page: 0 });
const firstPageResults = firstPage.results ?? [];
appendTinyFishSources(collected, firstPageResults);
if (
numResults > DEFAULT_NUM_RESULTS &&
collected.length < numResults &&
firstPageResults.length >= DEFAULT_NUM_RESULTS
) {
const secondPage = await callTinyFishSearch(key, { ...tinyFishParams, page: 1 });
appendTinyFishSources(collected, secondPage.results ?? []);
}
return collected.slice(0, numResults);
},
{
signal: params.signal,
missingKeyMessage:
'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".',
},
);
return {
provider: "tinyfish",
sources,
authMode: "api_key",
};
}
/** Search provider for TinyFish web search. */
export class TinyFishProvider extends SearchProvider {
readonly id = "tinyfish";
readonly label = "TinyFish";
isAvailable(authStorage: AuthStorage): boolean {
return authStorage.hasAuth("tinyfish") || !!getEnvApiKey("tinyfish");
}
search(params: SearchParams): Promise<SearchResponse> {
return searchTinyFish(params);
}
}
@@ -0,0 +1,250 @@
import { type ApiKey, type AuthStorage, withAuth } from "@oh-my-pi/pi-ai";
import type { SearchCitation, SearchResponse, SearchSource, SearchUsage } from "../../../web/search/types";
import { SearchProviderError } from "../../../web/search/types";
import { clampNumResults } from "../utils";
import type { SearchParams } from "./base";
import { SearchProvider } from "./base";
import { classifyProviderHttpError, withHardTimeout } from "./utils";
const XAI_RESPONSES_URL = "https://api.x.ai/v1/responses";
const XAI_WEB_SEARCH_MODEL = "grok-4.3";
const DEFAULT_NUM_RESULTS = 10;
const MAX_NUM_RESULTS = 30;
interface XAIUrlCitationAnnotation {
type?: string;
url?: string | null;
title?: string | null;
text?: string | null;
cited_text?: string | null;
}
interface XAIResponseContentPart {
type?: string;
text?: string | null;
output_text?: string | null;
annotations?: XAIUrlCitationAnnotation[] | null;
}
interface XAIResponseOutputItem {
content?: XAIResponseContentPart[] | null;
annotations?: XAIUrlCitationAnnotation[] | null;
}
interface XAIResponsesUsage {
input_tokens?: number;
output_tokens?: number;
total_tokens?: number;
inputTokens?: number;
outputTokens?: number;
totalTokens?: number;
}
interface XAIResponsesResponse {
id?: string;
model?: string;
output_text?: string | null;
output?: XAIResponseOutputItem[] | null;
annotations?: XAIUrlCitationAnnotation[] | null;
citations?: string[] | null;
usage?: XAIResponsesUsage | null;
}
function buildRequestBody(params: SearchParams): Record<string, unknown> {
const body: Record<string, unknown> = {
model: XAI_WEB_SEARCH_MODEL,
input: [
{ role: "system", content: params.systemPrompt },
{ role: "user", content: params.query },
],
tools: [{ type: "web_search" }],
};
if (params.maxOutputTokens !== undefined) {
body.max_output_tokens = params.maxOutputTokens;
}
if (params.temperature !== undefined) {
body.temperature = params.temperature;
}
return body;
}
async function postXAIResponses(
apiKey: string,
params: SearchParams,
body: Record<string, unknown>,
): Promise<Response> {
return (params.fetch ?? fetch)(XAI_RESPONSES_URL, {
method: "POST",
headers: {
"Content-Type": "application/json",
Authorization: `Bearer ${apiKey}`,
},
body: JSON.stringify(body),
signal: withHardTimeout(params.signal),
});
}
function throwXAIResponsesError(status: number, errorText: string): never {
const classified = classifyProviderHttpError("xai", status, errorText);
if (classified) throw classified;
throw new SearchProviderError("xai", `xAI Responses API error (${status}): ${errorText}`, status);
}
async function callXAIResponses(apiKey: string, params: SearchParams): Promise<XAIResponsesResponse> {
const requestBody = buildRequestBody(params);
const response = await postXAIResponses(apiKey, params, requestBody);
if (!response.ok) {
throwXAIResponsesError(response.status, await response.text());
}
return (await response.json()) as XAIResponsesResponse;
}
function addCitationSource(
sources: SearchSource[],
citations: SearchCitation[],
seenUrls: Set<string>,
url: string,
title?: string | null,
citedText?: string | null,
): void {
const trimmedUrl = url.trim();
if (!trimmedUrl || seenUrls.has(trimmedUrl)) return;
seenUrls.add(trimmedUrl);
const sourceTitle = title?.trim() || trimmedUrl;
const sourceSnippet = citedText?.trim() || undefined;
sources.push({
title: sourceTitle,
url: trimmedUrl,
snippet: sourceSnippet,
});
citations.push({
title: sourceTitle,
url: trimmedUrl,
citedText: sourceSnippet,
});
}
function collectAnnotationSources(
annotations: readonly XAIUrlCitationAnnotation[] | null | undefined,
sources: SearchSource[],
citations: SearchCitation[],
seenUrls: Set<string>,
): void {
if (!annotations) return;
for (const annotation of annotations) {
if (annotation.type !== "url_citation" || !annotation.url) continue;
addCitationSource(
sources,
citations,
seenUrls,
annotation.url,
annotation.title,
annotation.cited_text ?? annotation.text,
);
}
}
function parseAnswer(response: XAIResponsesResponse): string | undefined {
const topLevelText = response.output_text?.trim();
if (topLevelText) return topLevelText;
const answerParts: string[] = [];
for (const item of response.output ?? []) {
for (const part of item.content ?? []) {
const text = part.output_text ?? part.text;
if ((part.type === "output_text" || part.type === "text") && text?.trim()) {
answerParts.push(text.trim());
}
}
}
const answer = answerParts.join("\n").trim();
return answer ? answer : undefined;
}
function parseUsage(usage: XAIResponsesUsage | null | undefined): SearchUsage | undefined {
if (!usage) return undefined;
const parsed: SearchUsage = {};
const inputTokens = usage.input_tokens ?? usage.inputTokens;
const outputTokens = usage.output_tokens ?? usage.outputTokens;
const totalTokens = usage.total_tokens ?? usage.totalTokens;
if (typeof inputTokens === "number") parsed.inputTokens = inputTokens;
if (typeof outputTokens === "number") parsed.outputTokens = outputTokens;
if (typeof totalTokens === "number") parsed.totalTokens = totalTokens;
return Object.keys(parsed).length > 0 ? parsed : undefined;
}
function applyResultCap(
sources: SearchSource[],
citations: SearchCitation[],
resultCap: number,
): { sources: SearchSource[]; citations: SearchCitation[] } {
return {
sources: sources.slice(0, resultCap),
citations: citations.slice(0, resultCap),
};
}
function parseResponse(response: XAIResponsesResponse, resultCap: number): SearchResponse {
const sources: SearchSource[] = [];
const citations: SearchCitation[] = [];
const seenUrls = new Set<string>();
collectAnnotationSources(response.annotations, sources, citations, seenUrls);
for (const item of response.output ?? []) {
collectAnnotationSources(item.annotations, sources, citations, seenUrls);
for (const part of item.content ?? []) {
collectAnnotationSources(part.annotations, sources, citations, seenUrls);
}
}
for (const url of response.citations ?? []) {
addCitationSource(sources, citations, seenUrls, url);
}
const limited = applyResultCap(sources, citations, resultCap);
return {
provider: "xai",
answer: parseAnswer(response),
sources: limited.sources,
citations: limited.citations.length > 0 ? limited.citations : undefined,
usage: parseUsage(response.usage),
model: response.model,
requestId: response.id,
authMode: "api_key",
};
}
/** Execute xAI Responses API web search. */
export async function searchXAI(params: SearchParams): Promise<SearchResponse> {
const keyOrResolver: ApiKey = params.authStorage.resolver("xai", {
sessionId: params.sessionId,
});
const resultCap = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS);
const response = await withAuth(keyOrResolver, (key: string) => callXAIResponses(key, params), {
signal: params.signal,
missingKeyMessage: 'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".',
});
return parseResponse(response, resultCap);
}
/** Search provider for xAI web search. */
export class XAIProvider extends SearchProvider {
readonly id = "xai";
readonly label = "xAI";
isAvailable(authStorage: AuthStorage): boolean {
return authStorage.hasAuth("xai");
}
search(params: SearchParams): Promise<SearchResponse> {
return searchXAI(params);
}
}
@@ -30,16 +30,20 @@ export const SEARCH_PROVIDER_OPTIONS = [
label: "OpenAI",
description: "OpenAI's native web_search (uses ChatGPT OAuth via /login openai-codex)",
},
{ value: "xai", label: "xAI", description: "Grok web search via xAI Responses API (requires XAI_API_KEY)" },
{ value: "zai", label: "Z.AI", description: "Calls Z.AI webSearchPrime MCP" },
{ value: "exa", label: "Exa", description: "Uses Exa API when EXA_API_KEY is set; falls back to Exa MCP" },
{ value: "tinyfish", label: "TinyFish", description: "Requires TINYFISH_API_KEY" },
{ value: "jina", label: "Jina", description: "Requires JINA_API_KEY" },
{ value: "kagi", label: "Kagi", description: "Requires KAGI_API_KEY and Kagi Search API beta access" },
{ value: "tavily", label: "Tavily", description: "Requires TAVILY_API_KEY" },
{ value: "firecrawl", label: "Firecrawl", description: "Requires FIRECRAWL_API_KEY" },
{ value: "brave", label: "Brave", description: "Requires BRAVE_API_KEY" },
{ value: "kimi", label: "Kimi", description: "Requires MOONSHOT_SEARCH_API_KEY or MOONSHOT_API_KEY" },
{ value: "parallel", label: "Parallel", description: "Requires PARALLEL_API_KEY" },
{ value: "synthetic", label: "Synthetic", description: "Requires SYNTHETIC_API_KEY" },
{ value: "searxng", label: "SearXNG", description: "Requires SEARXNG_ENDPOINT or searxng.endpoint" },
{ value: "duckduckgo", label: "DuckDuckGo", description: "Uses DuckDuckGo Instant Answer API (no API key)" },
] as const;
/** Supported web search providers (every option except `auto`). */
@@ -81,7 +85,7 @@ export interface SearchSource {
author?: string;
}
/** Citation with text reference (anthropic, perplexity) */
/** Citation with text reference (LLM-mediated providers) */
export interface SearchCitation {
url: string;
title: string;
@@ -101,7 +105,7 @@ export interface SearchUsage {
/** Unified response across providers */
export interface SearchResponse {
provider: SearchProviderId | "none";
/** Synthesized answer text (anthropic, perplexity) */
/** Synthesized answer text (LLM-mediated providers) */
answer?: string;
/** Search result sources */
sources: SearchSource[];
@@ -34,6 +34,21 @@ describe("legacy pi compat compiled-mode subpath overrides (issue #3442)", () =>
expect(BUNDLED_PI_REGISTRY_KEYS.has("@oh-my-pi/pi-ai/oauth/openai-codex")).toBe(true);
});
it("expands web search provider wildcard exports for compiled plugin imports", () => {
const overrides = __buildLegacyPiPackageRootOverrides(true);
const providerKeys = [
"@oh-my-pi/pi-coding-agent/web/search/providers/xai",
"@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish",
"@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl",
"@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo",
] as const;
for (const key of providerKeys) {
expect(BUNDLED_PI_REGISTRY_KEYS.has(key)).toBe(true);
expect(overrides[key]).toBe(`omp-legacy-pi-bundled:${key}`);
}
});
it("does not enumerate root catch-all wildcards (./* / ./*.js)", () => {
// Root `./*` / `./*.js` patterns would static-import top-level files
// like the package's own `cli.ts` and explode the bundle through the
@@ -0,0 +1,188 @@
import { describe, expect, it } from "bun:test";
import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai";
import { searchDuckDuckGo } from "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo";
import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types";
const fakeAuthStorage = {
async getApiKey() {
throw new Error("DuckDuckGo must not request API keys");
},
resolver() {
throw new Error("DuckDuckGo must not request credential resolvers");
},
hasAuth() {
throw new Error("DuckDuckGo search must not check auth");
},
} as unknown as AuthStorage;
function makeParams(query: string, fetch: FetchImpl) {
return {
query,
authStorage: fakeAuthStorage,
systemPrompt: "DuckDuckGo test prompt",
fetch,
} as const;
}
describe("DuckDuckGo web search provider", () => {
it("calls the official Instant Answer API with unauthenticated JSON query params", async () => {
let capturedUrl: string | null = null;
let capturedInit: RequestInit | undefined;
const fetchMock: FetchImpl = (input, init) => {
capturedUrl = typeof input === "string" ? input : input.toString();
capturedInit = init;
return Promise.resolve(
new Response(JSON.stringify({ AbstractText: "Duck answer", Results: [] }), {
status: 200,
headers: { "Content-Type": "application/json" },
}),
);
};
await searchDuckDuckGo(makeParams("instant answer", fetchMock));
expect(capturedUrl).not.toBeNull();
const url = new URL(capturedUrl ?? "");
expect(`${url.origin}${url.pathname}`).toBe("https://api.duckduckgo.com/");
expect(url.searchParams.get("q")).toBe("instant answer");
expect(url.searchParams.get("format")).toBe("json");
expect(url.searchParams.get("no_redirect")).toBe("1");
expect(url.searchParams.get("no_html")).toBe("1");
expect(url.searchParams.get("skip_disambig")).toBe("1");
expect(url.searchParams.get("t")).toBe("oh-my-pi");
expect(capturedInit?.method).toBe("GET");
expect(capturedInit?.headers).toBeUndefined();
});
it("uses AbstractText as the answer and flattens abstract, result, and nested related topics within the local limit", async () => {
const fetchMock: FetchImpl = () =>
Promise.resolve(
new Response(
JSON.stringify({
AbstractText: " DuckDuckGo <b>abstract</b> &amp; answer ",
AbstractURL: " https://example.com/abstract ",
AbstractSource: " Example Abstract Source ",
Heading: "Example Heading",
Results: [
{
FirstURL: "https://example.com/result",
Text: "Result <i>snippet</i>",
},
],
RelatedTopics: [
{
FirstURL: "https://example.com/related",
Text: "Related topic",
},
{
Topics: [
{
FirstURL: "https://example.com/nested",
Text: "Nested related topic",
},
],
},
{
FirstURL: "https://example.com/omitted-by-limit",
Text: "Should be omitted by local limit",
},
],
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
),
);
const response = await searchDuckDuckGo({ ...makeParams("duck mapping", fetchMock), numSearchResults: 4 });
expect(response).toMatchObject({
provider: "duckduckgo",
answer: "DuckDuckGo abstract & answer",
sources: [
{
title: "Example Abstract Source",
url: "https://example.com/abstract",
snippet: "DuckDuckGo abstract & answer",
},
{
title: "Result snippet",
url: "https://example.com/result",
snippet: "Result snippet",
},
{
title: "Related topic",
url: "https://example.com/related",
snippet: "Related topic",
},
{
title: "Nested related topic",
url: "https://example.com/nested",
snippet: "Nested related topic",
},
],
});
expect(response.sources).toHaveLength(4);
expect(response.sources.some(source => source.url === "https://example.com/omitted-by-limit")).toBe(false);
});
it("clamps oversized local result limits to DuckDuckGo's provider maximum", async () => {
const fetchMock: FetchImpl = () =>
Promise.resolve(
new Response(
JSON.stringify({
RelatedTopics: Array.from({ length: 25 }, (_value, index) => ({
FirstURL: `https://example.com/topic-${index}`,
Text: `Topic ${index}`,
})),
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
),
);
const response = await searchDuckDuckGo({ ...makeParams("duck clamp", fetchMock), numSearchResults: 999 });
expect(response.sources).toHaveLength(20);
expect(response.sources.at(0)?.url).toBe("https://example.com/topic-0");
expect(response.sources.at(-1)?.url).toBe("https://example.com/topic-19");
expect(response.sources.some(source => source.url === "https://example.com/topic-20")).toBe(false);
});
it.each([
["Answer", { Answer: " Direct answer " }, "Direct answer"],
["Definition", { Definition: " Definition answer " }, "Definition answer"],
] as const)("falls back to %s when AbstractText is absent", async (_field, payload, expectedAnswer) => {
const fetchMock: FetchImpl = () =>
Promise.resolve(
new Response(JSON.stringify(payload), {
status: 200,
headers: { "Content-Type": "application/json" },
}),
);
const response = await searchDuckDuckGo(makeParams("fallback answer", fetchMock));
expect(response).toMatchObject({
provider: "duckduckgo",
answer: expectedAnswer,
});
});
it("throws a provider-tagged SearchProviderError for HTTP failures", async () => {
const fetchMock: FetchImpl = () =>
Promise.resolve(
new Response("upstream unavailable", {
status: 503,
}),
);
try {
await searchDuckDuckGo(makeParams("http failure", fetchMock));
expect.unreachable("DuckDuckGo HTTP failure should reject");
} catch (error) {
expect(error).toBeInstanceOf(SearchProviderError);
expect(error).toMatchObject({
provider: "duckduckgo",
status: 503,
message: "DuckDuckGo API error (503): upstream unavailable",
});
}
});
});
@@ -0,0 +1,138 @@
import { describe, expect, it } from "bun:test";
import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai";
import { searchFirecrawl } from "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl";
import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types";
const TEST_KEY = "test-firecrawl-key";
function makeAuthStorage(apiKey: string | undefined): AuthStorage {
return {
resolver(provider: string, options?: { sessionId?: string }) {
expect(provider).toBe("firecrawl");
expect(options?.sessionId).toBe("session-firecrawl-test");
return async () => apiKey;
},
hasAuth(provider: string) {
return provider === "firecrawl" && Boolean(apiKey);
},
} as unknown as AuthStorage;
}
function makeParams(query: string, authStorage: AuthStorage = makeAuthStorage(TEST_KEY)) {
return {
query,
authStorage,
systemPrompt: "Firecrawl test prompt",
sessionId: "session-firecrawl-test",
} as const;
}
function getHeader(headers: RequestInit["headers"] | undefined, name: string): string | null {
if (!headers) return null;
if (headers instanceof Headers) return headers.get(name);
if (Array.isArray(headers)) {
return headers.find(([key]) => key.toLowerCase() === name.toLowerCase())?.[1] ?? null;
}
const record = headers as Record<string, string>;
return record[name] ?? record[name.toLowerCase()] ?? null;
}
describe("Firecrawl web search provider", () => {
it("sends the Firecrawl POST request and maps web results", async () => {
const captured: { url?: string; init?: RequestInit; body?: unknown } = {};
const fetchMock: FetchImpl = async (input, init) => {
captured.url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url;
captured.init = init;
captured.body = JSON.parse(String(init?.body ?? "null")) as unknown;
return new Response(
JSON.stringify({
id: "firecrawl-request-123",
data: {
web: [
{
title: "Firecrawl result one",
url: "https://example.com/one",
description: "Description snippet",
markdown: "Ignored markdown",
},
{
title: "Firecrawl result two",
url: "https://example.com/two",
description: null,
markdown: "Markdown fallback snippet",
},
],
},
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
};
const response = await searchFirecrawl({
...makeParams("firecrawl query"),
numSearchResults: 2,
recency: "month",
fetch: fetchMock,
});
expect(captured.url).toBe("https://api.firecrawl.dev/v2/search");
expect(captured.init?.method).toBe("POST");
expect(getHeader(captured.init?.headers, "Authorization")).toBe(`Bearer ${TEST_KEY}`);
expect(getHeader(captured.init?.headers, "Content-Type")).toBe("application/json");
expect(captured.body).toEqual({
query: "firecrawl query",
limit: 2,
sources: [{ type: "web" }],
tbs: "qdr:m",
});
expect(response).toEqual({
provider: "firecrawl",
sources: [
{
title: "Firecrawl result one",
url: "https://example.com/one",
snippet: "Description snippet",
},
{
title: "Firecrawl result two",
url: "https://example.com/two",
snippet: "Markdown fallback snippet",
},
],
requestId: "firecrawl-request-123",
authMode: "api_key",
});
});
it.each([
[401, "firecrawl: 401 unauthorized"],
[402, "firecrawl: 402 credits exhausted"],
] as const)("maps HTTP %d to a SearchProviderError", async (status, message) => {
const fetchMock: FetchImpl = async () => new Response("upstream rejected", { status });
try {
await searchFirecrawl({ ...makeParams("bad auth"), fetch: fetchMock });
expect.unreachable("expected searchFirecrawl to throw");
} catch (error) {
expect(error).toBeInstanceOf(SearchProviderError);
expect(error).toMatchObject({ provider: "firecrawl", status, message });
}
});
it("throws a clear error when Firecrawl credentials are missing", async () => {
const fetchMock: FetchImpl = async () => {
throw new Error("fetch should not be called without credentials");
};
try {
await searchFirecrawl({ ...makeParams("missing creds", makeAuthStorage(undefined)), fetch: fetchMock });
expect.unreachable("expected searchFirecrawl to throw");
} catch (error) {
expect(error).toBeInstanceOf(Error);
expect((error as Error).message).toBe(
'Firecrawl credentials not found. Set FIRECRAWL_API_KEY or configure an API key for provider "firecrawl".',
);
}
});
});
@@ -0,0 +1,308 @@
import { describe, expect, it } from "bun:test";
import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai";
import { searchTinyFish } from "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish";
import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types";
const TEST_KEY = "test-tinyfish-key";
const UNSUPPORTED_TINYFISH_COUNT_PARAMS = ["limit", "num_results", "count", "size", "max_results"] as const;
function makeAuthStorage(apiKey: string | undefined): AuthStorage {
return {
resolver(provider: string, options?: { sessionId?: string }) {
expect(provider).toBe("tinyfish");
expect(options?.sessionId).toBe("session-tinyfish-test");
return async () => apiKey;
},
hasAuth(provider: string) {
return provider === "tinyfish" && Boolean(apiKey);
},
} as unknown as AuthStorage;
}
function makeParams(query: string, authStorage: AuthStorage = makeAuthStorage(TEST_KEY)) {
return {
query,
authStorage,
systemPrompt: "TinyFish test prompt",
sessionId: "session-tinyfish-test",
} as const;
}
function getHeader(headers: RequestInit["headers"] | undefined, name: string): string | null {
if (!headers) return null;
if (headers instanceof Headers) return headers.get(name);
if (Array.isArray(headers)) {
return headers.find(([key]) => key.toLowerCase() === name.toLowerCase())?.[1] ?? null;
}
const record = headers as Record<string, string>;
return record[name] ?? record[name.toLowerCase()] ?? null;
}
interface TinyFishMockResult {
title: string;
url: string | null;
snippet: string;
site_name?: string;
}
function tinyFishResults(prefix: string, count: number, start = 0): TinyFishMockResult[] {
return Array.from({ length: count }, (_, offset) => {
const index = start + offset;
return {
title: `${prefix} result ${index}`,
url: `https://example.com/${prefix}-${index}`,
snippet: `${prefix} snippet ${index}`,
site_name: index === 0 ? "Example Site" : undefined,
};
});
}
function tinyFishPage(results: TinyFishMockResult[], page = 0, totalResults = results.length) {
return { results, total_results: totalResults, page };
}
function expectOnlyDocumentedTinyFishParams(url: URL, expectedParams: readonly string[]): void {
expect([...url.searchParams.keys()].sort()).toEqual([...expectedParams].sort());
for (const unsupportedParam of UNSUPPORTED_TINYFISH_COUNT_PARAMS) {
expect(url.searchParams.has(unsupportedParam)).toBe(false);
}
}
describe("TinyFish web search provider", () => {
it("documents TinyFish's absent result-count parameter and applies numSearchResults across pages", async () => {
const captured: { url: URL; init?: RequestInit }[] = [];
const pages = new Map([
["0", tinyFishResults("tinyfish", 10)],
["1", tinyFishResults("tinyfish", 10, 10)],
]);
const fetchMock: FetchImpl = async (input, init) => {
const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url);
captured.push({ url, init });
const page = Number(url.searchParams.get("page") ?? 0);
return new Response(JSON.stringify(tinyFishPage(pages.get(String(page)) ?? [], page, 20)), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const response = await searchTinyFish({
...makeParams("fresh fish"),
numSearchResults: 12,
recency: "week",
fetch: fetchMock,
});
expect(captured).toHaveLength(2);
const [firstRequest, secondRequest] = captured;
const endpoint = `${firstRequest.url.origin}${firstRequest.url.pathname === "/" ? "" : firstRequest.url.pathname}`;
expect(endpoint).toBe("https://api.search.tinyfish.ai");
expect(firstRequest.init?.method ?? "GET").toBe("GET");
expect(getHeader(firstRequest.init?.headers, "X-API-Key")).toBe(TEST_KEY);
expect(firstRequest.url.searchParams.get("query")).toBe("fresh fish");
expect(firstRequest.url.searchParams.get("recency_minutes")).toBe("10080");
expect(firstRequest.url.searchParams.get("page")).toBe("0");
expect(secondRequest.url.searchParams.get("query")).toBe("fresh fish");
expect(secondRequest.url.searchParams.get("recency_minutes")).toBe("10080");
expect(secondRequest.url.searchParams.get("page")).toBe("1");
// TinyFish Search docs expose no result-count parameter; unified counts are applied after paginated responses.
expectOnlyDocumentedTinyFishParams(firstRequest.url, ["query", "recency_minutes", "page"]);
expectOnlyDocumentedTinyFishParams(secondRequest.url, ["query", "recency_minutes", "page"]);
expect(response.provider).toBe("tinyfish");
expect(response.authMode).toBe("api_key");
expect(response.sources).toHaveLength(12);
expect(response.sources[0]).toEqual({
title: "tinyfish result 0",
url: "https://example.com/tinyfish-0",
snippet: "tinyfish snippet 0",
author: "Example Site",
});
expect(response.sources.at(-1)).toEqual({
title: "tinyfish result 11",
url: "https://example.com/tinyfish-11",
snippet: "tinyfish snippet 11",
author: undefined,
});
expect(response.sources.some(source => source.url === "https://example.com/tinyfish-12")).toBe(false);
});
it("requests two documented TinyFish pages for limit 20 without unsupported count-like params", async () => {
const captured: URL[] = [];
const pages = new Map([
["0", tinyFishResults("limit", 10)],
["1", tinyFishResults("limit", 10, 10)],
]);
const fetchMock: FetchImpl = async input => {
const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url);
captured.push(url);
const page = Number(url.searchParams.get("page") ?? 0);
return new Response(JSON.stringify(tinyFishPage(pages.get(String(page)) ?? [], page, 20)), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const response = await searchTinyFish({
...makeParams("limit fish"),
limit: 20,
recency: "day",
fetch: fetchMock,
});
expect(captured).toHaveLength(2);
expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0", "1"]);
for (const url of captured) {
expect(url.searchParams.get("query")).toBe("limit fish");
expect(url.searchParams.get("recency_minutes")).toBe("1440");
expectOnlyDocumentedTinyFishParams(url, ["query", "recency_minutes", "page"]);
}
expect(response.sources).toHaveLength(20);
expect(response.sources.at(-1)?.url).toBe("https://example.com/limit-19");
});
it("requests page 1 when page 0 has 10 raw results but fewer usable sources", async () => {
const captured: URL[] = [];
const firstPageResults = tinyFishResults("raw-page", 10);
firstPageResults[0] = { ...firstPageResults[0], url: null };
const pages = new Map([
["0", firstPageResults],
["1", tinyFishResults("raw-page", 10, 10)],
]);
const fetchMock: FetchImpl = async input => {
const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url);
captured.push(url);
const page = Number(url.searchParams.get("page") ?? 0);
return new Response(JSON.stringify(tinyFishPage(pages.get(String(page)) ?? [], page, 20)), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const response = await searchTinyFish({ ...makeParams("raw page fish"), limit: 11, fetch: fetchMock });
expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0", "1"]);
expect(response.sources).toHaveLength(11);
expect(response.sources[0]?.url).toBe("https://example.com/raw-page-1");
expect(response.sources.at(-1)?.url).toBe("https://example.com/raw-page-11");
});
it("stops early for limit 20 when page 0 returns fewer than 10 raw results", async () => {
const captured: URL[] = [];
const fetchMock: FetchImpl = async input => {
const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url);
captured.push(url);
return new Response(JSON.stringify(tinyFishPage(tinyFishResults("short-page", 9), 0, 9)), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const response = await searchTinyFish({ ...makeParams("short page fish"), limit: 20, fetch: fetchMock });
expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0"]);
expect(response.sources).toHaveLength(9);
expect(response.sources.at(-1)?.url).toBe("https://example.com/short-page-8");
});
it("does not request a second page for the default 10-result page", async () => {
const captured: URL[] = [];
const fetchMock: FetchImpl = async input => {
const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url);
captured.push(url);
return new Response(JSON.stringify(tinyFishPage(tinyFishResults("default", 10), 0, 10)), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const response = await searchTinyFish({ ...makeParams("default fish"), fetch: fetchMock });
expect(captured).toHaveLength(1);
expect(captured[0].searchParams.get("query")).toBe("default fish");
expect(captured[0].searchParams.get("page")).toBe("0");
expectOnlyDocumentedTinyFishParams(captured[0], ["query", "page"]);
expect(response.sources).toHaveLength(10);
});
it("does not request a second page when the local limit is 10 or below", async () => {
const captured: URL[] = [];
const fetchMock: FetchImpl = async input => {
const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url);
captured.push(url);
return new Response(JSON.stringify(tinyFishPage(tinyFishResults("small-limit", 10), 0, 10)), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const response = await searchTinyFish({ ...makeParams("small limit fish"), limit: 7, fetch: fetchMock });
expect(captured).toHaveLength(1);
expect(captured[0].searchParams.get("query")).toBe("small limit fish");
expect(captured[0].searchParams.get("page")).toBe("0");
expectOnlyDocumentedTinyFishParams(captured[0], ["query", "page"]);
expect(response.sources).toHaveLength(7);
expect(response.sources.at(-1)?.url).toBe("https://example.com/small-limit-6");
});
it("propagates second-page HTTP errors", async () => {
const captured: URL[] = [];
const fetchMock: FetchImpl = async input => {
const url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url);
captured.push(url);
if (url.searchParams.get("page") === "1") {
return new Response("upstream rejected page 1", { status: 402 });
}
return new Response(JSON.stringify(tinyFishPage(tinyFishResults("page-error", 10), 0, 20)), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
try {
await searchTinyFish({ ...makeParams("page error fish"), limit: 20, fetch: fetchMock });
expect.unreachable("expected searchTinyFish to throw");
} catch (error) {
expect(captured.map(url => url.searchParams.get("page"))).toEqual(["0", "1"]);
expect(error).toBeInstanceOf(SearchProviderError);
expect(error).toMatchObject({ provider: "tinyfish", status: 402, message: "tinyfish: 402 credits exhausted" });
}
});
it.each([
[401, "tinyfish: 401 unauthorized"],
[402, "tinyfish: 402 credits exhausted"],
] as const)("maps HTTP %d to a SearchProviderError", async (status, message) => {
const fetchMock: FetchImpl = async () => new Response("upstream rejected", { status });
try {
await searchTinyFish({ ...makeParams("bad auth"), fetch: fetchMock });
expect.unreachable("expected searchTinyFish to throw");
} catch (error) {
expect(error).toBeInstanceOf(SearchProviderError);
expect(error).toMatchObject({ provider: "tinyfish", status, message });
}
});
it("throws a clear error when TinyFish credentials are missing", async () => {
const fetchMock: FetchImpl = async () => {
throw new Error("fetch should not be called without credentials");
};
try {
await searchTinyFish({ ...makeParams("missing creds", makeAuthStorage(undefined)), fetch: fetchMock });
expect.unreachable("expected searchTinyFish to throw");
} catch (error) {
expect(error).toBeInstanceOf(Error);
expect((error as Error).message).toBe(
'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".',
);
}
});
});
@@ -0,0 +1,485 @@
import { afterEach, describe, expect, it, vi } from "bun:test";
import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai";
import { searchXAI } from "@oh-my-pi/pi-coding-agent/web/search/providers/xai";
import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types";
type CapturedRequest = {
url: string;
method: string | undefined;
headers: RequestInit["headers"];
body: Record<string, unknown> | null;
};
function makeAuthStorage(apiKey: string | undefined) {
return {
resolver(provider: string, options?: { sessionId?: string }) {
expect(provider).toBe("xai");
expect(options?.sessionId).toBe("session-xai-test");
return async () => apiKey;
},
hasAuth(provider: string) {
return provider === "xai" && Boolean(apiKey);
},
} as unknown as AuthStorage;
}
function makeParams(fetch: FetchImpl, authStorage: AuthStorage = makeAuthStorage("test-xai-key")) {
return {
query: "latest xAI web search",
systemPrompt: "Use web search for current xAI facts.",
authStorage,
fetch,
sessionId: "session-xai-test",
} as const;
}
function captureFetch(responseBody: Record<string, unknown> | string, status = 200) {
const capturedRequests: CapturedRequest[] = [];
const fetchMock: FetchImpl = (input, init) => {
capturedRequests.push({
url: typeof input === "string" ? input : input.toString(),
method: init?.method,
headers: init?.headers,
body: init?.body ? (JSON.parse(String(init.body)) as Record<string, unknown>) : null,
});
return Promise.resolve(
new Response(typeof responseBody === "string" ? responseBody : JSON.stringify(responseBody), {
status,
headers: { "Content-Type": "application/json" },
}),
);
};
return {
fetchMock,
capturedRequests,
get capturedRequest() {
return capturedRequests.at(-1) ?? null;
},
};
}
function citationUrls(prefix: string, count: number): string[] {
return Array.from({ length: count }, (_, index) => `https://example.com/${prefix}-${index + 1}`);
}
describe("xAI web search provider", () => {
afterEach(() => {
vi.restoreAllMocks();
});
it("POSTs the Responses API with bearer auth and xAI web_search tool payload", async () => {
const capture = captureFetch({ id: "resp_request", model: "grok-4.3", output_text: "xAI answer" });
await searchXAI({
...makeParams(capture.fetchMock),
maxOutputTokens: 512,
temperature: 0.2,
});
expect(capture.capturedRequest).not.toBeNull();
expect(capture.capturedRequest?.url).toBe("https://api.x.ai/v1/responses");
expect(capture.capturedRequest?.method).toBe("POST");
expect(capture.capturedRequest?.headers).toMatchObject({
"Content-Type": "application/json",
Authorization: "Bearer test-xai-key",
});
expect(capture.capturedRequest?.body).toMatchObject({
model: "grok-4.3",
input: [
{ role: "system", content: "Use web search for current xAI facts." },
{ role: "user", content: "latest xAI web search" },
],
tools: [{ type: "web_search" }],
max_output_tokens: 512,
temperature: 0.2,
});
expect(capture.capturedRequest?.body?.tools).toEqual([{ type: "web_search" }]);
expect(capture.capturedRequest?.body).not.toHaveProperty("search_parameters");
});
it("omits search_parameters for minimal web_search requests", async () => {
const capture = captureFetch({ id: "resp_minimal", model: "grok-4.3", output_text: "minimal xAI answer" });
await searchXAI(makeParams(capture.fetchMock));
expect(capture.capturedRequest).not.toBeNull();
const body = capture.capturedRequest?.body;
expect(body?.tools).toEqual([{ type: "web_search" }]);
expect(body).not.toHaveProperty("search_parameters");
});
it.each([
["limit", { limit: 6 }],
["numSearchResults", { numSearchResults: 7 }],
["limit and numSearchResults", { limit: 2, numSearchResults: 50 }],
["recency", { recency: "week" }],
["limit, numSearchResults, and recency", { limit: 0, numSearchResults: 30, recency: "day" }],
] as const)("does not map %s to xAI search_parameters", async (_caseName, searchParams) => {
const capture = captureFetch({ id: "resp_agent_tools", model: "grok-4.3", output_text: "xAI answer" });
await searchXAI({
...makeParams(capture.fetchMock),
...searchParams,
});
expect(capture.capturedRequest).not.toBeNull();
const body = capture.capturedRequest?.body;
expect(body?.tools).toEqual([{ type: "web_search" }]);
expect(body).not.toHaveProperty("search_parameters");
expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]);
});
it("rejects deprecated live-search 410 responses without retrying", async () => {
const capture = captureFetch("Live search is deprecated. Please use the Agent Tools API.", 410);
try {
await searchXAI({
...makeParams(capture.fetchMock),
limit: 2,
numSearchResults: 5,
recency: "week",
});
expect.unreachable("xAI HTTP 410 deprecation failure should reject");
} catch (error) {
expect(error).toBeInstanceOf(SearchProviderError);
expect(error).toMatchObject({
provider: "xai",
status: 410,
message: "xAI Responses API error (410): Live search is deprecated. Please use the Agent Tools API.",
});
}
expect(capture.capturedRequests).toHaveLength(1);
const body = capture.capturedRequests[0]?.body;
expect(body?.tools).toEqual([{ type: "web_search" }]);
expect(body).not.toHaveProperty("search_parameters");
});
it("maps output_text, URL citation annotations, top-level citations, id, model, usage, and auth mode", async () => {
const capture = captureFetch({
id: "resp_xai_123",
model: "grok-4.3",
output_text: "Top-level xAI answer",
annotations: [
{
type: "url_citation",
url: "https://example.com/top-annotation",
title: "Top Annotation",
text: "Top annotation text",
},
],
output: [
{
type: "message",
annotations: [
{
type: "url_citation",
url: "https://example.com/item-annotation",
title: "Item Annotation",
cited_text: "Item annotation text",
},
],
content: [
{
type: "output_text",
text: "Ignored because output_text wins",
annotations: [
{
type: "url_citation",
url: "https://example.com/annotated",
title: "Annotated Source",
cited_text: "Annotated cited text",
},
],
},
],
},
],
citations: ["https://example.com/top-level-citation"],
usage: {
input_tokens: 12,
output_tokens: 8,
total_tokens: 20,
},
});
const response = await searchXAI(makeParams(capture.fetchMock));
expect(response).toMatchObject({
provider: "xai",
answer: "Top-level xAI answer",
requestId: "resp_xai_123",
model: "grok-4.3",
authMode: "api_key",
usage: {
inputTokens: 12,
outputTokens: 8,
totalTokens: 20,
},
sources: [
{
title: "Top Annotation",
url: "https://example.com/top-annotation",
snippet: "Top annotation text",
},
{
title: "Item Annotation",
url: "https://example.com/item-annotation",
snippet: "Item annotation text",
},
{
title: "Annotated Source",
url: "https://example.com/annotated",
snippet: "Annotated cited text",
},
{
title: "https://example.com/top-level-citation",
url: "https://example.com/top-level-citation",
},
],
citations: [
{
title: "Top Annotation",
url: "https://example.com/top-annotation",
citedText: "Top annotation text",
},
{
title: "Item Annotation",
url: "https://example.com/item-annotation",
citedText: "Item annotation text",
},
{
title: "Annotated Source",
url: "https://example.com/annotated",
citedText: "Annotated cited text",
},
{
title: "https://example.com/top-level-citation",
url: "https://example.com/top-level-citation",
},
],
});
});
it("defaults xAI local cap to 10 sources and citations when no count is requested", async () => {
const urls = citationUrls("default-cap", 12);
const capture = captureFetch({
id: "resp_default_cap",
model: "grok-4.3",
output_text: "Default capped xAI answer",
citations: urls,
});
const response = await searchXAI(makeParams(capture.fetchMock));
const expectedUrls = urls.slice(0, 10);
expect(response.sources).toHaveLength(10);
expect(response.citations).toHaveLength(10);
expect(response.sources.map(source => source.url)).toEqual(expectedUrls);
expect(response.citations?.map(citation => citation.url)).toEqual(expectedUrls);
expect(capture.capturedRequest).not.toBeNull();
const body = capture.capturedRequest?.body;
expect(body?.tools).toEqual([{ type: "web_search" }]);
expect(body).not.toHaveProperty("search_parameters");
expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]);
});
it("clamps oversized xAI local cap requests to 30 sources and citations", async () => {
const urls = citationUrls("max-cap", 35);
const capture = captureFetch({
id: "resp_max_cap",
model: "grok-4.3",
output_text: "Max capped xAI answer",
citations: urls,
});
const response = await searchXAI({
...makeParams(capture.fetchMock),
numSearchResults: 99,
});
const expectedUrls = urls.slice(0, 30);
expect(response.sources).toHaveLength(30);
expect(response.citations).toHaveLength(30);
expect(response.sources.map(source => source.url)).toEqual(expectedUrls);
expect(response.citations?.map(citation => citation.url)).toEqual(expectedUrls);
expect(capture.capturedRequest).not.toBeNull();
const body = capture.capturedRequest?.body;
expect(body?.tools).toEqual([{ type: "web_search" }]);
expect(body).not.toHaveProperty("search_parameters");
expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]);
});
it("caps parsed sources and citations locally without changing Agent Tools request shape", async () => {
const capture = captureFetch({
id: "resp_local_cap",
model: "grok-4.3",
output_text: "Capped xAI answer",
annotations: [
{
type: "url_citation",
url: "https://example.com/annotation-1",
title: "Annotation 1",
text: "Annotation 1 text",
},
],
output: [
{
annotations: [
{
type: "url_citation",
url: "https://example.com/annotation-2",
title: "Annotation 2",
cited_text: "Annotation 2 text",
},
],
content: [
{
type: "output_text",
text: "Ignored because output_text wins",
annotations: [
{
type: "url_citation",
url: "https://example.com/annotation-3",
title: "Annotation 3",
cited_text: "Annotation 3 text",
},
],
},
],
},
],
citations: ["https://example.com/top-level-4", "https://example.com/top-level-5"],
});
const response = await searchXAI({
...makeParams(capture.fetchMock),
limit: 4,
});
expect(response.sources).toHaveLength(4);
expect(response.citations).toHaveLength(4);
expect(response.sources.map(source => source.url)).toEqual([
"https://example.com/annotation-1",
"https://example.com/annotation-2",
"https://example.com/annotation-3",
"https://example.com/top-level-4",
]);
expect(response.citations?.map(citation => citation.url)).toEqual([
"https://example.com/annotation-1",
"https://example.com/annotation-2",
"https://example.com/annotation-3",
"https://example.com/top-level-4",
]);
expect(capture.capturedRequest).not.toBeNull();
const body = capture.capturedRequest?.body;
expect(body?.tools).toEqual([{ type: "web_search" }]);
expect(body).not.toHaveProperty("search_parameters");
expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]);
});
it("uses numSearchResults before limit for the local xAI output cap", async () => {
const capture = captureFetch({
id: "resp_num_search_results_cap",
model: "grok-4.3",
output_text: "numSearchResults capped xAI answer",
annotations: [
{
type: "url_citation",
url: "https://example.com/precedence-1",
title: "Precedence 1",
},
],
citations: [
"https://example.com/precedence-2",
"https://example.com/precedence-3",
"https://example.com/precedence-4",
],
});
const response = await searchXAI({
...makeParams(capture.fetchMock),
limit: 1,
numSearchResults: 3,
});
expect(response.sources).toHaveLength(3);
expect(response.citations).toHaveLength(3);
expect(response.sources.map(source => source.url)).toEqual([
"https://example.com/precedence-1",
"https://example.com/precedence-2",
"https://example.com/precedence-3",
]);
expect(response.citations?.map(citation => citation.url)).toEqual([
"https://example.com/precedence-1",
"https://example.com/precedence-2",
"https://example.com/precedence-3",
]);
expect(capture.capturedRequest).not.toBeNull();
const body = capture.capturedRequest?.body;
expect(body?.tools).toEqual([{ type: "web_search" }]);
expect(body).not.toHaveProperty("search_parameters");
expect(Object.keys(body ?? {}).sort()).toEqual(["input", "model", "tools"]);
});
it("falls back to output content parts when output_text is absent", async () => {
const capture = captureFetch({
id: "resp_content_parts",
model: "grok-4.3",
output: [
{
content: [
{ type: "output_text", text: "First content part" },
{ type: "text", output_text: "Second content part" },
],
},
],
});
const response = await searchXAI(makeParams(capture.fetchMock));
expect(response).toMatchObject({
answer: "First content part\nSecond content part",
});
});
it.each([
[401, "xai: 401 unauthorized"],
[402, "xai: 402 credits exhausted"],
] as const)("maps HTTP %s failures to SearchProviderError", async (status, message) => {
const fetchMock: FetchImpl = () =>
Promise.resolve(
new Response(JSON.stringify({ error: "request failed" }), {
status,
headers: { "Content-Type": "application/json" },
}),
);
try {
await searchXAI(makeParams(fetchMock));
expect.unreachable(`xAI HTTP ${status} failure should reject`);
} catch (error) {
expect(error).toBeInstanceOf(SearchProviderError);
expect(error).toMatchObject({
provider: "xai",
status,
message,
});
}
});
it("throws a clear missing-key error before fetch when credentials are unavailable", async () => {
const fetchMock = vi.fn(() => Promise.resolve(new Response("{}", { status: 200 }))) as unknown as FetchImpl;
try {
await searchXAI(makeParams(fetchMock, makeAuthStorage(undefined)));
expect.unreachable("missing xAI credentials should reject");
} catch (error) {
expect(error).toBeInstanceOf(Error);
expect(error).toHaveProperty(
"message",
'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".',
);
}
expect(fetchMock).not.toHaveBeenCalled();
});
});