diff --git a/README.md b/README.md index ab6e6682c..15e7a878e 100644 --- a/README.md +++ b/README.md @@ -151,7 +151,7 @@ _[Watch the capture ↗](https://omp.sh/clips/collab.mp4)_ ### 08 · Read a pdf on arxiv, why not? -web_search chains fourteen ranked providers and hands whatever URLs it finds straight to read. Arxiv PDFs, GitHub pages, Stack Overflow threads come back as structured markdown with anchors intact — the same tool surface you use on local files. Cite, follow, quote, never lose where you came from. +web_search chains eighteen ranked providers and hands whatever URLs it finds straight to read. Arxiv PDFs, GitHub pages, Stack Overflow threads come back as structured markdown with anchors intact — the same tool surface you use on local files. Cite, follow, quote, never lose where you came from. ![omp TUI: web_search returns 10 ranked Perplexity sources for inference-time compute scaling, the agent picks an arxiv paper, calls read https://arxiv.org/pdf/2604.10739v1, and summarizes the paper's headline result with real numbers.](https://omp.sh/clips/web-poster.webp) @@ -309,31 +309,35 @@ Ollama `local` · Ollama Cloud · LM Studio `local` · llama.cpp `local` · vLLM Full provider & routing reference at [omp.sh/docs/providers](https://omp.sh/docs/providers). -## Fourteen backends. _One tool the agent already knows_. +## Eighteen backends. _One tool the agent already knows_. -`web_search` is built in, not bolted on. `auto` walks a fourteen-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive. +`web_search` is built in, not bolted on. `auto` walks an eighteen-provider chain; pin one by name if you already pay for it. Behind every hit, site-aware extraction turns GitHub, registries, arXiv, Stack Overflow, and docs into structured markdown — anchors and link targets survive. ### Search providers -Fourteen backends. Pin one, or let `auto` walk the chain in order. +Eighteen backends. Pin one, or let `auto` walk the chain in order. | provider | auth | | ------------ | ---------------------- | | `auto` | chain | -| `exa` | `EXA_API_KEY` (or mcp) | -| `brave` | `BRAVE_API_KEY` | -| `jina` | `JINA_API_KEY` | -| `kimi` | `MOONSHOT_API_KEY` | -| `zai` | `ZAI_API_KEY` | -| `anthropic` | oauth | | `perplexity` | `PERPLEXITY_API_KEY` | | `gemini` | oauth | +| `anthropic` | oauth | | `codex` | oauth | -| `tavily` | `TAVILY_API_KEY` | -| `parallel` | `PARALLEL_API_KEY` | +| `xai` | `XAI_API_KEY` | +| `zai` | `ZAI_API_KEY` | +| `exa` | `EXA_API_KEY` (or mcp) | +| `tinyfish` | `TINYFISH_API_KEY` | +| `jina` | `JINA_API_KEY` | | `kagi` | `KAGI_API_KEY` | +| `tavily` | `TAVILY_API_KEY` | +| `firecrawl` | `FIRECRAWL_API_KEY` | +| `brave` | `BRAVE_API_KEY` | +| `kimi` | `MOONSHOT_API_KEY` | +| `parallel` | `PARALLEL_API_KEY` | | `synthetic` | `SYNTHETIC_API_KEY` | | `searxng` | self-hosted | +| `duckduckgo` | no key | ### Specialised handlers diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md index 26cd5f3cb..602824544 100644 --- a/docs/tools/web_search.md +++ b/docs/tools/web_search.md @@ -14,7 +14,9 @@ - `packages/coding-agent/src/web/search/providers/anthropic.ts` — Claude web-search provider. - `packages/coding-agent/src/web/search/providers/brave.ts` — Brave Search API adapter. - `packages/coding-agent/src/web/search/providers/codex.ts` — OpenAI Codex SSE adapter. + - `packages/coding-agent/src/web/search/providers/duckduckgo.ts` — DuckDuckGo Instant Answer API adapter. - `packages/coding-agent/src/web/search/providers/exa.ts` — Exa API or MCP adapter. + - `packages/coding-agent/src/web/search/providers/firecrawl.ts` — Firecrawl search adapter. - `packages/coding-agent/src/web/search/providers/gemini.ts` — Gemini grounding SSE adapter. - `packages/coding-agent/src/web/search/providers/jina.ts` — Jina Reader search adapter. - `packages/coding-agent/src/web/search/providers/kagi.ts` — Kagi provider wrapper. @@ -24,6 +26,8 @@ - `packages/coding-agent/src/web/search/providers/searxng.ts` — self-hosted SearXNG adapter. - `packages/coding-agent/src/web/search/providers/synthetic.ts` — Synthetic search adapter. - `packages/coding-agent/src/web/search/providers/tavily.ts` — Tavily search adapter. + - `packages/coding-agent/src/web/search/providers/tinyfish.ts` — TinyFish search adapter. + - `packages/coding-agent/src/web/search/providers/xai.ts` — xAI Responses web-search adapter. - `packages/coding-agent/src/web/search/providers/zai.ts` — Z.AI remote MCP adapter. - `packages/coding-agent/src/web/parallel.ts` — Parallel search/extract HTTP client. - `packages/coding-agent/src/web/kagi.ts` — Kagi HTTP client. @@ -34,11 +38,11 @@ | Field | Type | Required | Description | | --- | --- | --- | --- | | `query` | `string` | Yes | Search query, passed to providers unchanged. | -| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, and Kagi. | +| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it; code maps it for Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl. | | `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. | -| `max_tokens` | `number` | No | Passed through as `maxOutputTokens` / `max_tokens` only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. | -| `temperature` | `number` | No | Passed through only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. | -| `num_search_results` | `number` | No | Requested upstream search breadth. For most providers this is the same count used for returned sources. Perplexity is the only adapter that keeps it distinct from `limit`. | +| `max_tokens` | `number` | No | Passed through as provider token caps (`maxOutputTokens`, `max_tokens`, or xAI `max_output_tokens`) only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | +| `temperature` | `number` | No | Passed through only by Anthropic, Gemini, xAI, and Perplexity API-key mode. Ignored by the other providers. | +| `num_search_results` | `number` | No | Requested upstream search breadth. Most providers use this as returned source count. Perplexity keeps it distinct from `limit`; xAI does not send a source-count parameter to Responses API. | ## Outputs The tool returns a single text content block plus structured `details`. @@ -73,7 +77,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - if `params.provider` is set and not `"auto"`, it loads that provider with `getSearchProvider()`; if `isExplicitlyAvailable()` returns true, the list is `[that provider]`, otherwise it falls back to `resolveProviderChain(authStorage, "auto")`. - otherwise it calls `resolveProviderChain()` with the module-global preferred provider from `packages/coding-agent/src/web/search/provider.ts`. 3. `resolveProviderChain()` lazily loads each provider module on demand and returns only available providers. If a preferred provider is set, it is tried first (gated by `isExplicitlyAvailable()`), then the static `SEARCH_PROVIDER_ORDER` excluding that provider, each gated by `isAvailable()`. Providers in the excluded set (`setExcludedSearchProviders()`) are skipped entirely, including as the preferred candidate. -4. If no providers are available, `executeSearch()` returns `Error: No web search provider configured.` with `details.response.provider = "none"`. +4. If no providers are available (for example, after excluding DuckDuckGo and lacking configured keyed/OAuth providers), `executeSearch()` returns `Error: No web search provider configured.` with `details.response.provider = "none"`. 5. For each provider in order, `executeSearch()` calls `provider.search()` with: - `query`, - `limit`, `recency`, `temperature`, `maxOutputTokens`, `numSearchResults`, @@ -91,36 +95,20 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - **Forced provider**: internal callers may pass `provider`; unavailable forced providers fall back to the auto chain instead of hard-failing (`packages/coding-agent/src/web/search/index.ts`). This field is not in the model-facing schema. - **Preferred provider**: `setPreferredSearchProvider()` sets a module-global default used by `resolveProviderChain()`. `packages/coding-agent/src/sdk.ts` and `packages/coding-agent/src/modes/controllers/selector-controller.ts` wire this from settings. - **Excluded providers**: `setExcludedSearchProviders()` records providers `resolveProviderChain()` must never return, including as fallbacks. Wired from the `providers.webSearchExclude` setting (`providers.webSearch` drives the preferred provider) in `packages/coding-agent/src/sdk.ts`, `packages/coding-agent/src/modes/interactive-mode.ts`, and `packages/coding-agent/src/modes/controllers/selector-controller.ts`. - - **Auto chain order**: `perplexity`, `gemini`, `anthropic`, `codex`, `zai`, `exa`, `jina`, `kagi`, `tavily`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). + - **Auto chain order** (18 providers): `perplexity`, `gemini`, `anthropic`, `codex`, `xai`, `zai`, `exa`, `tinyfish`, `jina`, `kagi`, `tavily`, `firecrawl`, `brave`, `kimi`, `parallel`, `synthetic`, `searxng`, `duckduckgo` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). - **Provider adapters** - - **Tavily** — `packages/coding-agent/src/web/search/providers/tavily.ts` - - Availability: API key from env or `agent.db` via `findCredential()`. - - Querying: POST `https://api.tavily.com/search`. - - `recency` maps to Tavily `time_range`; code explicitly keeps `topic` at default general scope instead of narrowing to news. - - `limit` / `num_search_results`: adapter uses `params.numSearchResults ?? params.limit`, clamped to `5..20` with default `5`. - - Output: `answer`, `sources`, `requestId`, `authMode: "api_key"`. - **Perplexity** — `packages/coding-agent/src/web/search/providers/perplexity.ts` - Availability: auth precedence is `PERPLEXITY_COOKIES` -> OAuth token in `agent.db` -> `PERPLEXITY_API_KEY` / `PPLX_API_KEY` -> anonymous ask-endpoint fallback. `isAvailable()` gates the auto chain on credentials, but `isExplicitlyAvailable()` is always true, so explicit selection works unauthenticated. - OAuth/cookie/anonymous mode: POSTs to `https://www.perplexity.ai/rest/sse/perplexity_ask`, consumes SSE, merges partial events, extracts answer and source URLs, sets `authMode: "oauth"` (`"anonymous"` for the unauthenticated fallback). - API-key mode: POSTs to `https://api.perplexity.ai/chat/completions` with `model: "sonar-pro"`, `search_mode: "web"`, `num_search_results`, optional `search_recency_filter`, `max_tokens`, `temperature`. - `num_search_results` controls upstream API breadth only in API-key mode. `limit` is preserved separately as `num_results` and slices returned `sources` after parsing in both auth modes. - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode`. - - **Brave** — `packages/coding-agent/src/web/search/providers/brave.ts` - - Availability: `BRAVE_API_KEY` only. - - Querying: GET `https://api.search.brave.com/res/v1/web/search` with `count`, `extra_snippets=true`, and `freshness=pd|pw|pm|py` for `recency`. - - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. - - Output: `sources`, `requestId`. - - **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts` - - Availability: `JINA_API_KEY` only. - - Querying: GET-like fetch to `https://s.jina.ai/` with bearer auth. - - Ignores `recency`, `max_tokens`, and `temperature`. - - `limit` / `num_search_results`: adapter slices sources to `params.numSearchResults ?? params.limit` when provided; otherwise returns all payload items. - - Output: `sources` only. - - **Kimi** — `packages/coding-agent/src/web/search/providers/kimi.ts` - - Availability: `MOONSHOT_SEARCH_API_KEY`, `KIMI_SEARCH_API_KEY`, `MOONSHOT_API_KEY`, or `agent.db` credentials for `moonshot` / `kimi-code`. - - Querying: POST to `MOONSHOT_SEARCH_BASE_URL` / `KIMI_SEARCH_BASE_URL` / default `https://api.kimi.com/coding/v1/search` with `text_query`, `limit`, `enable_page_crawling`, `timeout_seconds: 30`. - - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. - - Output: `sources`, `requestId`. + - **Gemini** — `packages/coding-agent/src/web/search/providers/gemini.ts` + - Availability: OAuth credentials in `agent.db` for `google-gemini-cli` or `google-antigravity`. + - Querying: SSE `streamGenerateContent` call with Google Search grounding enabled. Antigravity auth tries two fallback endpoints and retries `401/403/400 invalid auth` once after token refresh; `429/5xx` retry with exponential backoff and server-provided retry delay, capped by a `5 * 60 * 1000` ms rate-limit budget. + - `max_tokens` and `temperature` pass through as `generationConfig.maxOutputTokens` / `generationConfig.temperature`. + - `limit` and `num_search_results` are collapsed together before dispatch. + - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage`, `model`. - **Anthropic** — `packages/coding-agent/src/web/search/providers/anthropic.ts` - Availability: `ANTHROPIC_SEARCH_API_KEY` env var, otherwise `authStorage.hasAuth("anthropic")`; search credentials come from `authStorage.getApiKey("anthropic")` when no search-specific key is set. - Env overrides specific to search (do not affect chat completions): @@ -131,18 +119,17 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - `max_tokens` and `temperature` pass through. - `limit` and `num_search_results` are collapsed together before dispatch: `num_results = params.numSearchResults ?? params.limit`. - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage.searchRequests`, `model`, `requestId`. - - **Gemini** — `packages/coding-agent/src/web/search/providers/gemini.ts` - - Availability: OAuth credentials in `agent.db` for `google-gemini-cli` or `google-antigravity`. - - Querying: SSE `streamGenerateContent` call with Google Search grounding enabled. Antigravity auth tries two fallback endpoints and retries `401/403/400 invalid auth` once after token refresh; `429/5xx` retry with exponential backoff and server-provided retry delay, capped by a `5 * 60 * 1000` ms rate-limit budget. - - `max_tokens` and `temperature` pass through as `generationConfig.maxOutputTokens` / `generationConfig.temperature`. - - `limit` and `num_search_results` are collapsed together before dispatch. - - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage`, `model`. - **Codex** — `packages/coding-agent/src/web/search/providers/codex.ts` - Availability: OAuth credential for `openai-codex` in `agent.db` (`hasOAuth()`; expiry is not checked here — refresh is lazy in `searchCodex`). - Querying: SSE POST to `https://chatgpt.com/backend-api/codex/responses` with `tool_choice: { type: "web_search" }` and `search_context_size: "high"` by default. - Ignores `recency`, `max_tokens`, and `temperature` in this tool path. - `limit` and `num_search_results` are collapsed together before dispatch. - Output may include `answer`, `sources`, `usage`, `model`, `requestId`. If the streamed response has no `url_citation` annotations, the adapter falls back to scraping markdown links and bare URLs from the answer text. + - **xAI** — `packages/coding-agent/src/web/search/providers/xai.ts` + - Availability: `XAI_API_KEY` or `agent.db` credential for `xai`. + - Querying: POST `https://api.x.ai/v1/responses` with model `grok-4.3` and `tools: [{ type: "web_search" }]`. + - `max_tokens` and `temperature` pass through; `limit`, `num_search_results`, and `recency` are not sent. + - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode: "api_key"`. - **Z.AI** — `packages/coding-agent/src/web/search/providers/zai.ts` - Availability: env or `agent.db` credential for `zai`. - Querying: JSON-RPC `tools/call` against `https://api.z.ai/api/mcp/web_search_prime/mcp` for remote MCP tool `web_search_prime`. @@ -154,17 +141,47 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Querying: POST `https://api.exa.ai/search` with the resolved Exa API key, otherwise JSON-RPC `tools/call` against `https://mcp.exa.ai/mcp` for remote MCP tool `web_search_exa`. - `limit` and `num_search_results` are collapsed together before dispatch. - Output: synthesized `answer` from up to 3 result summaries, `sources`, `requestId`. + - **TinyFish** — `packages/coding-agent/src/web/search/providers/tinyfish.ts` + - Availability: `TINYFISH_API_KEY` or `agent.db` credential for `tinyfish`. + - Querying: GET `https://api.search.tinyfish.ai` with `X-API-Key`; `recency` maps to `recency_minutes`. + - `limit` / `num_search_results`: collapsed and clamped to `1..20`, default `10`; output `sources`, `authMode: "api_key"`. + - **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts` + - Availability: `JINA_API_KEY` only. + - Querying: GET-like fetch to `https://s.jina.ai/` with bearer auth. + - Ignores `recency`, `max_tokens`, and `temperature`. + - `limit` / `num_search_results`: adapter slices sources to `params.numSearchResults ?? params.limit` when provided; otherwise returns all payload items. + - Output: `sources` only. + - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` + - Availability: env or `agent.db` credential for `kagi`. + - Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer ` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`). + - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. + - Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`). + - **Tavily** — `packages/coding-agent/src/web/search/providers/tavily.ts` + - Availability: API key from env or `agent.db` via `findCredential()`. + - Querying: POST `https://api.tavily.com/search`. + - `recency` maps to Tavily `time_range`; code explicitly keeps `topic` at default general scope instead of narrowing to news. + - `limit` / `num_search_results`: adapter uses `params.numSearchResults ?? params.limit`, clamped to `5..20` with default `5`. + - Output: `answer`, `sources`, `requestId`, `authMode: "api_key"`. + - **Firecrawl** — `packages/coding-agent/src/web/search/providers/firecrawl.ts` + - Availability: `FIRECRAWL_API_KEY` or `agent.db` credential for `firecrawl`. + - Querying: POST `https://api.firecrawl.dev/v2/search` with `sources: [{ type: "web" }]`; `recency` maps to Google-style `tbs`. + - `limit` / `num_search_results`: collapsed and clamped to `1..100`, default `10`; output `sources`, `requestId`, `authMode: "api_key"`. + - **Brave** — `packages/coding-agent/src/web/search/providers/brave.ts` + - Availability: `BRAVE_API_KEY` only. + - Querying: GET `https://api.search.brave.com/res/v1/web/search` with `count`, `extra_snippets=true`, and `freshness=pd|pw|pm|py` for `recency`. + - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. + - Output: `sources`, `requestId`. + - **Kimi** — `packages/coding-agent/src/web/search/providers/kimi.ts` + - Availability: `MOONSHOT_SEARCH_API_KEY`, `KIMI_SEARCH_API_KEY`, `MOONSHOT_API_KEY`, or `agent.db` credentials for `moonshot` / `kimi-code`. + - Querying: POST to `MOONSHOT_SEARCH_BASE_URL` / `KIMI_SEARCH_BASE_URL` / default `https://api.kimi.com/coding/v1/search` with `text_query`, `limit`, `enable_page_crawling`, `timeout_seconds: 30`. + - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. + - Output: `sources`, `requestId`. - **Parallel** — `packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts` - Availability: env or `agent.db` credential for `parallel`. - Querying: POST `https://api.parallel.ai/v1beta/search` with `objective=query`, `search_queries=[query]`, `mode:"fast"`, `max_chars_per_result: 10000`, beta header `search-extract-2025-10-10`. - There is no provider fan-out here despite the name; the current adapter always sends a one-element `search_queries` array. - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. - Output: `sources`, `requestId`. - - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` - - Availability: env or `agent.db` credential for `kagi`. - - Querying: POST `https://kagi.com/api/v1/search` with `Authorization: Bearer ` and JSON body `{ query, workflow: "search", limit, filters?: { after } }`. `recency` maps to `filters.after` as a UTC `YYYY-MM-DD` string (`day`/`week`/`month`/`year`). - - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. - - Output: `sources` (concatenated `data.search` + `data.video` + `data.news` + `data.infobox`, with video/news/infobox results tagged in the title), `relatedQuestions` (`data.adjacent_question` + `data.related_search` `props.question`), `answer` (`data.direct_answer[0].snippet ?? title`), `requestId` (`meta.trace`). - **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts` - Availability: env or `agent.db` credential for `synthetic`. - Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`. @@ -178,6 +195,10 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - `recency` maps to `time_range`; `week` is downgraded to `month` because SearXNG does not support week. - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..20`, default `10`. - Output: `sources`, `relatedQuestions` from `suggestions`. + - **DuckDuckGo** — `packages/coding-agent/src/web/search/providers/duckduckgo.ts` + - Availability: always available; no API key. + - Querying: GET official Instant Answer API `https://api.duckduckgo.com/` with JSON/no-HTML flags; no scraped HTML. + - `limit` / `num_search_results`: collapsed and clamped to `1..20`, default `10`; output may include `answer` and `sources` from abstracts/results/topics. ## Side Effects - Network @@ -193,11 +214,14 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec - Many provider adapters accept `AbortSignal`; `WebSearchTool.execute()` passes the tool call signal into `executeSearch()`, which forwards it as `params.signal` to providers and rethrows cancellation during fallback. ## Limits & Caps -- Provider auto-order length: 14 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). +- Provider auto-order length: 18 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/types.ts`). - `formatForLLM()` truncates source snippets and citation text to 240 chars (`packages/coding-agent/src/web/search/index.ts`). - `formatForLLM()` emits at most 3 search queries, each truncated to 120 chars (`packages/coding-agent/src/web/search/index.ts`). - Brave result count: default `10`, max `20` (`DEFAULT_NUM_RESULTS`, `MAX_NUM_RESULTS` in `packages/coding-agent/src/web/search/providers/brave.ts`). +- TinyFish result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/tinyfish.ts`). +- DuckDuckGo result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/duckduckgo.ts`). - Tavily result count: default `5`, max `20` (`packages/coding-agent/src/web/search/providers/tavily.ts`). +- Firecrawl result count: default `10`, max `100` (`packages/coding-agent/src/web/search/providers/firecrawl.ts`). - Kimi result count: default `10`, max `20`; request timeout field fixed to `30` seconds (`packages/coding-agent/src/web/search/providers/kimi.ts`). - Parallel result count: default `10`, max `40`; per-result excerpt cap `10_000` chars (`packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts`). - Kagi result count: default `10`, max `40` (`packages/coding-agent/src/web/search/providers/kagi.ts`). @@ -222,7 +246,8 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec ## Notes - The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. - `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. -- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity is the only implementation that preserves both concepts. -- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, and Kagi; the model-facing prompt does not name specific providers. +- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity preserves both concepts; xAI currently ignores both. +- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, and Firecrawl; the model-facing prompt does not name specific providers. - `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain. +- DuckDuckGo is intentionally last in the auto chain because it is always available without credentials. - Exa uses `authStorage.getApiKey("exa")`, then `EXA_API_KEY`, then unauthenticated `https://mcp.exa.ai/mcp` fallback. diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 576bd6116..67fab9948 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -166,6 +166,8 @@ const LEGACY_ENV_KEYS: Record = { exa: "EXA_API_KEY", jina: "JINA_API_KEY", brave: "BRAVE_API_KEY", + tinyfish: "TINYFISH_API_KEY", + firecrawl: "FIRECRAWL_API_KEY", }; /** diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 1334b25e0..4afb641fb 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added TinyFish, DuckDuckGo, xAI, and Firecrawl web_search providers. + ### Fixed - Fixed Kimi-family models defaulting to hashline edit mode; they now fall back to `replace` unless `edit.modelVariants`, `PI_EDIT_VARIANT`, or `PI_STRICT_EDIT_MODE` explicitly opts into hashline. diff --git a/packages/coding-agent/src/cli/args.ts b/packages/coding-agent/src/cli/args.ts index 364bb94ac..b0b8d0321 100644 --- a/packages/coding-agent/src/cli/args.ts +++ b/packages/coding-agent/src/cli/args.ts @@ -310,6 +310,8 @@ export function getExtraHelpText(): string { PERPLEXITY_API_KEY - Perplexity web search API key (optional; anonymous fallback) PERPLEXITY_COOKIES - Perplexity web search (session cookie) TAVILY_API_KEY - Tavily web search + TINYFISH_API_KEY - TinyFish web search + FIRECRAWL_API_KEY - Firecrawl web search ANTHROPIC_SEARCH_API_KEY - Anthropic web search (override; isolates search from main ANTHROPIC_API_KEY) ANTHROPIC_SEARCH_BASE_URL - Anthropic web search base URL (override; pairs with ANTHROPIC_SEARCH_API_KEY) diff --git a/packages/coding-agent/src/cli/web-search-cli.ts b/packages/coding-agent/src/cli/web-search-cli.ts index b7ba9b0ba..a4c6f8cc2 100644 --- a/packages/coding-agent/src/cli/web-search-cli.ts +++ b/packages/coding-agent/src/cli/web-search-cli.ts @@ -120,7 +120,7 @@ ${chalk.bold("Arguments:")} ${chalk.bold("Options:")} --provider Provider: ${PROVIDERS.join(", ")} - --recency Recency filter (Brave/Perplexity): ${RECENCY_OPTIONS.join(", ")} + --recency Recency filter (when supported): ${RECENCY_OPTIONS.join(", ")} -l, --limit Max results to return --compact Render condensed output -h, --help Show this help diff --git a/packages/coding-agent/src/web/search/provider.ts b/packages/coding-agent/src/web/search/provider.ts index b74b6a00a..ee1e40674 100644 --- a/packages/coding-agent/src/web/search/provider.ts +++ b/packages/coding-agent/src/web/search/provider.ts @@ -24,66 +24,81 @@ interface ProviderMeta { /** Lazy factories. Each `load()` dynamic-imports its provider module on first call. */ const PROVIDER_META: Record = { - exa: { - id: "exa", - label: SEARCH_PROVIDER_LABELS.exa, - load: async () => new (await import("./providers/exa")).ExaProvider(), - }, - brave: { - id: "brave", - label: SEARCH_PROVIDER_LABELS.brave, - load: async () => new (await import("./providers/brave")).BraveProvider(), - }, - jina: { - id: "jina", - label: SEARCH_PROVIDER_LABELS.jina, - load: async () => new (await import("./providers/jina")).JinaProvider(), - }, perplexity: { id: "perplexity", label: SEARCH_PROVIDER_LABELS.perplexity, load: async () => new (await import("./providers/perplexity")).PerplexityProvider(), }, - kimi: { - id: "kimi", - label: SEARCH_PROVIDER_LABELS.kimi, - load: async () => new (await import("./providers/kimi")).KimiProvider(), - }, - zai: { - id: "zai", - label: SEARCH_PROVIDER_LABELS.zai, - load: async () => new (await import("./providers/zai")).ZaiProvider(), - }, - anthropic: { - id: "anthropic", - label: SEARCH_PROVIDER_LABELS.anthropic, - load: async () => new (await import("./providers/anthropic")).AnthropicProvider(), - }, gemini: { id: "gemini", label: SEARCH_PROVIDER_LABELS.gemini, load: async () => new (await import("./providers/gemini")).GeminiProvider(), }, + anthropic: { + id: "anthropic", + label: SEARCH_PROVIDER_LABELS.anthropic, + load: async () => new (await import("./providers/anthropic")).AnthropicProvider(), + }, codex: { id: "codex", label: SEARCH_PROVIDER_LABELS.codex, load: async () => new (await import("./providers/codex")).CodexProvider(), }, + xai: { + id: "xai", + label: SEARCH_PROVIDER_LABELS.xai, + load: async () => new (await import("./providers/xai")).XAIProvider(), + }, + zai: { + id: "zai", + label: SEARCH_PROVIDER_LABELS.zai, + load: async () => new (await import("./providers/zai")).ZaiProvider(), + }, + exa: { + id: "exa", + label: SEARCH_PROVIDER_LABELS.exa, + load: async () => new (await import("./providers/exa")).ExaProvider(), + }, + tinyfish: { + id: "tinyfish", + label: SEARCH_PROVIDER_LABELS.tinyfish, + load: async () => new (await import("./providers/tinyfish")).TinyFishProvider(), + }, + jina: { + id: "jina", + label: SEARCH_PROVIDER_LABELS.jina, + load: async () => new (await import("./providers/jina")).JinaProvider(), + }, + kagi: { + id: "kagi", + label: SEARCH_PROVIDER_LABELS.kagi, + load: async () => new (await import("./providers/kagi")).KagiProvider(), + }, tavily: { id: "tavily", label: SEARCH_PROVIDER_LABELS.tavily, load: async () => new (await import("./providers/tavily")).TavilyProvider(), }, + firecrawl: { + id: "firecrawl", + label: SEARCH_PROVIDER_LABELS.firecrawl, + load: async () => new (await import("./providers/firecrawl")).FirecrawlProvider(), + }, + brave: { + id: "brave", + label: SEARCH_PROVIDER_LABELS.brave, + load: async () => new (await import("./providers/brave")).BraveProvider(), + }, + kimi: { + id: "kimi", + label: SEARCH_PROVIDER_LABELS.kimi, + load: async () => new (await import("./providers/kimi")).KimiProvider(), + }, parallel: { id: "parallel", label: SEARCH_PROVIDER_LABELS.parallel, load: async () => new (await import("./providers/parallel")).ParallelProvider(), }, - kagi: { - id: "kagi", - label: SEARCH_PROVIDER_LABELS.kagi, - load: async () => new (await import("./providers/kagi")).KagiProvider(), - }, synthetic: { id: "synthetic", label: SEARCH_PROVIDER_LABELS.synthetic, @@ -94,6 +109,11 @@ const PROVIDER_META: Record = { label: SEARCH_PROVIDER_LABELS.searxng, load: async () => new (await import("./providers/searxng")).SearXNGProvider(), }, + duckduckgo: { + id: "duckduckgo", + label: SEARCH_PROVIDER_LABELS.duckduckgo, + load: async () => new (await import("./providers/duckduckgo")).DuckDuckGoProvider(), + }, }; const instanceCache = new Map(); diff --git a/packages/coding-agent/src/web/search/providers/duckduckgo.ts b/packages/coding-agent/src/web/search/providers/duckduckgo.ts new file mode 100644 index 000000000..88b2b7afc --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/duckduckgo.ts @@ -0,0 +1,140 @@ +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const DUCKDUCKGO_SEARCH_URL = "https://api.duckduckgo.com/"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; + +interface DuckDuckGoTopic { + FirstURL?: string | null; + Text?: string | null; + Topics?: DuckDuckGoTopic[] | null; +} + +interface DuckDuckGoResponse { + AbstractText?: string | null; + AbstractURL?: string | null; + AbstractSource?: string | null; + Answer?: string | null; + Definition?: string | null; + Heading?: string | null; + Results?: DuckDuckGoTopic[] | null; + RelatedTopics?: DuckDuckGoTopic[] | null; +} + +function cleanText(value: string | null | undefined): string | undefined { + const cleaned = value + ?.replace(/<[^>]*>/g, " ") + .replace(/ /gi, " ") + .replace(/&/gi, "&") + .replace(/</gi, "<") + .replace(/>/gi, ">") + .replace(/"/gi, '"') + .replace(/'/gi, "'") + .replace(/\s+/g, " ") + .trim(); + return cleaned ? cleaned : undefined; +} + +function addSource(sources: SearchSource[], source: SearchSource): void { + if (!source.url || sources.some(existing => existing.url === source.url)) return; + sources.push(source); +} + +function addTopicSource(sources: SearchSource[], topic: DuckDuckGoTopic): void { + const url = topic.FirstURL?.trim(); + if (!url) return; + const text = cleanText(topic.Text); + addSource(sources, { + title: text ?? url, + url, + snippet: text, + }); +} + +function collectTopicSources(sources: SearchSource[], topics: readonly DuckDuckGoTopic[] | null | undefined): void { + if (!topics) return; + for (const topic of topics) { + addTopicSource(sources, topic); + collectTopicSources(sources, topic.Topics); + } +} + +async function callDuckDuckGoSearch(params: SearchParams): Promise { + const queryString = [ + ["q", params.query], + ["format", "json"], + ["no_redirect", "1"], + ["no_html", "1"], + ["skip_disambig", "1"], + ["t", "oh-my-pi"], + ] + .map(([key, value]) => `${encodeURIComponent(key)}=${encodeURIComponent(value)}`) + .join("&"); + const response = await (params.fetch ?? fetch)(`${DUCKDUCKGO_SEARCH_URL}?${queryString}`, { + method: "GET", + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("duckduckgo", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "duckduckgo", + `DuckDuckGo API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as DuckDuckGoResponse; +} + +/** Execute DuckDuckGo Instant Answer API search. */ +export async function searchDuckDuckGo(params: SearchParams): Promise { + const numResults = clampNumResults(params.numSearchResults ?? params.limit, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + const data = await callDuckDuckGoSearch(params); + const answer = cleanText(data.AbstractText) ?? cleanText(data.Answer) ?? cleanText(data.Definition); + const sources: SearchSource[] = []; + + const abstractUrl = data.AbstractURL?.trim(); + if (abstractUrl) { + addSource(sources, { + title: cleanText(data.AbstractSource) ?? cleanText(data.Heading) ?? abstractUrl, + url: abstractUrl, + snippet: cleanText(data.AbstractText), + }); + } + + collectTopicSources(sources, data.Results); + collectTopicSources(sources, data.RelatedTopics); + + return { + provider: "duckduckgo", + answer, + sources: sources.slice(0, numResults), + }; +} + +/** Search provider for DuckDuckGo Instant Answer API. */ +export class DuckDuckGoProvider extends SearchProvider { + readonly id = "duckduckgo"; + readonly label = "DuckDuckGo"; + + isAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + isExplicitlyAvailable(_authStorage: AuthStorage): boolean { + return true; + } + + search(params: SearchParams): Promise { + return searchDuckDuckGo(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/firecrawl.ts b/packages/coding-agent/src/web/search/providers/firecrawl.ts new file mode 100644 index 000000000..3d7a02916 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/firecrawl.ts @@ -0,0 +1,144 @@ +/** + * Firecrawl Web Search Provider + * + * Calls Firecrawl's search API and maps web results into the unified + * SearchResponse shape used by the web search tool. + */ +import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const FIRECRAWL_SEARCH_URL = "https://api.firecrawl.dev/v2/search"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 100; + +const RECENCY_TBS: Record, string> = { + day: "qdr:d", + week: "qdr:w", + month: "qdr:m", + year: "qdr:y", +}; + +export interface FirecrawlSearchParams { + query: string; + num_results?: number; + recency?: SearchParams["recency"]; + signal?: AbortSignal; + fetch?: FetchImpl; +} + +interface FirecrawlWebResult { + title?: string | null; + url?: string | null; + description?: string | null; + markdown?: string | null; +} + +interface FirecrawlSearchResponse { + id?: string | null; + data?: { + web?: FirecrawlWebResult[] | null; + } | null; +} + +/** Resolve Firecrawl API key through the shared auth storage pipeline. */ +export function findApiKey( + authStorage: AuthStorage, + sessionId?: string, + signal?: AbortSignal, +): Promise { + return authStorage.getApiKey("firecrawl", sessionId, { signal }); +} + +function buildRequestBody(params: FirecrawlSearchParams): Record { + const body: Record = { + query: params.query, + limit: clampNumResults(params.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS), + sources: [{ type: "web" }], + }; + if (params.recency) { + body.tbs = RECENCY_TBS[params.recency]; + } + return body; +} + +async function callFirecrawlSearch(apiKey: string, params: FirecrawlSearchParams): Promise { + const response = await (params.fetch ?? fetch)(FIRECRAWL_SEARCH_URL, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${apiKey}`, + }, + body: JSON.stringify(buildRequestBody(params)), + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("firecrawl", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "firecrawl", + `Firecrawl API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as FirecrawlSearchResponse; +} + +/** Execute Firecrawl web search. */ +export async function searchFirecrawl(params: SearchParams): Promise { + const firecrawlParams: FirecrawlSearchParams = { + query: params.query, + num_results: params.numSearchResults ?? params.limit, + recency: params.recency, + signal: params.signal, + fetch: params.fetch, + }; + const keyOrResolver: ApiKey = params.authStorage.resolver("firecrawl", { + sessionId: params.sessionId, + }); + const numResults = clampNumResults(firecrawlParams.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + + const data = await withAuth(keyOrResolver, key => callFirecrawlSearch(key, firecrawlParams), { + signal: params.signal, + missingKeyMessage: + 'Firecrawl credentials not found. Set FIRECRAWL_API_KEY or configure an API key for provider "firecrawl".', + }); + const sources: SearchSource[] = []; + + for (const result of data.data?.web ?? []) { + if (!result.url) continue; + sources.push({ + title: result.title ?? result.url, + url: result.url, + snippet: result.description ?? result.markdown ?? undefined, + }); + } + + return { + provider: "firecrawl", + sources: sources.slice(0, numResults), + requestId: data.id ?? undefined, + authMode: "api_key", + }; +} + +/** Search provider for Firecrawl web search. */ +export class FirecrawlProvider extends SearchProvider { + readonly id = "firecrawl"; + readonly label = "Firecrawl"; + + isAvailable(authStorage: AuthStorage): boolean { + return authStorage.hasAuth("firecrawl") || !!getEnvApiKey("firecrawl"); + } + + search(params: SearchParams): Promise { + return searchFirecrawl(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/tinyfish.ts b/packages/coding-agent/src/web/search/providers/tinyfish.ts new file mode 100644 index 000000000..06d343b84 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/tinyfish.ts @@ -0,0 +1,134 @@ +/** + * TinyFish Web Search Provider + * + * Calls TinyFish's search API and maps results into the unified + * SearchResponse shape used by the web search tool. + */ +import { type ApiKey, type AuthStorage, type FetchImpl, getEnvApiKey, withAuth } from "@oh-my-pi/pi-ai"; +import type { SearchResponse, SearchSource } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import { clampNumResults } from "../utils"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const TINYFISH_SEARCH_URL = "https://api.search.tinyfish.ai"; +const DEFAULT_NUM_RESULTS = 10; +const MAX_NUM_RESULTS = 20; + +const RECENCY_MINUTES: Record, number> = { + day: 1440, + week: 10080, + month: 43200, + year: 525600, +}; + +export interface TinyFishSearchParams { + query: string; + num_results?: number; + recency?: SearchParams["recency"]; + signal?: AbortSignal; + fetch?: FetchImpl; +} + +interface TinyFishSearchResult { + title?: string | null; + url?: string | null; + snippet?: string | null; + site_name?: string | null; +} + +interface TinyFishSearchResponse { + results?: TinyFishSearchResult[] | null; +} + +/** Resolve TinyFish API key through the shared auth storage pipeline. */ +export function findApiKey( + authStorage: AuthStorage, + sessionId?: string, + signal?: AbortSignal, +): Promise { + return authStorage.getApiKey("tinyfish", sessionId, { signal }); +} + +async function callTinyFishSearch(apiKey: string, params: TinyFishSearchParams): Promise { + const url = new URL(TINYFISH_SEARCH_URL); + url.searchParams.set("query", params.query); + if (params.recency) { + url.searchParams.set("recency_minutes", String(RECENCY_MINUTES[params.recency])); + } + + const response = await (params.fetch ?? fetch)(url, { + method: "GET", + headers: { + Accept: "application/json", + "X-API-Key": apiKey, + }, + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("tinyfish", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "tinyfish", + `TinyFish API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as TinyFishSearchResponse; +} + +/** Execute TinyFish web search. */ +export async function searchTinyFish(params: SearchParams): Promise { + const tinyFishParams: TinyFishSearchParams = { + query: params.query, + num_results: params.numSearchResults ?? params.limit, + recency: params.recency, + signal: params.signal, + fetch: params.fetch, + }; + const keyOrResolver: ApiKey = params.authStorage.resolver("tinyfish", { + sessionId: params.sessionId, + }); + const numResults = clampNumResults(tinyFishParams.num_results, DEFAULT_NUM_RESULTS, MAX_NUM_RESULTS); + + const data = await withAuth(keyOrResolver, key => callTinyFishSearch(key, tinyFishParams), { + signal: params.signal, + missingKeyMessage: + 'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".', + }); + const sources: SearchSource[] = []; + + for (const result of data.results ?? []) { + if (!result.url) continue; + sources.push({ + title: result.title ?? result.site_name ?? result.url, + url: result.url, + snippet: result.snippet ?? undefined, + author: result.site_name ?? undefined, + }); + } + + return { + provider: "tinyfish", + sources: sources.slice(0, numResults), + authMode: "api_key", + }; +} + +/** Search provider for TinyFish web search. */ +export class TinyFishProvider extends SearchProvider { + readonly id = "tinyfish"; + readonly label = "TinyFish"; + + isAvailable(authStorage: AuthStorage): boolean { + return authStorage.hasAuth("tinyfish") || !!getEnvApiKey("tinyfish"); + } + + search(params: SearchParams): Promise { + return searchTinyFish(params); + } +} diff --git a/packages/coding-agent/src/web/search/providers/xai.ts b/packages/coding-agent/src/web/search/providers/xai.ts new file mode 100644 index 000000000..d5c047130 --- /dev/null +++ b/packages/coding-agent/src/web/search/providers/xai.ts @@ -0,0 +1,226 @@ +import { type ApiKey, type AuthStorage, withAuth } from "@oh-my-pi/pi-ai"; +import type { SearchCitation, SearchResponse, SearchSource, SearchUsage } from "../../../web/search/types"; +import { SearchProviderError } from "../../../web/search/types"; +import type { SearchParams } from "./base"; +import { SearchProvider } from "./base"; +import { classifyProviderHttpError, withHardTimeout } from "./utils"; + +const XAI_RESPONSES_URL = "https://api.x.ai/v1/responses"; +const XAI_WEB_SEARCH_MODEL = "grok-4.3"; + +interface XAIUrlCitationAnnotation { + type?: string; + url?: string | null; + title?: string | null; + text?: string | null; + cited_text?: string | null; +} + +interface XAIResponseContentPart { + type?: string; + text?: string | null; + output_text?: string | null; + annotations?: XAIUrlCitationAnnotation[] | null; +} + +interface XAIResponseOutputItem { + content?: XAIResponseContentPart[] | null; + annotations?: XAIUrlCitationAnnotation[] | null; +} + +interface XAIResponsesUsage { + input_tokens?: number; + output_tokens?: number; + total_tokens?: number; + inputTokens?: number; + outputTokens?: number; + totalTokens?: number; +} + +interface XAIResponsesResponse { + id?: string; + model?: string; + output_text?: string | null; + output?: XAIResponseOutputItem[] | null; + annotations?: XAIUrlCitationAnnotation[] | null; + citations?: string[] | null; + usage?: XAIResponsesUsage | null; +} + +function buildRequestBody(params: SearchParams): Record { + const body: Record = { + model: XAI_WEB_SEARCH_MODEL, + input: [ + { role: "system", content: params.systemPrompt }, + { role: "user", content: params.query }, + ], + tools: [{ type: "web_search" }], + }; + + if (params.maxOutputTokens !== undefined) { + body.max_output_tokens = params.maxOutputTokens; + } + if (params.temperature !== undefined) { + body.temperature = params.temperature; + } + + return body; +} + +async function callXAIResponses(apiKey: string, params: SearchParams): Promise { + const response = await (params.fetch ?? fetch)(XAI_RESPONSES_URL, { + method: "POST", + headers: { + "Content-Type": "application/json", + Authorization: `Bearer ${apiKey}`, + }, + body: JSON.stringify(buildRequestBody(params)), + signal: withHardTimeout(params.signal), + }); + + if (!response.ok) { + const errorText = await response.text(); + const classified = classifyProviderHttpError("xai", response.status, errorText); + if (classified) throw classified; + throw new SearchProviderError( + "xai", + `xAI Responses API error (${response.status}): ${errorText}`, + response.status, + ); + } + + return (await response.json()) as XAIResponsesResponse; +} + +function addCitationSource( + sources: SearchSource[], + citations: SearchCitation[], + seenUrls: Set, + url: string, + title?: string | null, + citedText?: string | null, +): void { + const trimmedUrl = url.trim(); + if (!trimmedUrl || seenUrls.has(trimmedUrl)) return; + seenUrls.add(trimmedUrl); + const sourceTitle = title?.trim() || trimmedUrl; + const sourceSnippet = citedText?.trim() || undefined; + + sources.push({ + title: sourceTitle, + url: trimmedUrl, + snippet: sourceSnippet, + }); + citations.push({ + title: sourceTitle, + url: trimmedUrl, + citedText: sourceSnippet, + }); +} + +function collectAnnotationSources( + annotations: readonly XAIUrlCitationAnnotation[] | null | undefined, + sources: SearchSource[], + citations: SearchCitation[], + seenUrls: Set, +): void { + if (!annotations) return; + for (const annotation of annotations) { + if (annotation.type !== "url_citation" || !annotation.url) continue; + addCitationSource( + sources, + citations, + seenUrls, + annotation.url, + annotation.title, + annotation.cited_text ?? annotation.text, + ); + } +} + +function parseAnswer(response: XAIResponsesResponse): string | undefined { + const topLevelText = response.output_text?.trim(); + if (topLevelText) return topLevelText; + + const answerParts: string[] = []; + for (const item of response.output ?? []) { + for (const part of item.content ?? []) { + const text = part.output_text ?? part.text; + if ((part.type === "output_text" || part.type === "text") && text?.trim()) { + answerParts.push(text.trim()); + } + } + } + + const answer = answerParts.join("\n").trim(); + return answer ? answer : undefined; +} + +function parseUsage(usage: XAIResponsesUsage | null | undefined): SearchUsage | undefined { + if (!usage) return undefined; + const parsed: SearchUsage = {}; + const inputTokens = usage.input_tokens ?? usage.inputTokens; + const outputTokens = usage.output_tokens ?? usage.outputTokens; + const totalTokens = usage.total_tokens ?? usage.totalTokens; + + if (typeof inputTokens === "number") parsed.inputTokens = inputTokens; + if (typeof outputTokens === "number") parsed.outputTokens = outputTokens; + if (typeof totalTokens === "number") parsed.totalTokens = totalTokens; + + return Object.keys(parsed).length > 0 ? parsed : undefined; +} + +function parseResponse(response: XAIResponsesResponse): SearchResponse { + const sources: SearchSource[] = []; + const citations: SearchCitation[] = []; + const seenUrls = new Set(); + + collectAnnotationSources(response.annotations, sources, citations, seenUrls); + for (const item of response.output ?? []) { + collectAnnotationSources(item.annotations, sources, citations, seenUrls); + for (const part of item.content ?? []) { + collectAnnotationSources(part.annotations, sources, citations, seenUrls); + } + } + for (const url of response.citations ?? []) { + addCitationSource(sources, citations, seenUrls, url); + } + + return { + provider: "xai", + answer: parseAnswer(response), + sources, + citations: citations.length > 0 ? citations : undefined, + usage: parseUsage(response.usage), + model: response.model, + requestId: response.id, + authMode: "api_key", + }; +} + +/** Execute xAI Responses API web search. */ +export async function searchXAI(params: SearchParams): Promise { + const keyOrResolver: ApiKey = params.authStorage.resolver("xai", { + sessionId: params.sessionId, + }); + + const response = await withAuth(keyOrResolver, (key: string) => callXAIResponses(key, params), { + signal: params.signal, + missingKeyMessage: 'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".', + }); + return parseResponse(response); +} + +/** Search provider for xAI web search. */ +export class XAIProvider extends SearchProvider { + readonly id = "xai"; + readonly label = "xAI"; + + isAvailable(authStorage: AuthStorage): boolean { + return authStorage.hasAuth("xai"); + } + + search(params: SearchParams): Promise { + return searchXAI(params); + } +} diff --git a/packages/coding-agent/src/web/search/types.ts b/packages/coding-agent/src/web/search/types.ts index ab15f2ab5..d2654211c 100644 --- a/packages/coding-agent/src/web/search/types.ts +++ b/packages/coding-agent/src/web/search/types.ts @@ -30,16 +30,20 @@ export const SEARCH_PROVIDER_OPTIONS = [ label: "OpenAI", description: "OpenAI's native web_search (uses ChatGPT OAuth via /login openai-codex)", }, + { value: "xai", label: "xAI", description: "Grok web search via xAI Responses API (requires XAI_API_KEY)" }, { value: "zai", label: "Z.AI", description: "Calls Z.AI webSearchPrime MCP" }, { value: "exa", label: "Exa", description: "Uses Exa API when EXA_API_KEY is set; falls back to Exa MCP" }, + { value: "tinyfish", label: "TinyFish", description: "Requires TINYFISH_API_KEY" }, { value: "jina", label: "Jina", description: "Requires JINA_API_KEY" }, { value: "kagi", label: "Kagi", description: "Requires KAGI_API_KEY and Kagi Search API beta access" }, { value: "tavily", label: "Tavily", description: "Requires TAVILY_API_KEY" }, + { value: "firecrawl", label: "Firecrawl", description: "Requires FIRECRAWL_API_KEY" }, { value: "brave", label: "Brave", description: "Requires BRAVE_API_KEY" }, { value: "kimi", label: "Kimi", description: "Requires MOONSHOT_SEARCH_API_KEY or MOONSHOT_API_KEY" }, { value: "parallel", label: "Parallel", description: "Requires PARALLEL_API_KEY" }, { value: "synthetic", label: "Synthetic", description: "Requires SYNTHETIC_API_KEY" }, { value: "searxng", label: "SearXNG", description: "Requires SEARXNG_ENDPOINT or searxng.endpoint" }, + { value: "duckduckgo", label: "DuckDuckGo", description: "Uses DuckDuckGo Instant Answer API (no API key)" }, ] as const; /** Supported web search providers (every option except `auto`). */ @@ -81,7 +85,7 @@ export interface SearchSource { author?: string; } -/** Citation with text reference (anthropic, perplexity) */ +/** Citation with text reference (LLM-mediated providers) */ export interface SearchCitation { url: string; title: string; @@ -101,7 +105,7 @@ export interface SearchUsage { /** Unified response across providers */ export interface SearchResponse { provider: SearchProviderId | "none"; - /** Synthesized answer text (anthropic, perplexity) */ + /** Synthesized answer text (LLM-mediated providers) */ answer?: string; /** Search result sources */ sources: SearchSource[]; diff --git a/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts new file mode 100644 index 000000000..e99861b27 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-duckduckgo.test.ts @@ -0,0 +1,188 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchDuckDuckGo } from "@oh-my-pi/pi-coding-agent/web/search/providers/duckduckgo"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const fakeAuthStorage = { + async getApiKey() { + throw new Error("DuckDuckGo must not request API keys"); + }, + resolver() { + throw new Error("DuckDuckGo must not request credential resolvers"); + }, + hasAuth() { + throw new Error("DuckDuckGo search must not check auth"); + }, +} as unknown as AuthStorage; + +function makeParams(query: string, fetch: FetchImpl) { + return { + query, + authStorage: fakeAuthStorage, + systemPrompt: "DuckDuckGo test prompt", + fetch, + } as const; +} + +describe("DuckDuckGo web search provider", () => { + it("calls the official Instant Answer API with unauthenticated JSON query params", async () => { + let capturedUrl: string | null = null; + let capturedInit: RequestInit | undefined; + const fetchMock: FetchImpl = (input, init) => { + capturedUrl = typeof input === "string" ? input : input.toString(); + capturedInit = init; + return Promise.resolve( + new Response(JSON.stringify({ AbstractText: "Duck answer", Results: [] }), { + status: 200, + headers: { "Content-Type": "application/json" }, + }), + ); + }; + + await searchDuckDuckGo(makeParams("instant answer", fetchMock)); + + expect(capturedUrl).not.toBeNull(); + const url = new URL(capturedUrl ?? ""); + expect(`${url.origin}${url.pathname}`).toBe("https://api.duckduckgo.com/"); + expect(url.searchParams.get("q")).toBe("instant answer"); + expect(url.searchParams.get("format")).toBe("json"); + expect(url.searchParams.get("no_redirect")).toBe("1"); + expect(url.searchParams.get("no_html")).toBe("1"); + expect(url.searchParams.get("skip_disambig")).toBe("1"); + expect(url.searchParams.get("t")).toBe("oh-my-pi"); + expect(capturedInit?.method).toBe("GET"); + expect(capturedInit?.headers).toBeUndefined(); + }); + + it("uses AbstractText as the answer and flattens abstract, result, and nested related topics within the local limit", async () => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response( + JSON.stringify({ + AbstractText: " DuckDuckGo abstract & answer ", + AbstractURL: " https://example.com/abstract ", + AbstractSource: " Example Abstract Source ", + Heading: "Example Heading", + Results: [ + { + FirstURL: "https://example.com/result", + Text: "Result snippet", + }, + ], + RelatedTopics: [ + { + FirstURL: "https://example.com/related", + Text: "Related topic", + }, + { + Topics: [ + { + FirstURL: "https://example.com/nested", + Text: "Nested related topic", + }, + ], + }, + { + FirstURL: "https://example.com/omitted-by-limit", + Text: "Should be omitted by local limit", + }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ), + ); + + const response = await searchDuckDuckGo({ ...makeParams("duck mapping", fetchMock), numSearchResults: 4 }); + + expect(response).toMatchObject({ + provider: "duckduckgo", + answer: "DuckDuckGo abstract & answer", + sources: [ + { + title: "Example Abstract Source", + url: "https://example.com/abstract", + snippet: "DuckDuckGo abstract & answer", + }, + { + title: "Result snippet", + url: "https://example.com/result", + snippet: "Result snippet", + }, + { + title: "Related topic", + url: "https://example.com/related", + snippet: "Related topic", + }, + { + title: "Nested related topic", + url: "https://example.com/nested", + snippet: "Nested related topic", + }, + ], + }); + expect(response.sources).toHaveLength(4); + expect(response.sources.some(source => source.url === "https://example.com/omitted-by-limit")).toBe(false); + }); + + it("clamps oversized local result limits to DuckDuckGo's provider maximum", async () => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response( + JSON.stringify({ + RelatedTopics: Array.from({ length: 25 }, (_value, index) => ({ + FirstURL: `https://example.com/topic-${index}`, + Text: `Topic ${index}`, + })), + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ), + ); + + const response = await searchDuckDuckGo({ ...makeParams("duck clamp", fetchMock), numSearchResults: 999 }); + + expect(response.sources).toHaveLength(20); + expect(response.sources.at(0)?.url).toBe("https://example.com/topic-0"); + expect(response.sources.at(-1)?.url).toBe("https://example.com/topic-19"); + expect(response.sources.some(source => source.url === "https://example.com/topic-20")).toBe(false); + }); + + it.each([ + ["Answer", { Answer: " Direct answer " }, "Direct answer"], + ["Definition", { Definition: " Definition answer " }, "Definition answer"], + ] as const)("falls back to %s when AbstractText is absent", async (_field, payload, expectedAnswer) => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response(JSON.stringify(payload), { + status: 200, + headers: { "Content-Type": "application/json" }, + }), + ); + + const response = await searchDuckDuckGo(makeParams("fallback answer", fetchMock)); + expect(response).toMatchObject({ + provider: "duckduckgo", + answer: expectedAnswer, + }); + }); + + it("throws a provider-tagged SearchProviderError for HTTP failures", async () => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response("upstream unavailable", { + status: 503, + }), + ); + + try { + await searchDuckDuckGo(makeParams("http failure", fetchMock)); + expect.unreachable("DuckDuckGo HTTP failure should reject"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "duckduckgo", + status: 503, + message: "DuckDuckGo API error (503): upstream unavailable", + }); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-firecrawl.test.ts b/packages/coding-agent/test/tools/web-search-firecrawl.test.ts new file mode 100644 index 000000000..d4b2b34a8 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-firecrawl.test.ts @@ -0,0 +1,138 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchFirecrawl } from "@oh-my-pi/pi-coding-agent/web/search/providers/firecrawl"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const TEST_KEY = "test-firecrawl-key"; + +function makeAuthStorage(apiKey: string | undefined): AuthStorage { + return { + resolver(provider: string, options?: { sessionId?: string }) { + expect(provider).toBe("firecrawl"); + expect(options?.sessionId).toBe("session-firecrawl-test"); + return async () => apiKey; + }, + hasAuth(provider: string) { + return provider === "firecrawl" && Boolean(apiKey); + }, + } as unknown as AuthStorage; +} + +function makeParams(query: string, authStorage: AuthStorage = makeAuthStorage(TEST_KEY)) { + return { + query, + authStorage, + systemPrompt: "Firecrawl test prompt", + sessionId: "session-firecrawl-test", + } as const; +} + +function getHeader(headers: RequestInit["headers"] | undefined, name: string): string | null { + if (!headers) return null; + if (headers instanceof Headers) return headers.get(name); + if (Array.isArray(headers)) { + return headers.find(([key]) => key.toLowerCase() === name.toLowerCase())?.[1] ?? null; + } + const record = headers as Record; + return record[name] ?? record[name.toLowerCase()] ?? null; +} + +describe("Firecrawl web search provider", () => { + it("sends the Firecrawl POST request and maps web results", async () => { + const captured: { url?: string; init?: RequestInit; body?: unknown } = {}; + + const fetchMock: FetchImpl = async (input, init) => { + captured.url = typeof input === "string" ? input : input instanceof URL ? input.toString() : input.url; + captured.init = init; + captured.body = JSON.parse(String(init?.body ?? "null")) as unknown; + return new Response( + JSON.stringify({ + id: "firecrawl-request-123", + data: { + web: [ + { + title: "Firecrawl result one", + url: "https://example.com/one", + description: "Description snippet", + markdown: "Ignored markdown", + }, + { + title: "Firecrawl result two", + url: "https://example.com/two", + description: null, + markdown: "Markdown fallback snippet", + }, + ], + }, + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + }; + + const response = await searchFirecrawl({ + ...makeParams("firecrawl query"), + numSearchResults: 2, + recency: "month", + fetch: fetchMock, + }); + + expect(captured.url).toBe("https://api.firecrawl.dev/v2/search"); + expect(captured.init?.method).toBe("POST"); + expect(getHeader(captured.init?.headers, "Authorization")).toBe(`Bearer ${TEST_KEY}`); + expect(getHeader(captured.init?.headers, "Content-Type")).toBe("application/json"); + expect(captured.body).toEqual({ + query: "firecrawl query", + limit: 2, + sources: [{ type: "web" }], + tbs: "qdr:m", + }); + expect(response).toEqual({ + provider: "firecrawl", + sources: [ + { + title: "Firecrawl result one", + url: "https://example.com/one", + snippet: "Description snippet", + }, + { + title: "Firecrawl result two", + url: "https://example.com/two", + snippet: "Markdown fallback snippet", + }, + ], + requestId: "firecrawl-request-123", + authMode: "api_key", + }); + }); + + it.each([ + [401, "firecrawl: 401 unauthorized"], + [402, "firecrawl: 402 credits exhausted"], + ] as const)("maps HTTP %d to a SearchProviderError", async (status, message) => { + const fetchMock: FetchImpl = async () => new Response("upstream rejected", { status }); + + try { + await searchFirecrawl({ ...makeParams("bad auth"), fetch: fetchMock }); + expect.unreachable("expected searchFirecrawl to throw"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "firecrawl", status, message }); + } + }); + + it("throws a clear error when Firecrawl credentials are missing", async () => { + const fetchMock: FetchImpl = async () => { + throw new Error("fetch should not be called without credentials"); + }; + + try { + await searchFirecrawl({ ...makeParams("missing creds", makeAuthStorage(undefined)), fetch: fetchMock }); + expect.unreachable("expected searchFirecrawl to throw"); + } catch (error) { + expect(error).toBeInstanceOf(Error); + expect((error as Error).message).toBe( + 'Firecrawl credentials not found. Set FIRECRAWL_API_KEY or configure an API key for provider "firecrawl".', + ); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-tinyfish.test.ts b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts new file mode 100644 index 000000000..c8b735e62 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-tinyfish.test.ts @@ -0,0 +1,127 @@ +import { describe, expect, it } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchTinyFish } from "@oh-my-pi/pi-coding-agent/web/search/providers/tinyfish"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +const TEST_KEY = "test-tinyfish-key"; + +function makeAuthStorage(apiKey: string | undefined): AuthStorage { + return { + resolver(provider: string, options?: { sessionId?: string }) { + expect(provider).toBe("tinyfish"); + expect(options?.sessionId).toBe("session-tinyfish-test"); + return async () => apiKey; + }, + hasAuth(provider: string) { + return provider === "tinyfish" && Boolean(apiKey); + }, + } as unknown as AuthStorage; +} + +function makeParams(query: string, authStorage: AuthStorage = makeAuthStorage(TEST_KEY)) { + return { + query, + authStorage, + systemPrompt: "TinyFish test prompt", + sessionId: "session-tinyfish-test", + } as const; +} + +function getHeader(headers: RequestInit["headers"] | undefined, name: string): string | null { + if (!headers) return null; + if (headers instanceof Headers) return headers.get(name); + if (Array.isArray(headers)) { + return headers.find(([key]) => key.toLowerCase() === name.toLowerCase())?.[1] ?? null; + } + const record = headers as Record; + return record[name] ?? record[name.toLowerCase()] ?? null; +} + +describe("TinyFish web search provider", () => { + it("sends the TinyFish GET request and locally clamps results", async () => { + const captured: { url?: URL; init?: RequestInit } = {}; + + const fetchMock: FetchImpl = async (input, init) => { + captured.url = input instanceof URL ? input : new URL(typeof input === "string" ? input : input.url); + captured.init = init; + return new Response( + JSON.stringify({ + results: [ + { + title: "TinyFish result one", + url: "https://example.com/one", + snippet: "First snippet", + site_name: "Example Site", + }, + { + title: "TinyFish result two", + url: "https://example.com/two", + snippet: "Second snippet", + }, + ], + }), + { status: 200, headers: { "Content-Type": "application/json" } }, + ); + }; + + const response = await searchTinyFish({ + ...makeParams("fresh fish"), + numSearchResults: 1, + recency: "week", + fetch: fetchMock, + }); + + const capturedUrl = captured.url; + if (!capturedUrl) throw new Error("TinyFish request was not captured"); + const endpoint = `${capturedUrl.origin}${capturedUrl.pathname === "/" ? "" : capturedUrl.pathname}`; + expect(endpoint).toBe("https://api.search.tinyfish.ai"); + expect(captured.init?.method ?? "GET").toBe("GET"); + expect(getHeader(captured.init?.headers, "X-API-Key")).toBe(TEST_KEY); + expect(capturedUrl.searchParams.get("query")).toBe("fresh fish"); + expect(capturedUrl.searchParams.get("recency_minutes")).toBe("10080"); + expect([...capturedUrl.searchParams.keys()].sort()).toEqual(["query", "recency_minutes"]); + expect(response).toEqual({ + provider: "tinyfish", + sources: [ + { + title: "TinyFish result one", + url: "https://example.com/one", + snippet: "First snippet", + author: "Example Site", + }, + ], + authMode: "api_key", + }); + }); + + it.each([ + [401, "tinyfish: 401 unauthorized"], + [402, "tinyfish: 402 credits exhausted"], + ] as const)("maps HTTP %d to a SearchProviderError", async (status, message) => { + const fetchMock: FetchImpl = async () => new Response("upstream rejected", { status }); + + try { + await searchTinyFish({ ...makeParams("bad auth"), fetch: fetchMock }); + expect.unreachable("expected searchTinyFish to throw"); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ provider: "tinyfish", status, message }); + } + }); + + it("throws a clear error when TinyFish credentials are missing", async () => { + const fetchMock: FetchImpl = async () => { + throw new Error("fetch should not be called without credentials"); + }; + + try { + await searchTinyFish({ ...makeParams("missing creds", makeAuthStorage(undefined)), fetch: fetchMock }); + expect.unreachable("expected searchTinyFish to throw"); + } catch (error) { + expect(error).toBeInstanceOf(Error); + expect((error as Error).message).toBe( + 'TinyFish credentials not found. Set TINYFISH_API_KEY or configure an API key for provider "tinyfish".', + ); + } + }); +}); diff --git a/packages/coding-agent/test/tools/web-search-xai.test.ts b/packages/coding-agent/test/tools/web-search-xai.test.ts new file mode 100644 index 000000000..4b0db7dd3 --- /dev/null +++ b/packages/coding-agent/test/tools/web-search-xai.test.ts @@ -0,0 +1,259 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import type { AuthStorage, FetchImpl } from "@oh-my-pi/pi-ai"; +import { searchXAI } from "@oh-my-pi/pi-coding-agent/web/search/providers/xai"; +import { SearchProviderError } from "@oh-my-pi/pi-coding-agent/web/search/types"; + +type CapturedRequest = { + url: string; + method: string | undefined; + headers: RequestInit["headers"]; + body: Record | null; +}; + +function makeAuthStorage(apiKey: string | undefined) { + return { + resolver(provider: string, options?: { sessionId?: string }) { + expect(provider).toBe("xai"); + expect(options?.sessionId).toBe("session-xai-test"); + return async () => apiKey; + }, + hasAuth(provider: string) { + return provider === "xai" && Boolean(apiKey); + }, + } as unknown as AuthStorage; +} + +function makeParams(fetch: FetchImpl, authStorage: AuthStorage = makeAuthStorage("test-xai-key")) { + return { + query: "latest xAI web search", + systemPrompt: "Use web search for current xAI facts.", + authStorage, + fetch, + sessionId: "session-xai-test", + } as const; +} + +function captureFetch(responseBody: Record, status = 200) { + let capturedRequest: CapturedRequest | null = null; + const fetchMock: FetchImpl = (input, init) => { + capturedRequest = { + url: typeof input === "string" ? input : input.toString(), + method: init?.method, + headers: init?.headers, + body: init?.body ? (JSON.parse(String(init.body)) as Record) : null, + }; + return Promise.resolve( + new Response(JSON.stringify(responseBody), { + status, + headers: { "Content-Type": "application/json" }, + }), + ); + }; + return { + fetchMock, + get capturedRequest() { + return capturedRequest; + }, + }; +} + +describe("xAI web search provider", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("POSTs the Responses API with bearer auth and xAI web_search tool payload", async () => { + const capture = captureFetch({ id: "resp_request", model: "grok-4.3", output_text: "xAI answer" }); + + await searchXAI({ + ...makeParams(capture.fetchMock), + maxOutputTokens: 512, + temperature: 0.2, + }); + + expect(capture.capturedRequest).not.toBeNull(); + expect(capture.capturedRequest?.url).toBe("https://api.x.ai/v1/responses"); + expect(capture.capturedRequest?.method).toBe("POST"); + expect(capture.capturedRequest?.headers).toMatchObject({ + "Content-Type": "application/json", + Authorization: "Bearer test-xai-key", + }); + expect(capture.capturedRequest?.body).toMatchObject({ + model: "grok-4.3", + input: [ + { role: "system", content: "Use web search for current xAI facts." }, + { role: "user", content: "latest xAI web search" }, + ], + tools: [{ type: "web_search" }], + max_output_tokens: 512, + temperature: 0.2, + }); + }); + + it("maps output_text, URL citation annotations, top-level citations, id, model, usage, and auth mode", async () => { + const capture = captureFetch({ + id: "resp_xai_123", + model: "grok-4.3", + output_text: "Top-level xAI answer", + annotations: [ + { + type: "url_citation", + url: "https://example.com/top-annotation", + title: "Top Annotation", + text: "Top annotation text", + }, + ], + output: [ + { + type: "message", + annotations: [ + { + type: "url_citation", + url: "https://example.com/item-annotation", + title: "Item Annotation", + cited_text: "Item annotation text", + }, + ], + content: [ + { + type: "output_text", + text: "Ignored because output_text wins", + annotations: [ + { + type: "url_citation", + url: "https://example.com/annotated", + title: "Annotated Source", + cited_text: "Annotated cited text", + }, + ], + }, + ], + }, + ], + citations: ["https://example.com/top-level-citation"], + usage: { + input_tokens: 12, + output_tokens: 8, + total_tokens: 20, + }, + }); + + const response = await searchXAI(makeParams(capture.fetchMock)); + + expect(response).toMatchObject({ + provider: "xai", + answer: "Top-level xAI answer", + requestId: "resp_xai_123", + model: "grok-4.3", + authMode: "api_key", + usage: { + inputTokens: 12, + outputTokens: 8, + totalTokens: 20, + }, + sources: [ + { + title: "Top Annotation", + url: "https://example.com/top-annotation", + snippet: "Top annotation text", + }, + { + title: "Item Annotation", + url: "https://example.com/item-annotation", + snippet: "Item annotation text", + }, + { + title: "Annotated Source", + url: "https://example.com/annotated", + snippet: "Annotated cited text", + }, + { + title: "https://example.com/top-level-citation", + url: "https://example.com/top-level-citation", + }, + ], + citations: [ + { + title: "Top Annotation", + url: "https://example.com/top-annotation", + citedText: "Top annotation text", + }, + { + title: "Item Annotation", + url: "https://example.com/item-annotation", + citedText: "Item annotation text", + }, + { + title: "Annotated Source", + url: "https://example.com/annotated", + citedText: "Annotated cited text", + }, + { + title: "https://example.com/top-level-citation", + url: "https://example.com/top-level-citation", + }, + ], + }); + }); + + it("falls back to output content parts when output_text is absent", async () => { + const capture = captureFetch({ + id: "resp_content_parts", + model: "grok-4.3", + output: [ + { + content: [ + { type: "output_text", text: "First content part" }, + { type: "text", output_text: "Second content part" }, + ], + }, + ], + }); + + const response = await searchXAI(makeParams(capture.fetchMock)); + expect(response).toMatchObject({ + answer: "First content part\nSecond content part", + }); + }); + + it.each([ + [401, "xai: 401 unauthorized"], + [402, "xai: 402 credits exhausted"], + ] as const)("maps HTTP %s failures to SearchProviderError", async (status, message) => { + const fetchMock: FetchImpl = () => + Promise.resolve( + new Response(JSON.stringify({ error: "request failed" }), { + status, + headers: { "Content-Type": "application/json" }, + }), + ); + + try { + await searchXAI(makeParams(fetchMock)); + expect.unreachable(`xAI HTTP ${status} failure should reject`); + } catch (error) { + expect(error).toBeInstanceOf(SearchProviderError); + expect(error).toMatchObject({ + provider: "xai", + status, + message, + }); + } + }); + + it("throws a clear missing-key error before fetch when credentials are unavailable", async () => { + const fetchMock = vi.fn(() => Promise.resolve(new Response("{}", { status: 200 }))) as unknown as FetchImpl; + + try { + await searchXAI(makeParams(fetchMock, makeAuthStorage(undefined))); + expect.unreachable("missing xAI credentials should reject"); + } catch (error) { + expect(error).toBeInstanceOf(Error); + expect(error).toHaveProperty( + "message", + 'xAI credentials not found. Set XAI_API_KEY or configure an API key for provider "xai".', + ); + } + expect(fetchMock).not.toHaveBeenCalled(); + }); +});