Merge remote-tracking branch 'origin/main' into feat/security-native
This commit is contained in:
@@ -393,6 +393,8 @@ Twenty-five backends. Pin one, or let `auto` walk the chain in order.
|
||||
| `mojeek` | no key (browser) |
|
||||
| `public` | no key (all of the above, consolidated) |
|
||||
|
||||
Exa also accepts a stored API key through `/login exa`; explicit keyless selection uses the public MCP fallback.
|
||||
|
||||
### Specialised handlers
|
||||
|
||||
The agent gets structured content, not stripped HTML.
|
||||
|
||||
@@ -244,7 +244,7 @@ OAuth host chain: `KIMI_CODE_OAUTH_HOST` → `KIMI_OAUTH_HOST` → `https://auth
|
||||
|
||||
| Variable | Used by |
|
||||
| --------------------------------------------------- | ------------------------------------------------------------- |
|
||||
| `EXA_API_KEY` | Exa search provider and Exa MCP tools |
|
||||
| `EXA_API_KEY` | Exa search/MCP; alternatively use `/login exa` |
|
||||
| `BRAVE_API_KEY` | Brave search provider |
|
||||
| `PERPLEXITY_API_KEY` | Perplexity search provider API-key mode |
|
||||
| `PERPLEXITY_COOKIES` | Perplexity cookie-auth search mode |
|
||||
|
||||
@@ -158,6 +158,7 @@ Handlers and tool `execute` receive `ctx` with:
|
||||
- `modelRegistry`, `model`
|
||||
- `models` (read-only model query — see below)
|
||||
- `getContextUsage()`
|
||||
- `getAsyncJobSnapshot()` returns the current session's read-only async-job snapshot, or `null` when no session owns the context
|
||||
- `compact(...)`
|
||||
- `isIdle()`, `hasPendingMessages()`, `abort()`
|
||||
- `shutdown()`
|
||||
@@ -268,6 +269,23 @@ Cancelable pre-events:
|
||||
- `goal_updated`
|
||||
- `credential_disabled`
|
||||
|
||||
### MCP notifications
|
||||
|
||||
- `mcp_notification` — fired for every JSON-RPC notification received from a connected MCP server, AFTER the manager's own handling of known list/update methods (`notifications/tools/list_changed`, `notifications/resources/list_changed`, `notifications/resources/updated`, `notifications/prompts/list_changed`). Unknown or server-custom methods are also delivered. Payload: `{ server: string; method: string; params: unknown }`. Multiple extensions may subscribe; a handler that throws does not prevent other handlers from firing. Notifications received before any listener attaches are buffered (bounded FIFO, cap 100, drop-oldest) and drained into the first subscriber — so startup-time frames aren't lost even if the extension binds after MCP discovery.
|
||||
|
||||
Bridging a push-capable MCP into a session steer:
|
||||
|
||||
```ts
|
||||
pi.on("mcp_notification", event => {
|
||||
if (event.server !== "peer-bus") return;
|
||||
if (event.method !== "notifications/peer_message") return;
|
||||
const params = event.params as { from: string; text: string };
|
||||
pi.sendUserMessage(`[from ${params.from}] ${params.text}`, { deliverAs: "steer" });
|
||||
});
|
||||
```
|
||||
|
||||
The runtime handles the JSON-RPC transport and its own list/update refresh first; the handler runs afterwards and can inject a mid-turn steer via `pi.sendMessage` / `pi.sendUserMessage`.
|
||||
|
||||
### User command interception
|
||||
|
||||
- `user_bash` (override with `{ result }`)
|
||||
|
||||
@@ -158,6 +158,19 @@ Both return structured tool output and convert remaining transport/tool errors i
|
||||
|
||||
There is also a follow-up path for late connections: after waiting for a specific server, if status becomes `connected`, it re-runs `session.refreshMCPTools(...)` so newly available tools are rebound in-session.
|
||||
|
||||
|
||||
## Server-initiated notifications
|
||||
|
||||
MCP servers may push JSON-RPC notification frames at any point after `initialize` completes. The transport surfaces them via `onNotification`; the manager fans them out in two paths:
|
||||
|
||||
1. **Internal refresh** for known methods:
|
||||
- `notifications/tools/list_changed` → `refreshServerTools`
|
||||
- `notifications/resources/list_changed` → `refreshServerResources`
|
||||
- `notifications/resources/updated` → `#onResourcesChanged` (only for currently subscribed URIs)
|
||||
- `notifications/prompts/list_changed` → `refreshServerPrompts`
|
||||
2. **Listener fanout**: every notification (including the known ones AND server-custom methods) is delivered to registered listeners AFTER the internal refresh runs. Registered via `MCPManager.addNotificationListener(listener)`, which returns an unsubscribe function. Multiple listeners are supported; each is invoked with independent error isolation — a synchronous throw in one listener does not prevent others from firing (thrown errors are logged at `debug`).
|
||||
|
||||
`sdk.ts` registers one listener that bridges to the extension runner's `mcp_notification` event, so extensions receive every server-initiated frame with `{ server, method, params }`. The listener is captured with `postmortem` so it is released on session teardown.
|
||||
## Health, reconnect, and partial failure behavior
|
||||
|
||||
Current runtime behavior is connection-event driven:
|
||||
|
||||
+84
@@ -124,6 +124,7 @@ Important edge behavior from runtime:
|
||||
### State
|
||||
|
||||
- `{ id?, type: "get_state" }`
|
||||
- `{ id?, type: "set_fast_mode", enabled: boolean }`
|
||||
- `{ id?, type: "get_available_commands" }`
|
||||
- `{ id?, type: "set_todos", phases: TodoPhase[] }`
|
||||
- `{ id?, type: "set_host_tools", tools: RpcHostToolDefinition[] }`
|
||||
@@ -231,6 +232,19 @@ Local-only slash commands may emit `command_output` frames before completing via
|
||||
|
||||
### `get_state` payload
|
||||
|
||||
`tokensPerSecond` is a number when output throughput is available and `null`
|
||||
otherwise. `fastModeEnabled` reports the session setting, while
|
||||
`fastModeActive` reports the actual computed active state. For Fireworks,
|
||||
`providers.fireworksTier: priority` is a provider-level setting independent of
|
||||
the `/fast` family setting, so `fastModeActive` may remain `true` for an
|
||||
unsupported Fireworks model.
|
||||
|
||||
For direct Anthropic, a provider rejection of `speed: "fast"` uses a sticky
|
||||
fallback scoped by the resolved endpoint and exact model: `fastModeEnabled` may
|
||||
remain `true` while `fastModeActive` is `false`. An explicit `set_fast_mode`
|
||||
enable expresses retry intent and clears that fallback so the provider attempt
|
||||
is re-armed.
|
||||
|
||||
```json
|
||||
{
|
||||
"model": { "provider": "...", "id": "..." },
|
||||
@@ -243,6 +257,9 @@ Local-only slash commands may emit `command_output` frames before completing via
|
||||
"sessionFile": "...",
|
||||
"sessionId": "...",
|
||||
"sessionName": "...",
|
||||
"fastModeEnabled": false,
|
||||
"tokensPerSecond": null,
|
||||
"fastModeActive": false,
|
||||
"autoCompactionEnabled": true,
|
||||
"messageCount": 0,
|
||||
"queuedMessageCount": 0,
|
||||
@@ -275,6 +292,73 @@ Local-only slash commands may emit `command_output` frames before completing via
|
||||
}
|
||||
```
|
||||
|
||||
### `set_fast_mode` payload
|
||||
|
||||
`set_fast_mode` changes whether fast mode is enabled for the session. The
|
||||
request is:
|
||||
|
||||
```json
|
||||
{ "id": "req_fast_on", "type": "set_fast_mode", "enabled": true }
|
||||
```
|
||||
|
||||
On success, `data` always contains both `enabled` and `active`. These are the
|
||||
actual computed values: `enabled` reports the session setting, and `active`
|
||||
reports the resulting active state, including any provider-level Fireworks
|
||||
priority setting:
|
||||
|
||||
For direct Anthropic, an explicit enable also re-arms a provider attempt after
|
||||
the sticky rejection fallback, even when fast mode was already enabled.
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "req_fast_on",
|
||||
"type": "response",
|
||||
"command": "set_fast_mode",
|
||||
"success": true,
|
||||
"data": { "enabled": true, "active": true }
|
||||
}
|
||||
```
|
||||
|
||||
Enabling fast mode on a model without a service-tier family fails with the
|
||||
exact error below:
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "req_fast_on",
|
||||
"type": "response",
|
||||
"command": "set_fast_mode",
|
||||
"success": false,
|
||||
"error": "Fast mode is unavailable for the current model."
|
||||
}
|
||||
```
|
||||
|
||||
Disabling fast mode is idempotent, including on an unsupported model. It
|
||||
succeeds as an off/no-op result, but disabling `/fast` does not override
|
||||
provider-level settings, so a successful disable does not guarantee
|
||||
`active: false`. For example, with an unsupported
|
||||
`fireworks/deepseek-v4-flash` model and `providers.fireworksTier: priority`,
|
||||
the response reports the session setting as disabled while the provider
|
||||
priority keeps the computed active state true:
|
||||
|
||||
```json
|
||||
{
|
||||
"id": "req_fast_off",
|
||||
"type": "response",
|
||||
"command": "set_fast_mode",
|
||||
"success": true,
|
||||
"data": { "enabled": false, "active": true }
|
||||
}
|
||||
```
|
||||
|
||||
The corresponding `get_state` result reports the same computed state:
|
||||
|
||||
```json
|
||||
{
|
||||
"fastModeEnabled": false,
|
||||
"fastModeActive": true
|
||||
}
|
||||
```
|
||||
|
||||
### `set_todos` payload
|
||||
|
||||
Replaces the in-memory todo state for the current session and returns the normalized phase list:
|
||||
|
||||
@@ -385,6 +385,7 @@ thinkingBudgets:
|
||||
| `thinkingBudgets.high` | number | `16384` | Token budget for `high`. |
|
||||
| `thinkingBudgets.xhigh` | number | `32768` | Token budget for `xhigh`. |
|
||||
| `thinkingBudgets.max` | number | `32768` | Token budget for `max`. |
|
||||
| `providers.autoThinkingMaxEffort` | enum | `xhigh` | Highest effort `defaultThinkingLevel: auto` may resolve. `xhigh` keeps the classifier one tier below the top, so only `ultrathink` reaches `max`; `max` lets the classifier bill the top tier on models that expose it. The local on-device classifier stays capped at `xhigh` either way. This governs what `auto` *resolves*: a model whose ladder offers nothing under the ceiling gets no auto level at all, and one that also sets `thinking.requiresEffort` still receives its lowest supported effort from the transport — on a `["max"]` ladder that is `max`, because the model accepts nothing else. |
|
||||
|
||||
### Sampling
|
||||
|
||||
|
||||
@@ -55,7 +55,7 @@
|
||||
| `viewport` | `{ width: number; height: number; scale?: number }` | No | Requested viewport. For headless launch this becomes the initial viewport; for a page it is applied with `page.setViewport()`. `scale` maps to Puppeteer `deviceScaleFactor`. |
|
||||
| `wait_until` | `"load" \| "domcontentloaded" \| "networkidle0" \| "networkidle2"` | No | Navigation wait condition. Defaults to `"load"` where omitted, including `open` navigation and later `tab.goto(...)`. |
|
||||
| `dialogs` | `"accept" \| "dismiss"` | No | Installs a page `dialog` handler that auto-accepts or auto-dismisses dialogs. Omitted means no handler. |
|
||||
| `app` | `{ path?: string; cdp_url?: string; args?: string[]; target?: string }` | No | Selects browser kind. With no `app`, the cmux backend is used when a cmux socket is available (`CMUX_SOCKET_PATH`, gated by the `browser.cmux` setting / `PI_BROWSER_CMUX` override); otherwise the session `browser.headless` setting applies. `app.path` is resolved against the session cwd and used as the executable path for spawn/attach reuse. `app.cdp_url` connects to an existing CDP endpoint. `args` are appended only when spawning `app.path`. `target` is only used for attached/spawned-app page selection. |
|
||||
| `app` | `{ path?: string; cdp_url?: string; args?: string[]; target?: string }` | No | Selects browser kind. With no `app`, a configured `browser.cdpUrl` setting attaches to that endpoint; otherwise the cmux backend is used when a cmux socket is available (`CMUX_SOCKET_PATH`, gated by the `browser.cmux` setting / `PI_BROWSER_CMUX` override); otherwise the session `browser.headless` setting applies. `app.path` is resolved against the session cwd and used as the executable path for spawn/attach reuse. `app.cdp_url` connects to an existing CDP endpoint. `args` are appended only when spawning `app.path`. `target` is only used for attached/spawned-app page selection. |
|
||||
|
||||
### `action: "close"`
|
||||
|
||||
@@ -92,6 +92,7 @@ The tool returns one result per call; no streaming partial output is emitted fro
|
||||
2. `open` resolves browser kind with `resolveBrowserKind()`:
|
||||
- `app.cdp_url` → `{ kind: "connected" }` after trimming trailing slashes.
|
||||
- `app.path` → `{ kind: "spawned" }` after resolving against session cwd.
|
||||
- otherwise, a non-empty `browser.cdpUrl` setting → `{ kind: "connected" }` after trimming whitespace and trailing slashes.
|
||||
- otherwise, `resolveCmuxKind()` → `{ kind: "cmux", socketPath, password?, surface? }` when `CMUX_SOCKET_PATH` is set and cmux is enabled (`browser.cmux` setting, overridable by `PI_BROWSER_CMUX`).
|
||||
- otherwise → `{ kind: "headless", headless: session.settings.get("browser.headless") }`.
|
||||
3. `open` rejects reusing the same tab name across different browser kinds (`sameBrowserKind()`); callers must close first.
|
||||
@@ -163,7 +164,7 @@ The tool returns one result per call; no streaming partial output is emitted fro
|
||||
- **Browser kind**
|
||||
- **Headless**: launches local Chromium with Puppeteer, applies stealth patches, and creates a fresh page per tab.
|
||||
- **Spawned app (`app.path`)**: reuses an existing CDP-enabled process for that executable when possible; otherwise kills same-path processes, spawns the executable with remote debugging enabled, then attaches. No stealth patches are injected.
|
||||
- **Connected browser (`app.cdp_url`)**: attaches to an already-running CDP endpoint. No process ownership; close only disconnects.
|
||||
- **Connected browser (`app.cdp_url`, or the `browser.cdpUrl` setting when the call carries no `app`)**: attaches to an already-running CDP endpoint. No process ownership; close only disconnects.
|
||||
- **Cmux surface (`browser.cmux`)**: with no `app` and a cmux socket available (`CMUX_SOCKET_PATH`, enabled by the `browser.cmux` setting / `PI_BROWSER_CMUX` override), drives a cmux WKWebView surface over a unix-socket JSON-RPC client instead of Puppeteer. No Bun worker and no stealth patches; `open` opens a split (owning that surface), `run` executes via `runCmuxCode()`, and `close` issues `surface.close` for surfaces it owns (leaving the workspace's last surface open).
|
||||
- **Target selection for attached/spawned browsers**
|
||||
- With `app.target`, `pickElectronTarget()` returns the first page whose URL or title contains the case-insensitive substring.
|
||||
|
||||
@@ -147,7 +147,7 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
|
||||
- `limit` and `num_search_results` are collapsed together before dispatch.
|
||||
- Output may include parsed free-text `answer`, `sources`, `requestId`.
|
||||
- **Exa** — `packages/coding-agent/src/web/search/providers/exa.ts`
|
||||
- Availability: env or `agent.db` credential for `exa` admits Exa to the auto chain; settings must not explicitly disable `exa.enabled` or `exa.enableSearch`. Explicit selection (listing `exa` in `providers.webSearchOrder`, or a forced `provider: exa`) reaches Exa even without a credential and falls back to public MCP.
|
||||
- Availability: `EXA_API_KEY` or a stored credential for `exa` (including one added through `/login exa`) admits Exa to the auto chain; settings must not explicitly disable `exa.enabled` or `exa.enableSearch`. Explicit selection (listing `exa` in `providers.webSearchOrder`, or a forced `provider: exa`) reaches Exa even without a credential and falls back to public MCP.
|
||||
- Querying: POST `https://api.exa.ai/search` with the resolved Exa API key, otherwise JSON-RPC `tools/call` against `https://mcp.exa.ai/mcp` for remote MCP tool `web_search_exa`.
|
||||
- `limit` and `num_search_results` are collapsed together before dispatch.
|
||||
- Output: synthesized `answer` from up to 3 result summaries, `sources`, `requestId`.
|
||||
@@ -279,4 +279,4 @@ Streaming: none. `WebSearchTool.execute()` forwards its `AbortSignal` into `exec
|
||||
- `recency` is implemented by Brave, Perplexity, Tavily, SearXNG, Kagi, TinyFish, Firecrawl, xAI, DuckDuckGo, Bing, Yahoo, Startpage, Google, and Mojeek (Ecosia ignores it; Public Web passes it through). The model-facing prompt does not name specific providers.
|
||||
- `packages/coding-agent/src/config/settings-schema.ts` uses the shared `SEARCH_PROVIDER_PREFERENCES` / `SEARCH_PROVIDER_OPTIONS` metadata, so the settings selector and setup wizard expose `auto` plus every provider in the auto chain.
|
||||
- The credential-free scrapers close the auto chain, cheap plain-fetch engines first (`duckduckgo`, `bing`, `yahoo`, `startpage`) and browser-backed ones after (`google`, `ecosia`, `mojeek`); `public` is listed last and never auto-selected.
|
||||
- Exa uses `authStorage.getApiKey("exa")`, then `EXA_API_KEY`, then unauthenticated `https://mcp.exa.ai/mcp` fallback.
|
||||
- `/login exa` stores the pasted key in AuthStorage; Exa resolves credentials in order from `authStorage.getApiKey("exa")`, then `EXA_API_KEY`, then the unauthenticated `https://mcp.exa.ai/mcp` fallback.
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Provider-native compaction failures now surface their transport error instead of silently switching to generic summarization; streaming V2 still falls back to native V1 when available.
|
||||
|
||||
## [17.1.7] - 2026-07-27
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
*/
|
||||
|
||||
import type { Api, CodexCompactionContext, FetchImpl, Model, ProviderSessionState } from "@oh-my-pi/pi-ai";
|
||||
import { isTransientStatus, ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
||||
import * as AIError from "@oh-my-pi/pi-ai/error";
|
||||
import { applyCodexResponsesLiteShape } from "@oh-my-pi/pi-ai/providers/openai-codex/request-transformer";
|
||||
import {
|
||||
createOpenAICodexCompactionRequestContext,
|
||||
@@ -21,6 +21,7 @@ import {
|
||||
parseAzureDeploymentNameMap,
|
||||
resolveOpenAIRequestSetup,
|
||||
} from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
import { captureOpenAIHttpError } from "@oh-my-pi/pi-ai/utils/openai-http";
|
||||
import {
|
||||
CODEX_BASE_URL,
|
||||
getCodexAccountId,
|
||||
@@ -334,18 +335,19 @@ async function attemptCompactionV2Streaming(
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text().catch(() => "");
|
||||
const cause = await captureOpenAIHttpError(response);
|
||||
logger.warn("V2 remote compaction failed", {
|
||||
endpoint,
|
||||
status: response.status,
|
||||
statusText: response.statusText,
|
||||
errorText,
|
||||
errorText: cause.captured.bodyText ?? "",
|
||||
});
|
||||
throw new ProviderHttpError(
|
||||
throw new AIError.ProviderHttpError(
|
||||
`V2 remote compaction failed (${response.status} ${response.statusText})`,
|
||||
response.status,
|
||||
{
|
||||
headers: response.headers,
|
||||
cause,
|
||||
},
|
||||
);
|
||||
}
|
||||
@@ -560,6 +562,10 @@ function formatCompactionV2Failure(event: Record<string, unknown>, type: string)
|
||||
}
|
||||
|
||||
function isRetryableCompactionError(error: Error): boolean {
|
||||
// The gateway's synthetic auth_unavailable is an HTTP 503, but the
|
||||
// captured response cause classifies it as auth. Let provider fallback run
|
||||
// immediately instead of spending the transient retry budget.
|
||||
if (AIError.is(AIError.classify(error), AIError.Flag.AuthFailed)) return false;
|
||||
if (
|
||||
error.name === "AbortError" ||
|
||||
error.name === "TimeoutError" ||
|
||||
@@ -567,8 +573,8 @@ function isRetryableCompactionError(error: Error): boolean {
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
if (error instanceof ProviderHttpError) {
|
||||
return isTransientStatus(error.status);
|
||||
if (error instanceof AIError.ProviderHttpError) {
|
||||
return AIError.isTransientStatus(error.status);
|
||||
}
|
||||
const message = error.message.toLowerCase();
|
||||
return (
|
||||
|
||||
@@ -22,7 +22,7 @@ import {
|
||||
type Usage,
|
||||
withAuth,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { ProviderHttpError } from "@oh-my-pi/pi-ai/error";
|
||||
import * as AIError from "@oh-my-pi/pi-ai/error";
|
||||
import { createOpenAICodexCompactionRequestContext } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import { convertTools } from "@oh-my-pi/pi-ai/providers/openai-responses";
|
||||
import { buildResponsesInput, resolveOpenAICompatPolicy } from "@oh-my-pi/pi-ai/providers/openai-shared";
|
||||
@@ -43,6 +43,7 @@ import {
|
||||
V2_RETAINED_MESSAGE_TOKEN_BUDGET,
|
||||
} from "./compaction-v2-streaming";
|
||||
import type { CompactionEntry, SessionEntry } from "./entries";
|
||||
import { NativeCompactionError } from "./errors";
|
||||
import { isEstimateCacheable, readEstimateCache, writeEstimateCache } from "./message-cache";
|
||||
import { type ConvertToLlm, createBranchSummaryMessage, createCustomMessage, defaultConvertToLlm } from "./messages";
|
||||
import {
|
||||
@@ -202,6 +203,18 @@ export const DEFAULT_COMPACTION_SETTINGS: CompactionSettings = {
|
||||
v2RetainedMessageBudget: V2_RETAINED_MESSAGE_TOKEN_BUDGET,
|
||||
};
|
||||
|
||||
/** Whether a compaction candidate preserves provider-native transport under the effective settings. */
|
||||
export function shouldUseProviderNativeCompaction(
|
||||
model: Model,
|
||||
settings: Pick<CompactionSettings, "remoteEnabled" | "remoteStreamingV2Enabled">,
|
||||
): boolean {
|
||||
if (settings.remoteEnabled === false) return false;
|
||||
return (
|
||||
shouldUseOpenAiRemoteCompaction(model) ||
|
||||
(settings.remoteStreamingV2Enabled !== false && shouldUseCompactionV2Streaming(model))
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Token calculation
|
||||
// ============================================================================
|
||||
@@ -725,7 +738,9 @@ function resolveCompactionEffort(model: Model, level: ThinkingLevel | undefined)
|
||||
*/
|
||||
function createSummarizationError(prefix: string, response: AssistantMessage): Error {
|
||||
const text = `${prefix}: ${response.errorMessage || "Unknown error"}`;
|
||||
return response.errorStatus === undefined ? new Error(text) : new ProviderHttpError(text, response.errorStatus);
|
||||
return response.errorStatus === undefined
|
||||
? new Error(text)
|
||||
: new AIError.ProviderHttpError(text, response.errorStatus);
|
||||
}
|
||||
|
||||
function shouldRetryHandoffWithAutoToolChoice(response: AssistantMessage): boolean {
|
||||
@@ -1332,6 +1347,17 @@ function buildCompactionV2Reasoning(
|
||||
return { effort: reasoning.wireEffort ?? reasoning.requestedEffort, summary: "auto" };
|
||||
}
|
||||
|
||||
/**
|
||||
* Keep any non-auth native protocol failure ahead of authentication failures.
|
||||
* Downstream may retry compaction with another provider only when every native
|
||||
* protocol failed authentication, so a later auth error must not hide an
|
||||
* earlier transport or protocol failure.
|
||||
*/
|
||||
function selectNativeCompactionError(previousError: unknown, nextError: unknown): unknown {
|
||||
if (previousError === undefined) return nextError;
|
||||
return AIError.is(AIError.classify(previousError), AIError.Flag.AuthFailed) ? nextError : previousError;
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate summaries for compaction using prepared data.
|
||||
* Returns CompactionResult - SessionManager adds id/parentId when saving.
|
||||
@@ -1406,6 +1432,7 @@ export async function compact(
|
||||
...recentMessages,
|
||||
];
|
||||
let usedRemoteCompaction = false;
|
||||
let nativeCompactionError: unknown;
|
||||
if (
|
||||
settings.remoteEnabled !== false &&
|
||||
settings.remoteStreamingV2Enabled !== false &&
|
||||
@@ -1467,7 +1494,8 @@ export async function compact(
|
||||
// swallowing it here would downgrade Esc into "fall back to local
|
||||
// summarization" and keep compaction running on an aborted signal.
|
||||
if (signal?.aborted) throw err;
|
||||
logger.warn("OpenAI V2 remote compaction failed, falling back to V1/local summarization", {
|
||||
nativeCompactionError = selectNativeCompactionError(nativeCompactionError, err);
|
||||
logger.warn("OpenAI V2 remote compaction failed, falling back to V1 remote compaction", {
|
||||
error: err instanceof Error ? err.message : String(err),
|
||||
model: model.id,
|
||||
provider: model.provider,
|
||||
@@ -1517,7 +1545,8 @@ export async function compact(
|
||||
// swallowing it here would downgrade Esc into "fall back to local
|
||||
// summarization" and keep compaction running on an aborted signal.
|
||||
if (signal?.aborted) throw err;
|
||||
logger.warn("OpenAI remote compaction failed, falling back to local summarization", {
|
||||
nativeCompactionError = selectNativeCompactionError(nativeCompactionError, err);
|
||||
logger.warn("OpenAI remote compaction failed", {
|
||||
error: err instanceof Error ? err.message : String(err),
|
||||
model: model.id,
|
||||
provider: model.provider,
|
||||
@@ -1526,6 +1555,10 @@ export async function compact(
|
||||
}
|
||||
}
|
||||
|
||||
if (!usedRemoteCompaction && nativeCompactionError !== undefined && !summaryOptions.remoteEndpoint) {
|
||||
throw new NativeCompactionError(nativeCompactionError);
|
||||
}
|
||||
|
||||
// Generate summaries (can be parallel if both needed) and merge into one
|
||||
let summary: string;
|
||||
|
||||
|
||||
@@ -18,6 +18,22 @@ export class CompactionCancelledError extends Error {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A provider-native compaction request failed after every native protocol
|
||||
* available for the selected model was exhausted.
|
||||
*
|
||||
* The cause stays attached so AI error classification can still recognize
|
||||
* authentication failures. Non-auth failures remain distinguishable from
|
||||
* ordinary summarization errors and must not fall through to another provider.
|
||||
*/
|
||||
export class NativeCompactionError extends Error {
|
||||
readonly name = "NativeCompactionError" as const;
|
||||
|
||||
constructor(cause: unknown) {
|
||||
super(cause instanceof Error ? cause.message : String(cause), { cause });
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Outcome of a compaction attempt, surfaced by `CommandController.executeCompaction`
|
||||
* so callers (e.g. the plan-mode approval flow) can distinguish a deliberate abort
|
||||
|
||||
@@ -38,6 +38,7 @@ import {
|
||||
getOpenAIResponsesHistoryPayload,
|
||||
normalizeResponsesToolCallId,
|
||||
} from "@oh-my-pi/pi-ai/utils";
|
||||
import { captureOpenAIHttpError } from "@oh-my-pi/pi-ai/utils/openai-http";
|
||||
import {
|
||||
CODEX_BASE_URL,
|
||||
getCodexAccountId,
|
||||
@@ -840,18 +841,19 @@ export async function requestOpenAiRemoteCompaction(
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text().catch(() => "");
|
||||
const cause = await captureOpenAIHttpError(response);
|
||||
logger.warn("OpenAI remote compaction failed", {
|
||||
endpoint,
|
||||
status: response.status,
|
||||
statusText: response.statusText,
|
||||
errorText,
|
||||
errorText: cause.captured.bodyText ?? "",
|
||||
});
|
||||
throw new ProviderHttpError(
|
||||
`Remote compaction failed (${response.status} ${response.statusText})`,
|
||||
response.status,
|
||||
{
|
||||
headers: response.headers,
|
||||
cause,
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import {
|
||||
compact,
|
||||
createFileOps,
|
||||
DEFAULT_COMPACTION_SETTINGS,
|
||||
NativeCompactionError,
|
||||
prepareCompaction,
|
||||
type SessionEntry,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction";
|
||||
@@ -20,6 +21,7 @@ import {
|
||||
trimRemoteCompactionInputToContextWindow,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction/openai";
|
||||
import * as ai from "@oh-my-pi/pi-ai";
|
||||
import * as AIError from "@oh-my-pi/pi-ai/error";
|
||||
import { getOpenAICodexTransportDetails } from "@oh-my-pi/pi-ai/providers/openai-codex-responses";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
@@ -733,6 +735,36 @@ describe("requestCompactionV2Streaming", () => {
|
||||
|
||||
expect(attempts).toBe(2);
|
||||
});
|
||||
|
||||
test("does not retry and preserves auth_unavailable from V2 HTTP failures", async () => {
|
||||
const model = makeOpenAiModel({
|
||||
remoteCompaction: {
|
||||
enabled: true,
|
||||
v2StreamingEnabled: true,
|
||||
v2Endpoint: "https://compact.example/v1/responses",
|
||||
},
|
||||
});
|
||||
const request = buildCompactionV2Request(
|
||||
model,
|
||||
[{ type: "message", role: "user", content: [{ type: "input_text", text: "real user" }] }],
|
||||
"instructions",
|
||||
);
|
||||
const fetchMock = vi.fn(async () =>
|
||||
Response.json(
|
||||
{ error: { type: "auth_unavailable", message: "no auth available for codex" } },
|
||||
{ status: 503, statusText: "Service Unavailable" },
|
||||
),
|
||||
);
|
||||
|
||||
const error = await requestCompactionV2Streaming(model, "test-key", request, undefined, {
|
||||
fetch: fetchMock,
|
||||
retryWait: async () => {},
|
||||
}).catch(cause => cause);
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
expect(error).toBeInstanceOf(AIError.ProviderHttpError);
|
||||
expect(AIError.is(AIError.classify(error), AIError.Flag.AuthFailed)).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
describe("Responses Lite remote compaction", () => {
|
||||
@@ -1687,6 +1719,78 @@ describe("compact() remote compaction failure handling", () => {
|
||||
expect(JSON.stringify(sameProviderActive?.messagesToSummarize ?? [])).not.toContain("ORIGINAL ALPHA port 4242");
|
||||
});
|
||||
|
||||
test("retains the V2 non-auth failure when the V1 fallback fails authentication", async () => {
|
||||
const preparation = makePreparation();
|
||||
preparation.settings = { ...preparation.settings, remoteStreamingV2Enabled: true };
|
||||
const model = makeOpenAiModel({
|
||||
remoteCompaction: { enabled: true, v2StreamingEnabled: true },
|
||||
});
|
||||
const requestedUrls: string[] = [];
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
const url = String(input);
|
||||
requestedUrls.push(url);
|
||||
return url.endsWith("/responses/compact")
|
||||
? new Response("authentication failed", { status: 401, statusText: "Unauthorized" })
|
||||
: new Response("V2 transport failed", { status: 400, statusText: "Bad Request" });
|
||||
};
|
||||
|
||||
const error = await compact(preparation, model, "test-key", undefined, undefined, { fetch: fetchMock }).catch(
|
||||
cause => cause,
|
||||
);
|
||||
|
||||
expect(requestedUrls.map(url => new URL(url).pathname)).toEqual(["/v1/responses", "/v1/responses/compact"]);
|
||||
expect(error).toBeInstanceOf(NativeCompactionError);
|
||||
expect(error).toMatchObject({ cause: { status: 400 } });
|
||||
expect(AIError.is(AIError.classify(error), AIError.Flag.AuthFailed)).toBe(false);
|
||||
});
|
||||
|
||||
test("keeps native compaction auth-classified when every attempted protocol fails authentication", async () => {
|
||||
const preparation = makePreparation();
|
||||
preparation.settings = { ...preparation.settings, remoteStreamingV2Enabled: true };
|
||||
const model = makeOpenAiModel({
|
||||
remoteCompaction: { enabled: true, v2StreamingEnabled: true },
|
||||
});
|
||||
const requestedUrls: string[] = [];
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
requestedUrls.push(String(input));
|
||||
return new Response("authentication failed", { status: 401, statusText: "Unauthorized" });
|
||||
};
|
||||
|
||||
const error = await compact(preparation, model, "test-key", undefined, undefined, { fetch: fetchMock }).catch(
|
||||
cause => cause,
|
||||
);
|
||||
|
||||
expect(requestedUrls.map(url => new URL(url).pathname)).toEqual(["/v1/responses", "/v1/responses/compact"]);
|
||||
expect(error).toBeInstanceOf(NativeCompactionError);
|
||||
expect(error).toMatchObject({ cause: { status: 401 } });
|
||||
expect(AIError.is(AIError.classify(error), AIError.Flag.AuthFailed)).toBe(true);
|
||||
});
|
||||
|
||||
test("V2 native failure falls back to V1 without generic summarization", async () => {
|
||||
const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local summary"));
|
||||
const preparation = makePreparation();
|
||||
preparation.settings = { ...preparation.settings, remoteStreamingV2Enabled: true };
|
||||
const model = makeOpenAiModel({
|
||||
remoteCompaction: { enabled: true, v2StreamingEnabled: true },
|
||||
});
|
||||
const requestedUrls: string[] = [];
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
const url = String(input);
|
||||
requestedUrls.push(url);
|
||||
if (url.endsWith("/responses/compact")) {
|
||||
return Response.json({ output: [{ type: "compaction", encrypted_content: "enc-v1" }] });
|
||||
}
|
||||
return new Response("V2 unavailable", { status: 502, statusText: "Bad Gateway" });
|
||||
};
|
||||
|
||||
const result = await compact(preparation, model, "test-key", undefined, undefined, { fetch: fetchMock });
|
||||
|
||||
expect(requestedUrls.some(url => url.endsWith("/responses"))).toBe(true);
|
||||
expect(requestedUrls.some(url => url.endsWith("/responses/compact"))).toBe(true);
|
||||
expect(result.shortSummary).toBe("Remote compaction");
|
||||
expect(completeSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test("user abort during the remote compact request rejects without falling back to local summarization", async () => {
|
||||
// Contract: Esc is a cancellation, not a remote failure. Before the fix
|
||||
// the AbortError was swallowed by the fallback catch and compaction kept
|
||||
@@ -1760,16 +1864,54 @@ describe("compact() remote compaction failure handling", () => {
|
||||
});
|
||||
});
|
||||
|
||||
test("remote compact server failure without abort still falls back to local summarization", async () => {
|
||||
test("uses an explicit remote endpoint after provider-native compaction fails", async () => {
|
||||
const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local fallback"));
|
||||
const preparation = makePreparation();
|
||||
preparation.settings = {
|
||||
...preparation.settings,
|
||||
remoteEndpoint: "http://summary.test/v1/chat/completions",
|
||||
remoteStreamingV2Enabled: true,
|
||||
};
|
||||
const model = makeOpenAiModel({
|
||||
remoteCompaction: { enabled: true, v2StreamingEnabled: true },
|
||||
});
|
||||
const requestedUrls: string[] = [];
|
||||
const fetchMock: FetchImpl = async input => {
|
||||
const url = String(input);
|
||||
requestedUrls.push(url);
|
||||
if (url === preparation.settings.remoteEndpoint) {
|
||||
const summary =
|
||||
requestedUrls.filter(requested => requested === url).length === 1
|
||||
? "configured remote history summary"
|
||||
: "configured remote short summary";
|
||||
return Response.json({ choices: [{ message: { content: summary } }] });
|
||||
}
|
||||
return new Response("native compaction unavailable", { status: 400, statusText: "Bad Request" });
|
||||
};
|
||||
|
||||
const result = await compact(preparation, model, "test-key", undefined, undefined, { fetch: fetchMock });
|
||||
|
||||
expect(requestedUrls.map(url => new URL(url).pathname)).toEqual([
|
||||
"/v1/responses",
|
||||
"/v1/responses/compact",
|
||||
"/v1/chat/completions",
|
||||
"/v1/chat/completions",
|
||||
]);
|
||||
expect(result.summary).toContain("configured remote history summary");
|
||||
expect(result.shortSummary).toBe("configured remote short summary");
|
||||
expect(completeSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test("native compaction server failure rejects without generic summarization", async () => {
|
||||
const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local summary"));
|
||||
const fetchMock: FetchImpl = async () =>
|
||||
new Response("nope", { status: 500, statusText: "Internal Server Error" });
|
||||
|
||||
const result = await compact(makePreparation(), makeOpenAiModel(), "test-key", undefined, undefined, {
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
expect(result.summary).toContain("local summary");
|
||||
expect(completeSpy).toHaveBeenCalled();
|
||||
await expect(
|
||||
compact(makePreparation(), makeOpenAiModel(), "test-key", undefined, undefined, {
|
||||
fetch: fetchMock,
|
||||
}),
|
||||
).rejects.toThrow("Remote compaction failed");
|
||||
expect(completeSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -7,12 +7,16 @@
|
||||
- Added first-class parentTurnId support for nested Codex requests, allowing stream options and metadata helpers to accept and safely propagate the initiating turn's ID.
|
||||
- Added preservation of the Codex `encrypted_function_args` plaintext-collaboration marker on replayed function calls, keeping server-marked plaintext tool arguments from being reinterpreted as encrypted on subsequent turns.
|
||||
- Added exact OAuth credential-row resolution by durable credential id. The targeted path refreshes only that row and never ranks, rotates, or falls back to sibling accounts.
|
||||
- Added interactive Exa API-key login through `/login exa`, opening the official API-key dashboard and saving pasted keys to the credential store ([#1798](https://github.com/can1357/oh-my-pi/issues/1798)).
|
||||
- Cursor's modern exec wire protocol is now handled end to end. `agent.proto` models the frames current Cursor CLI builds emit — the seven Pi tools (`ExecServerMessage` 45-51), hooks, subagents, allowlist prechecks, MCP state, smart-mode classification, canvas diagnostics, conversation search, agent-store conflicts and git diff — and every one of them gets a typed answer. The Pi frames run their local equivalents (`read`/`bash`/`edit`/`write`/`grep`/`glob`); the rest answer with the error, not-found or empty-but-valid variant that is actually true of this client. Frames this build cannot name at all now raise `ExecClientControlMessage.throw` with `unknown_exec_variant`, and recognised frames with no truthful answer (`git_diff_request`, whose `GetDiffResponse` has no error variant) raise `exec_variant_unsupported`, instead of a silent ack that leaves the server waiting.
|
||||
- `lsp` is advertised in the MCP tool catalog again. It was filtered out as a Cursor-native tool, but the native `diagnostics` frame covers one of roughly ten LSP actions, so the other nine were unreachable.
|
||||
|
||||
### Changed
|
||||
|
||||
- Codex turn metadata now reserves the codex-rs `code_mode_tool_names` key, preventing caller-supplied client metadata extras from colliding with the core-owned field.
|
||||
- Codex SSE requests to the official endpoint now use zstd-compressed bodies by default to match the official client, which can be disabled with PI_CODEX_ZSTD=0.
|
||||
- API-key validation now preserves provider HTTP status and retry headers, allowing authentication, rate-limit, and server failures to retain their original error classifications.
|
||||
- The Cursor Pi arg translation (`piReadPath`, `piJoinPath`, `piLsPath`, `piEscapeRegexLiteral`, `piLimit`) moved to `providers/cursor-pi-args`, re-exported from `providers/cursor/exec-modern` so existing imports are unaffected. The legacy pi shim shares these helpers and is compiled into the bundled virtual module registry, where a nested `providers/<dir>/<mod>` specifier is unresolvable under bunfs — and importing them from the exec module would drag the whole protobuf graph in for two string functions.
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -26,6 +30,31 @@
|
||||
- Fixed direct Anthropic Claude Opus requests failing with HTTP 400 when the endpoint rejects strict tool fields.
|
||||
- Fixed usage-based credential ranking for Anthropic accounts where a missing long-window (7-day) metric was incorrectly treated as a short-window metric.
|
||||
- Fixed legacy Codex usage blocks continuing to gate all models after per-meter backoff was introduced, splitting the old shared scope into independent chat and spark blocks while maintaining backward compatibility with older clients and database schemas.
|
||||
- Fixed Anthropic retry loops ignoring `maxRetryDelayMs` for long server `retry-after` hints, so over-budget delays surface immediately without losing response details or abort cleanup ([#7003](https://github.com/can1357/oh-my-pi/issues/7003)).
|
||||
- Added interactive xAI API-key login with key validation through the xAI models endpoint.
|
||||
- Fixed Google Gemini and Vertex tool declarations carrying numeric, boolean, object-valued, or mixed `enum` arrays that the Google Schema wire type cannot represent. Unsupported enums are omitted while valid string enums remain constrained.
|
||||
- Umans usage provider: fetches `GET /v1/usage` and surfaces the rolling 5h request window + concurrency limits in `/usage`, `omp usage`, and the TUI status bar.
|
||||
- Fixed ranged legacy Cursor reads reporting the returned window byte length as the full file size.
|
||||
- Updated the Cursor client build advertisement to activate the modern exec-frame protocol handled by this provider.
|
||||
- Fixed a windowed Cursor `read` reporting the window's line count as the file's. `total_lines` and `file_size` were derived from the payload, which is the whole file only for an unranged read — a 20-line page of a 100-line file answered `total_lines: 20`, which a paginating server reads as the end of the file. The count now comes from the read's own record of the file (`details.meta.truncation.totalLines`), falling back to counting the payload when the read returned the file whole.
|
||||
- Fixed a `pi_grep` that hit the native backend's internal match ceiling answering as an unqualified success. `GrepTool` folds that cap into the flat `details.truncated` alone, setting neither `details.truncation` nor `perFileLimitReached` — the two fields the Pi result was built from — so the one truncation a caller can neither detect nor page around was the one it was never told about. The flat flag is now translated into a `PiTruncation`, and only when no specific cap already reported itself.
|
||||
- Fixed a `pi_grep` frame's `context` and `limit` vanishing from the transcript. The bridge honors both by building a scoped `grep`, but neither is expressible in the model-facing schema, so the synthesized block recorded a plain pattern/path search — replaying a context-widened or capped search as an ordinary grep sitting beside output no ordinary grep produces. Both are now recorded on the block.
|
||||
- Fixed a Cursor MCP resource listing shrinking to a count in the transcript. The full URI/name/mime catalog goes out on the wire, but the paired local result recorded `Listed N MCP resource(s)` — and rebuilt history is serialized from that result, so one reload later the model knew it had seen N resources and could name none of them. The paired result now lists what the answer carried.
|
||||
- Fixed the `pi_read` range translation padding the slice it asks for. `piReadPath` composed a plain `:N+K` selector, which the local `read` tool expands by one leading and three trailing context line — so a frame naming offset 5/limit 20 received lines 4-27. Ranged Pi reads now compose `:raw:N+K`; the wire result is an opaque output string, so the line-number gutter `raw` also drops carries nothing the contract needs.
|
||||
- Fixed four Cursor exec frames answering with a result whose oneof was never set. In proto3 that is not an empty result — the server reads it as "the tool ran and produced nothing", indistinguishable from real success. `listMcpResourcesExecResult`, `readMcpResourceExecResult`, `recordScreenResult` and `computerUseResult` now send `ListMcpResourcesSuccess{resources: []}`, `ReadMcpResourceNotFound{uri}`, `RecordScreenFailure` and `ComputerUseError` respectively.
|
||||
- The MCP resource frames now answer from the host instead of a fixed verdict. `CursorExecHandlers` gained `listMcpResources`/`readMcpResource`, so a host holding live MCP connections advertises them; the empty catalog and `not_found` above remain the answer when no handler is supplied. A handler that throws surfaces as `ListMcpResourcesError`/`ReadMcpResourceError` rather than collapsing into "none exist", which the model cannot retry. A read carrying `download_path` forwards it and answers with `ReadMcpResourceSuccess.download_path` and no content, which is what that mode means.
|
||||
- Fixed Cursor `connect_scm` calls losing their repository and settling on a fabricated verdict. The target rides in the `ConnectScmArgs.target` oneof, so reading a flat `github` property always saw `undefined`; and the authoritative `success`/`error`/`rejected` result only arrives on the completion frame, so answering at the announcement persisted a fixed failure for every call — including the ones the server went on to accept. The block now opens on the start frame and settles from the completion's decoded result.
|
||||
- Fixed interleaved Cursor tool calls corrupting each other. The stream decoder tracked a single "current" block and settled it on any `toolCallCompleted`, ignoring the envelope's `call_id`: a completion for one call closed whichever block happened to be open and paired it with the wrong result, and `start A, start B` orphaned A entirely so its own completion settled B while A was never paired — which strips the whole interaction from every rebuilt transcript. Open blocks are now retained per envelope `call_id`, and end-of-stream closes all of them rather than only the last.
|
||||
- Fixed a Cursor `search_conversations` call leaving no transcript block. The frame is answered from a fixed verdict, so nothing downstream pairs a result for it, and an unpaired call takes its whole interaction out of every rebuilt transcript.
|
||||
- Fixed a Cursor `read_mcp_resource` call leaving no transcript block. The frame runs locally — and in download mode writes a workspace file — but synthesized no tool call and paired no result, so the read was invisible in the UI and absent from every rebuilt history; a resource download could mutate the workspace with nothing on record. The frame now synthesizes a `read_mcp_resource` block (not `read`: it is a remote MCP operation, and the name drives rendering and prune semantics) and pairs a result on success, not-found and error alike. Frames answered without a handler still synthesize nothing, since nothing ran.
|
||||
- Fixed a Cursor `list_mcp_resources` call leaving no transcript block. The model consumed the catalog, but the frame synthesized no tool call and paired no result — its streamed `ListMcpResourcesToolCall` announcement was equally unrecognized — so the listing was invisible in the UI and absent from every rebuilt history. Frames a handler answered now synthesize a `list_mcp_resources` block and pair a result derived from the same answer that went on the wire; frames answered from the fixed no-handler catalog still synthesize nothing, since nothing ran.
|
||||
- Fixed an unavailable `pi_edit`/`pi_write` answering with the error variant. Both results model refusal and failure as separate oneof cases, and a denial reported as `error` reads as "the tool ran and broke" — inviting a retry of an operation that was never permitted. A frame whose tool is not granted, or whose handler produced nothing, now answers with `PiEditExecRejected`/`PiWriteExecRejected`; execution failures keep the error variant.
|
||||
- Fixed a Cursor MCP approval probe actually running the tool. A modern `mcpArgs` frame carrying `smart_mode_approval_only` asks only whether a call would be permitted, not for the call itself. The decoder dropped the flag, so the frame ran a side-effecting MCP tool the user had not been asked about, then ran it again when the real call followed. The flag is now carried through and the probe is answered from the host's policy without executing: approved only for a definite allow, refused for a deny, for a mode that demands a prompt the frame cannot raise, and for a tool the session does not have. No transcript block is synthesized either, since nothing ran.
|
||||
- Fixed the Cursor stream's end-of-transport cleanup erasing the arguments of every block still open. Blocks whose args arrive whole (todo, connect-SCM, MCP) never feed the streamed partial-JSON buffer, and reparsing an absent buffer yields `{}`, so a truncated or disconnected turn rebuilt those calls with no arguments at all. Only blocks that actually streamed their args are reparsed now.
|
||||
- Fixed a Cursor stream dying mid-turn stranding the call it left open. `connect_scm` and native todo blocks are stamped resolved the moment they open, so the agent loop synthesizes no placeholder and only their completion frame pairs a result — a transport that closed first left the card animating and the call unpaired, which takes the whole interaction out of every rebuilt transcript. The terminal-error path now closes open blocks and pairs those server-owned calls with an interrupted result; the flush ran only on clean completion before, which is not the path a dying stream takes. Exec-settled MCP blocks are left alone, since the dispatch that ran them owns their result.
|
||||
- Fixed the Pi exec frames displaying a different operation than the one they run. The provider synthesized its transcript block from a second, hand-rolled translation of the frame args, so `pi_read`'s `offset`/`limit` were shown as a whole-file read, `pi_grep`'s `literal` pattern as an unescaped regex, and `pi_find`'s path/glob join differed from the executed one. Both sides now share a single translation.
|
||||
- Fixed the streamed `pi_*_tool_call` announcements that modern builds send alongside each exec frame being unrecognized. The exec channel already synthesizes those blocks when it runs the tool; the duplicate was avoided only because the decoder recognized none of the variants, which would have started double-rendering as soon as any one was added.
|
||||
- Fixed `pi_bash` results reaching Cursor clipped with no truncation notice. Two truncation records exist locally: `read`/`grep` set `details.truncation`, which carries an explicit `truncated` flag, while `bash` sets `details.meta.truncation`, whose record has no such flag — its presence is the signal. `piTruncation` read only the first shape and required the flag, so every real Bash truncation was dropped and the server was told the clipped output was complete. Both shapes now translate, and an explicit `truncated: false` still suppresses the field.
|
||||
|
||||
## [17.1.8] - 2026-07-28
|
||||
|
||||
|
||||
@@ -65,6 +65,7 @@ import {
|
||||
} from "./usage/openai-codex-reset";
|
||||
import { opencodeGoUsageProvider } from "./usage/opencode-go";
|
||||
import { syntheticUsageProvider } from "./usage/synthetic";
|
||||
import { umansUsageProvider } from "./usage/umans";
|
||||
import { xaiOauthUsageProvider } from "./usage/xai-oauth";
|
||||
import { zaiRankingStrategy, zaiUsageProvider } from "./usage/zai";
|
||||
|
||||
@@ -660,6 +661,7 @@ const DEFAULT_USAGE_PROVIDERS: UsageProvider[] = [
|
||||
ollamaCloudUsageProvider,
|
||||
claudeUsageProvider,
|
||||
zaiUsageProvider,
|
||||
umansUsageProvider,
|
||||
opencodeGoUsageProvider,
|
||||
githubCopilotUsageProvider,
|
||||
cursorUsageProvider,
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import type { CapturedHttpErrorResponse } from "../utils/http-inspector";
|
||||
import { AbortError } from "./abort";
|
||||
|
||||
/** Prefix on errors raised when an Anthropic SSE stream envelope is malformed. */
|
||||
export const STREAM_ENVELOPE_ERROR_PREFIX = "Anthropic stream envelope error:";
|
||||
@@ -68,6 +69,22 @@ export class OpenAIHttpError extends ProviderHttpError {
|
||||
}
|
||||
}
|
||||
|
||||
/** Non-2xx response from the Anthropic API. */
|
||||
const DEFAULT_ANTHROPIC_ERROR_BODY_READ_TIMEOUT_MS = 5_000;
|
||||
const MAX_ANTHROPIC_ERROR_BODY_BYTES = 64 * 1024;
|
||||
const ANTHROPIC_ERROR_BODY_TRUNCATION_MARKER = "\n[Response body truncated after 64 KiB]";
|
||||
let anthropicErrorBodyReadTimeoutMs = DEFAULT_ANTHROPIC_ERROR_BODY_READ_TIMEOUT_MS;
|
||||
|
||||
/** Test-only control for the otherwise bounded Anthropic error-body drain. */
|
||||
export const __anthropicApiErrorForTesting = {
|
||||
setBodyReadTimeoutMs(timeoutMs: number | undefined): void {
|
||||
if (timeoutMs !== undefined && (!Number.isFinite(timeoutMs) || timeoutMs < 0)) {
|
||||
throw new RangeError("Anthropic error-body timeout must be a non-negative finite number.");
|
||||
}
|
||||
anthropicErrorBodyReadTimeoutMs = timeoutMs ?? DEFAULT_ANTHROPIC_ERROR_BODY_READ_TIMEOUT_MS;
|
||||
},
|
||||
};
|
||||
|
||||
/** Non-2xx response from the Anthropic API. */
|
||||
export class AnthropicApiError extends ProviderHttpError {
|
||||
declare readonly headers: Headers;
|
||||
@@ -79,9 +96,87 @@ export class AnthropicApiError extends ProviderHttpError {
|
||||
this.requestId = headers.get("request-id");
|
||||
}
|
||||
|
||||
static async fromResponse(response: Response): Promise<AnthropicApiError> {
|
||||
const body = await response.text().catch(() => "");
|
||||
const detail = body.trim() || "status code (no body)";
|
||||
static async fromResponse(response: Response, signal?: AbortSignal): Promise<AnthropicApiError> {
|
||||
// Avoid getReader() throwing when another consumer already owns the body.
|
||||
const reader = response.body?.locked ? undefined : response.body?.getReader();
|
||||
|
||||
if (!reader) {
|
||||
if (signal?.aborted) throw new AbortError("Request was aborted.");
|
||||
const detail = "status code (no body)";
|
||||
return new AnthropicApiError(response.status, `${response.status} ${detail}`, response.headers);
|
||||
}
|
||||
|
||||
let aborted = false;
|
||||
let timedOut = false;
|
||||
let readerCancelled = false;
|
||||
const cancelReader = () => {
|
||||
if (readerCancelled) return;
|
||||
readerCancelled = true;
|
||||
void reader.cancel().catch(() => {});
|
||||
};
|
||||
const onAbort = () => {
|
||||
if (aborted) return;
|
||||
aborted = true;
|
||||
cancelReader();
|
||||
};
|
||||
if (signal?.aborted) onAbort();
|
||||
else signal?.addEventListener("abort", onAbort, { once: true });
|
||||
|
||||
const deadline = performance.now() + anthropicErrorBodyReadTimeoutMs;
|
||||
let timeout: Timer | undefined;
|
||||
timeout = setTimeout(() => {
|
||||
timedOut = true;
|
||||
cancelReader();
|
||||
}, anthropicErrorBodyReadTimeoutMs);
|
||||
|
||||
let capturedBytes = 0;
|
||||
let truncated = false;
|
||||
const bodyChunks: string[] = [];
|
||||
try {
|
||||
const decoder = new TextDecoder();
|
||||
let cleanEof = false;
|
||||
while (!aborted && !timedOut) {
|
||||
if (performance.now() >= deadline) {
|
||||
timedOut = true;
|
||||
cancelReader();
|
||||
break;
|
||||
}
|
||||
|
||||
let result: { readonly done: boolean; readonly value?: Uint8Array };
|
||||
try {
|
||||
result = await reader.read();
|
||||
} catch {
|
||||
break;
|
||||
}
|
||||
if (aborted || timedOut) break;
|
||||
if (result.done) {
|
||||
cleanEof = true;
|
||||
break;
|
||||
}
|
||||
|
||||
const chunk = result.value;
|
||||
if (!chunk) break;
|
||||
const bytesToCapture = Math.min(MAX_ANTHROPIC_ERROR_BODY_BYTES - capturedBytes, chunk.byteLength);
|
||||
if (bytesToCapture > 0) {
|
||||
bodyChunks.push(decoder.decode(chunk.subarray(0, bytesToCapture), { stream: true }));
|
||||
capturedBytes += bytesToCapture;
|
||||
}
|
||||
if (bytesToCapture < chunk.byteLength) {
|
||||
truncated = true;
|
||||
cancelReader();
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (aborted || signal?.aborted) throw new AbortError("Request was aborted.");
|
||||
if (cleanEof) bodyChunks.push(decoder.decode());
|
||||
if (truncated) bodyChunks.push(ANTHROPIC_ERROR_BODY_TRUNCATION_MARKER);
|
||||
} finally {
|
||||
if (timeout !== undefined) clearTimeout(timeout);
|
||||
signal?.removeEventListener("abort", onAbort);
|
||||
reader.releaseLock();
|
||||
}
|
||||
|
||||
const detail = bodyChunks.join("").trim() || "status code (no body)";
|
||||
return new AnthropicApiError(response.status, `${response.status} ${detail}`, response.headers);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -43,6 +43,12 @@ export interface AnthropicRequestOptions {
|
||||
timeout?: number;
|
||||
/** Per-request retry budget override. */
|
||||
maxRetries?: number;
|
||||
/**
|
||||
* Maximum delay in milliseconds to wait for a server-directed retry. If the
|
||||
* server's `retry-after` hint exceeds this value, the retry is declined and
|
||||
* the original error is surfaced. Non-positive values disable the cap. Defaults to 60000.
|
||||
*/
|
||||
maxRetryDelayMs?: number;
|
||||
/** Per-request headers merged after client defaults. */
|
||||
headers?: Record<string, string>;
|
||||
}
|
||||
@@ -74,6 +80,12 @@ export interface AnthropicClientOptions {
|
||||
authToken?: string | null;
|
||||
baseURL?: string | null;
|
||||
maxRetries?: number;
|
||||
/**
|
||||
* Maximum delay in milliseconds to wait for a server-directed retry. If the
|
||||
* server's `retry-after` hint exceeds this value, the retry is declined and
|
||||
* the original error is surfaced. Non-positive values disable the cap. Defaults to 60000.
|
||||
*/
|
||||
maxRetryDelayMs?: number;
|
||||
/** Pre-response timeout in milliseconds. Defaults to 10 minutes. */
|
||||
timeout?: number;
|
||||
defaultHeaders?: Record<string, string>;
|
||||
@@ -97,7 +109,7 @@ function shouldRetryResponse(response: Response): boolean {
|
||||
}
|
||||
|
||||
/** Server-suggested delay (`retry-after-ms`, then `retry-after` seconds or HTTP date). */
|
||||
export function retryDelayFromHeaders(headers: Headers | undefined): number | undefined {
|
||||
export function retryDelayFromHeaders(headers: Pick<Headers, "get"> | undefined): number | undefined {
|
||||
if (!headers) return undefined;
|
||||
const retryAfterMs = headers.get("retry-after-ms");
|
||||
if (retryAfterMs) {
|
||||
@@ -216,6 +228,7 @@ export class AnthropicMessagesClient implements AnthropicMessagesClientLike {
|
||||
const callerSignal = options?.signal;
|
||||
const timeoutMs = options?.timeout ?? opts.timeout ?? DEFAULT_TIMEOUT_MS;
|
||||
const maxRetries = Math.max(0, options?.maxRetries ?? opts.maxRetries ?? DEFAULT_MAX_RETRIES);
|
||||
const maxRetryDelayMs = options?.maxRetryDelayMs ?? opts.maxRetryDelayMs ?? 60_000;
|
||||
const url = `${opts.baseURL ?? "https://api.anthropic.com"}${path}`;
|
||||
const headers = this.#buildHeaders(options?.headers);
|
||||
const body = JSON.stringify(params);
|
||||
@@ -239,11 +252,20 @@ export class AnthropicMessagesClient implements AnthropicMessagesClientLike {
|
||||
if (response.ok) return response;
|
||||
|
||||
if (attempt < maxRetries && shouldRetryResponse(response)) {
|
||||
// Bound the server-directed wait: an over-cap `retry-after` declines
|
||||
// the retry and surfaces the original error (status/body/headers
|
||||
// intact) so higher-level recovery can run. A non-positive cap disables enforcement.
|
||||
// Checked before draining the body so `fromResponse` can still read it.
|
||||
const headerDelayMs = retryDelayFromHeaders(response.headers);
|
||||
if (headerDelayMs !== undefined && maxRetryDelayMs > 0 && headerDelayMs > maxRetryDelayMs) {
|
||||
throw await AIError.AnthropicApiError.fromResponse(response, callerSignal);
|
||||
}
|
||||
await response.body?.cancel().catch(() => {});
|
||||
await this.#backoff(attempt, response.headers, callerSignal);
|
||||
continue;
|
||||
}
|
||||
throw await AIError.AnthropicApiError.fromResponse(response);
|
||||
|
||||
throw await AIError.AnthropicApiError.fromResponse(response, callerSignal);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -60,19 +60,18 @@ import { isFoundryEnabled } from "../utils/foundry";
|
||||
import { finalizeErrorMessage, type RawHttpRequestDump } from "../utils/http-inspector";
|
||||
import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
|
||||
import { notifyProviderResponse } from "../utils/provider-response";
|
||||
import { getHeadersFromError, getRetryAfterMsFromHeaders } from "../utils/retry-after";
|
||||
import { COMBINATOR_KEYS, NO_STRICT, toolWireSchema } from "../utils/schema";
|
||||
import { spillToDescription } from "../utils/schema/spill";
|
||||
import { createSdkStreamRequestOptions } from "../utils/sdk-stream-timeout";
|
||||
import { notifyRawSseEvent } from "../utils/sse-debug";
|
||||
import { isForcedToolChoice } from "../utils/tool-choice";
|
||||
import {
|
||||
AnthropicApiError,
|
||||
AnthropicConnectionTimeoutError,
|
||||
type AnthropicFetchOptions,
|
||||
AnthropicMessagesClient,
|
||||
type AnthropicMessagesClientLike,
|
||||
calculateAnthropicRetryDelayMs,
|
||||
retryDelayFromHeaders,
|
||||
} from "./anthropic-client";
|
||||
import {
|
||||
type ToolInputSchema as AnthropicToolInputSchema,
|
||||
@@ -450,6 +449,20 @@ export function clearAnthropicFastModeFallback(
|
||||
(value as AnthropicProviderSessionState).fastModeDisabled = false;
|
||||
}
|
||||
}
|
||||
/**
|
||||
* Whether the direct Anthropic model's endpoint-scoped fast-mode fallback is
|
||||
* currently active. Reading the map directly is intentional: inspection must
|
||||
* not materialize a state entry for a model that has never streamed.
|
||||
*/
|
||||
export function isAnthropicFastModeFallbackDisabled(
|
||||
providerSessionState: Map<string, ProviderSessionState> | undefined,
|
||||
model: Model<Api>,
|
||||
): boolean {
|
||||
if (!providerSessionState || model.provider !== "anthropic" || model.api !== "anthropic-messages") return false;
|
||||
const baseUrl = resolveAnthropicBaseUrl(model as Model<"anthropic-messages">) ?? "https://api.anthropic.com";
|
||||
const key = anthropicProviderSessionStateKey(baseUrl, model.id);
|
||||
return (providerSessionState.get(key) as AnthropicProviderSessionState | undefined)?.fastModeDisabled ?? false;
|
||||
}
|
||||
|
||||
function hasStrictAnthropicTools(params: MessageCreateParamsStreaming): boolean {
|
||||
return params.tools?.some(tool => tool.strict === true) ?? false;
|
||||
@@ -1155,6 +1168,7 @@ export type AnthropicClientOptionsArgs = {
|
||||
thinkingDisplay?: AnthropicThinkingDisplay;
|
||||
disableStrictTools?: boolean;
|
||||
fetch?: FetchImpl;
|
||||
maxRetryDelayMs?: number;
|
||||
claudeCodeSessionId?: string;
|
||||
};
|
||||
|
||||
@@ -1164,6 +1178,7 @@ export type AnthropicClientOptionsResult = {
|
||||
authToken?: string | null;
|
||||
baseURL?: string;
|
||||
maxRetries: number;
|
||||
maxRetryDelayMs?: number;
|
||||
defaultHeaders: Record<string, string>;
|
||||
fetch?: FetchImpl;
|
||||
fetchOptions?: AnthropicFetchOptions;
|
||||
@@ -1952,6 +1967,7 @@ const streamAnthropicOnce = (
|
||||
thinkingEnabled: options?.thinkingEnabled,
|
||||
thinkingDisplay: options?.thinkingDisplay,
|
||||
fetch: options?.fetch,
|
||||
maxRetryDelayMs: options?.maxRetryDelayMs,
|
||||
claudeCodeSessionId: options?.sessionId ?? extractClaudeMetadataSessionId(options?.metadata?.user_id),
|
||||
disableStrictTools,
|
||||
});
|
||||
@@ -2699,10 +2715,14 @@ const streamAnthropicOnce = (
|
||||
// Honor the server's retry hint (`retry-after-ms`/`retry-after`) on
|
||||
// 429/529-style failures: retrying sooner than the server asked is a
|
||||
// guaranteed failure that just burns the retry budget.
|
||||
const headerDelayMs =
|
||||
streamFailure instanceof Error && streamFailure instanceof AnthropicApiError
|
||||
? retryDelayFromHeaders(streamFailure.headers)
|
||||
: undefined;
|
||||
const headerDelayMs = getRetryAfterMsFromHeaders(getHeadersFromError(streamFailure));
|
||||
// Bound the server-directed wait so a multi-hour `retry-after` cannot
|
||||
// park the provider stream before higher-level recovery runs. A non-positive cap
|
||||
// disables the bound; an over-cap hint surfaces the original error immediately.
|
||||
const maxRetryDelayMs = options?.maxRetryDelayMs ?? 60_000;
|
||||
if (headerDelayMs !== undefined && maxRetryDelayMs > 0 && headerDelayMs > maxRetryDelayMs) {
|
||||
throw streamFailure;
|
||||
}
|
||||
const delayMs = headerDelayMs !== undefined ? Math.max(headerDelayMs, backoffDelayMs) : backoffDelayMs;
|
||||
if (options?.providerRetryWait) {
|
||||
await options.providerRetryWait(delayMs, options.signal);
|
||||
@@ -2847,6 +2867,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
thinkingEnabled = false,
|
||||
thinkingDisplay,
|
||||
isOAuth,
|
||||
maxRetryDelayMs,
|
||||
claudeCodeSessionId,
|
||||
disableStrictTools: disableStrictToolsOverride,
|
||||
} = args;
|
||||
@@ -2917,6 +2938,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: copilotApiKey,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
@@ -2964,6 +2986,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: null,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
@@ -2982,6 +3005,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: null,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
@@ -3004,6 +3028,7 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
authToken: oauthToken ? apiKey : undefined,
|
||||
baseURL: baseUrl,
|
||||
maxRetries: 5,
|
||||
maxRetryDelayMs,
|
||||
defaultHeaders,
|
||||
fetch: cchFetch,
|
||||
fetchOptions,
|
||||
|
||||
@@ -0,0 +1,135 @@
|
||||
/**
|
||||
* Translate a Pi frame's args into the local tool kwargs that run it.
|
||||
*
|
||||
* Shared deliberately by three consumers: the provider synthesizes a display
|
||||
* block from these, the coding-agent bridge executes with them, and the legacy
|
||||
* pi shim performs the identical translation for the old wire. Separate
|
||||
* hand-rolled copies drift, and the drift is invisible — the transcript shows
|
||||
* one operation while a different one runs.
|
||||
*
|
||||
* Kept apart from `cursor/exec-modern.ts` on purpose: these are pure
|
||||
* string/path functions with no protobuf coupling, while that module pulls in
|
||||
* `@bufbuild/protobuf` and the generated `agent_pb` graph. The legacy shim is
|
||||
* compiled into the bundled virtual module registry, so importing it from a
|
||||
* nested path would drag the whole exec implementation in with it — and
|
||||
* `./providers/*` is a single-segment wildcard export that cannot serve a
|
||||
* nested specifier under bunfs (issue #3442).
|
||||
*
|
||||
* Every `optional int32` here is presence-sensitive: `0` is a supplied value,
|
||||
* not "unset", so it must never be folded into a default.
|
||||
*/
|
||||
|
||||
import * as path from "node:path";
|
||||
|
||||
/**
|
||||
* A `pi_read` range composed onto the path as `read`'s inline `:raw:N+K`
|
||||
* selector.
|
||||
*
|
||||
* `read` exposes no range kwargs, so an uncomposed range reads the whole file.
|
||||
* `offset` is a 1-indexed start clamped like the reference's
|
||||
* `Math.max(0, offset - 1)` over 0-indexed lines; `limit` is a line count.
|
||||
* `null` marks a present `limit: 0` — zero lines, which no selector expresses
|
||||
* and which must not degrade into a whole-file read.
|
||||
*
|
||||
* The range is `raw` because a plain `:N+K` deliberately pads with one leading
|
||||
* and three trailing context lines: helpful for a human reading a snippet,
|
||||
* wrong for a caller that asked for exactly `limit` lines from `offset`. The
|
||||
* wire result is an opaque `output` string, so the hashline and line-number
|
||||
* gutter that `raw` also drops carry nothing the frame's contract needs.
|
||||
* A range-free read keeps the ordinary form — whole-file reads want them.
|
||||
*/
|
||||
export function piReadPath(readPath: string, offset?: number, limit?: number): string | null {
|
||||
if (limit !== undefined && Math.floor(limit) <= 0) return null;
|
||||
const start = offset !== undefined ? Math.max(1, Math.floor(offset)) : undefined;
|
||||
const count = limit !== undefined ? Math.floor(limit) : undefined;
|
||||
if (start === undefined && count === undefined) return readPath;
|
||||
if (start === undefined) return `${readPath}:raw:1+${count}`;
|
||||
return count === undefined ? `${readPath}:raw:${start}-` : `${readPath}:raw:${start}+${count}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* The same range as {@link piReadPath}, rendered for a transcript block rather
|
||||
* than for execution.
|
||||
*
|
||||
* Differs only at `limit: 0`, where `piReadPath` returns `null` because no
|
||||
* selector reads zero lines and the frame is answered with empty output
|
||||
* directly. The block still has to say so: falling back to the bare path there
|
||||
* would record a whole-file read whose result is empty, which is the widest
|
||||
* possible gap between what a rebuilt transcript shows and what happened.
|
||||
* `+0` is never executed — it exists to be read.
|
||||
*/
|
||||
export function piReadDisplayPath(readPath: string, offset?: number, limit?: number): string {
|
||||
const composed = piReadPath(readPath, offset, limit);
|
||||
if (composed !== null) return composed;
|
||||
const start = offset !== undefined ? Math.max(1, Math.floor(offset)) : 1;
|
||||
return `${readPath}:raw:${start}+0`;
|
||||
}
|
||||
|
||||
/**
|
||||
* A legacy `grep` frame's pagination `offset` as the local tool's file `skip`.
|
||||
*
|
||||
* `grep` paginates by file and reports "use skip=N for the next page" in that
|
||||
* same unit, so the offset maps across directly. A present `0` means "start at
|
||||
* the beginning", which is the unskipped search rather than a skip of zero.
|
||||
*
|
||||
* Shared because both the executing bridge and the provider's transcript
|
||||
* synthesis need it: a block showing an unskipped search beside a result from
|
||||
* a later file window misreports what was searched.
|
||||
*/
|
||||
export function piGrepSkip(offset?: number): number | undefined {
|
||||
return offset !== undefined && offset > 0 ? Math.floor(offset) : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Join a Pi frame's optional `path` with the `glob`/`pattern` it scopes.
|
||||
*
|
||||
* The local `grep`/`glob` tools take one combined path spec. An absolute
|
||||
* pattern ignores the path, and an absent or `.` path leaves the pattern
|
||||
* standing alone rather than building a `./`- or `//`-prefixed spec.
|
||||
*
|
||||
* Uses `node:path` rather than string surgery so Windows absolutes (`C:\…`,
|
||||
* UNC) are recognised and separators stay normalized.
|
||||
*/
|
||||
export function piJoinPath(basePath: string | undefined, pattern: string): string {
|
||||
if (path.isAbsolute(pattern)) return pattern;
|
||||
if (!basePath || basePath === ".") return pattern;
|
||||
return path.join(basePath, pattern);
|
||||
}
|
||||
|
||||
/**
|
||||
* The path a `pi_ls` frame lists.
|
||||
*
|
||||
* The frame's `limit` is deliberately NOT mapped. It caps directory *entries*
|
||||
* (the reference does a flat `readdir` and slices the entry array), while the
|
||||
* local `read` tool renders a depth-2 tree with per-directory caps and elision
|
||||
* summaries and applies a selector as a *rendered line* slice. Nested rows,
|
||||
* headers and "N more" lines all count toward that slice, so `:1+K` would cap
|
||||
* a different unit while looking honored — worse than leaving it unset, which
|
||||
* at least reports the local listing's own truncation faithfully.
|
||||
*/
|
||||
export function piLsPath(basePath: string | undefined): string {
|
||||
return basePath || ".";
|
||||
}
|
||||
|
||||
/** Escape a literal string so the regex-only local `grep` tool matches it verbatim. */
|
||||
export function piEscapeRegexLiteral(value: string): string {
|
||||
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
}
|
||||
|
||||
/** Clamp a present `optional int32` result cap the way the reference does; `undefined` stays unset. */
|
||||
export function piLimit(limit: number | undefined): number | undefined {
|
||||
return limit === undefined ? undefined : Math.max(1, Math.floor(limit));
|
||||
}
|
||||
|
||||
/**
|
||||
* A `pi_bash` frame's timeout as the local `bash` tool's kwarg.
|
||||
*
|
||||
* Presence-sensitive like every other `optional int32` here, and unusually
|
||||
* load-bearing: `bash` documents `timeout: 0` as "disables the command
|
||||
* deadline", so folding a supplied `0` into `undefined` applies the 300s
|
||||
* default and kills exactly the long-running command that asked not to be.
|
||||
* Negative values have no local meaning and fall back to the default.
|
||||
*/
|
||||
export function piTimeout(timeout: number | undefined): number | undefined {
|
||||
return timeout !== undefined && timeout >= 0 ? timeout : undefined;
|
||||
}
|
||||
+1138
-66
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,495 @@
|
||||
/**
|
||||
* Proto builders for the modern Cursor CLI exec frames (`ExecServerMessage`
|
||||
* 27-31, 36-38, 40-55).
|
||||
*
|
||||
* Split out of `cursor.ts` because these are pure `create(...)` shapes with no
|
||||
* transport, stream, or block-state coupling: the dispatcher decides *which*
|
||||
* answer a frame gets, this module knows *what* that answer looks like on the
|
||||
* wire. Every builder returns a result whose oneof case is set — an
|
||||
* `ExecClientMessage` carrying a result with an unset oneof is a fake success
|
||||
* the server reads as "the tool ran and produced nothing".
|
||||
*/
|
||||
|
||||
import { create } from "@bufbuild/protobuf";
|
||||
import {
|
||||
AfterAgentResponseRequestResponseSchema,
|
||||
AfterAgentThoughtRequestResponseSchema,
|
||||
BeforeSubmitPromptRequestResponseSchema,
|
||||
type ExecuteHookRequest,
|
||||
type ExecuteHookResponse,
|
||||
ExecuteHookResponseSchema,
|
||||
type ExecuteHookResult,
|
||||
ExecuteHookResultSchema,
|
||||
type McpStateExecResult,
|
||||
McpStateExecResultSchema,
|
||||
McpStateServerSchema,
|
||||
McpStateSuccessSchema,
|
||||
type McpToolDefinition,
|
||||
PiBashExecErrorSchema,
|
||||
type PiBashExecResult,
|
||||
PiBashExecResultSchema,
|
||||
PiBashExecSuccessSchema,
|
||||
PiEditExecErrorSchema,
|
||||
PiEditExecRejectedSchema,
|
||||
type PiEditExecResult,
|
||||
PiEditExecResultSchema,
|
||||
PiEditExecSuccessSchema,
|
||||
PiFindExecErrorSchema,
|
||||
type PiFindExecResult,
|
||||
PiFindExecResultSchema,
|
||||
PiFindExecSuccessSchema,
|
||||
PiGrepExecErrorSchema,
|
||||
type PiGrepExecResult,
|
||||
PiGrepExecResultSchema,
|
||||
PiGrepExecSuccessSchema,
|
||||
PiLsExecErrorSchema,
|
||||
type PiLsExecResult,
|
||||
PiLsExecResultSchema,
|
||||
PiLsExecSuccessSchema,
|
||||
PiReadExecErrorSchema,
|
||||
type PiReadExecResult,
|
||||
PiReadExecResultSchema,
|
||||
PiReadExecSuccessSchema,
|
||||
type PiTruncation,
|
||||
PiTruncationSchema,
|
||||
PiWriteExecErrorSchema,
|
||||
PiWriteExecRejectedSchema,
|
||||
type PiWriteExecResult,
|
||||
PiWriteExecResultSchema,
|
||||
PiWriteExecSuccessSchema,
|
||||
PostToolUseFailureRequestResponseSchema,
|
||||
PostToolUseRequestResponseSchema,
|
||||
PreCompactRequestResponseSchema,
|
||||
PreToolUseRequestResponseSchema,
|
||||
StopRequestResponseSchema,
|
||||
SubagentStartRequestResponseSchema,
|
||||
SubagentStopRequestResponseSchema,
|
||||
} from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb";
|
||||
import type { ToolResultMessage } from "../../types";
|
||||
|
||||
/**
|
||||
* The pure arg translation lives in `../cursor-pi-args` so the legacy pi shim
|
||||
* can share it without pulling this module's protobuf graph into the bundled
|
||||
* virtual registry. Re-exported here because this is where the frame builders
|
||||
* and their translation are consumed together.
|
||||
*/
|
||||
export {
|
||||
piEscapeRegexLiteral,
|
||||
piGrepSkip,
|
||||
piJoinPath,
|
||||
piLimit,
|
||||
piLsPath,
|
||||
piReadDisplayPath,
|
||||
piReadPath,
|
||||
piTimeout,
|
||||
} from "../cursor-pi-args";
|
||||
|
||||
/** Flatten a tool result's content into the single `output` string the Pi frames carry. */
|
||||
export function piOutputText(toolResult: ToolResultMessage): string {
|
||||
return toolResult.content.map(item => (item.type === "text" ? item.text : `[${item.mimeType} image]`)).join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
* Read one field off a tool result's `details` bag, or off a nested object
|
||||
* inside it.
|
||||
*
|
||||
* `details` is `unknown` by design — every tool ships its own shape — so each
|
||||
* read is narrowed rather than asserted, and the caller decides what a value of
|
||||
* the wrong type means.
|
||||
*/
|
||||
function bagValue(bag: unknown, key: string): unknown {
|
||||
if (!bag || typeof bag !== "object" || !(key in bag)) return undefined;
|
||||
return Reflect.get(bag, key);
|
||||
}
|
||||
|
||||
/**
|
||||
* A positive integer count from `details`, or `undefined`.
|
||||
*
|
||||
* The Pi frames model their limit counters as `optional uint32`: zero means
|
||||
* "the limit was reached at zero results", so a missing, non-numeric, or
|
||||
* non-positive value must stay unset rather than be sent as 0.
|
||||
*/
|
||||
function detailCount(toolResult: ToolResultMessage, key: string): number | undefined {
|
||||
const value = bagValue(toolResult.details, key);
|
||||
return positiveCount(value);
|
||||
}
|
||||
|
||||
function positiveCount(value: unknown): number | undefined {
|
||||
return typeof value === "number" && Number.isFinite(value) && value > 0 ? Math.floor(value) : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* The entry cap a listing hit, from either shape a local tool records it in.
|
||||
*
|
||||
* `glob` sets a flat `details.resultLimitReached` alongside the structured
|
||||
* meta; `read` — which serves `pi_ls` — records the cap only through
|
||||
* `OutputMeta` at `details.meta.limits.resultLimit.reached`. Reading just the
|
||||
* flat field dropped `entry_limit_reached` for every real listing, so Cursor
|
||||
* received clipped output with no incompleteness signal.
|
||||
*/
|
||||
function resultLimitReached(toolResult: ToolResultMessage): number | undefined {
|
||||
const flat = detailCount(toolResult, "resultLimitReached");
|
||||
if (flat !== undefined) return flat;
|
||||
const limits = bagValue(bagValue(toolResult.details, "meta"), "limits");
|
||||
return positiveCount(bagValue(bagValue(limits, "resultLimit"), "reached"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Translate a local tool's truncation summary into `PiTruncation`.
|
||||
*
|
||||
* Two shapes reach here. `read`/`grep` set `details.truncation`
|
||||
* (`TruncationResult`), which carries an explicit `truncated` boolean. `bash`
|
||||
* sets `details.meta.truncation` (`TruncationMeta`), which has **no** such
|
||||
* flag — its presence *is* the signal, and requiring the boolean silently
|
||||
* dropped every Bash truncation, handing Cursor clipped output with no notice
|
||||
* that it was clipped.
|
||||
*
|
||||
* Returns `undefined` when nothing was truncated: the field is `optional` on
|
||||
* every Pi success message, and emitting a zeroed `PiTruncation` would tell the
|
||||
* server the output was trimmed to nothing.
|
||||
*/
|
||||
export function piTruncation(toolResult: ToolResultMessage): PiTruncation | undefined {
|
||||
const direct = bagValue(toolResult.details, "truncation");
|
||||
// `TruncationResult` is authoritative when present and explicitly false.
|
||||
const truncation = direct !== undefined ? direct : bagValue(bagValue(toolResult.details, "meta"), "truncation");
|
||||
if (truncation === undefined || truncation === null) return undefined;
|
||||
// `TruncationResult` gates on its flag; `TruncationMeta` has none and is
|
||||
// only ever attached when output was actually trimmed.
|
||||
const flag = bagValue(truncation, "truncated");
|
||||
if (flag !== undefined && flag !== true) return undefined;
|
||||
const truncatedBy = bagValue(truncation, "truncatedBy");
|
||||
const totalLines = bagValue(truncation, "totalLines");
|
||||
const outputLines = bagValue(truncation, "outputLines");
|
||||
const outputBytes = bagValue(truncation, "outputBytes");
|
||||
return create(PiTruncationSchema, {
|
||||
truncated: true,
|
||||
truncatedBy: typeof truncatedBy === "string" ? truncatedBy : "",
|
||||
totalLines: typeof totalLines === "number" ? totalLines : 0,
|
||||
outputLines: typeof outputLines === "number" ? outputLines : 0,
|
||||
outputBytes: typeof outputBytes === "number" ? outputBytes : 0,
|
||||
firstLineExceedsLimit: bagValue(truncation, "firstLineExceedsLimit") === true,
|
||||
lastLinePartial: bagValue(truncation, "lastLinePartial") === true,
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiReadResult(toolResult: ToolResultMessage): PiReadExecResult {
|
||||
const text = piOutputText(toolResult);
|
||||
if (toolResult.isError) return buildPiReadError(text || "Read failed");
|
||||
return create(PiReadExecResultSchema, {
|
||||
result: {
|
||||
case: "success",
|
||||
value: create(PiReadExecSuccessSchema, { output: text, truncation: piTruncation(toolResult) }),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiReadError(error: string): PiReadExecResult {
|
||||
return create(PiReadExecResultSchema, {
|
||||
result: { case: "error", value: create(PiReadExecErrorSchema, { error }) },
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiBashResult(toolResult: ToolResultMessage): PiBashExecResult {
|
||||
const text = piOutputText(toolResult);
|
||||
const truncation = piTruncation(toolResult);
|
||||
if (toolResult.isError) {
|
||||
return create(PiBashExecResultSchema, {
|
||||
result: {
|
||||
case: "error",
|
||||
value: create(PiBashExecErrorSchema, { error: text || "Command failed", truncation }),
|
||||
},
|
||||
});
|
||||
}
|
||||
return create(PiBashExecResultSchema, {
|
||||
result: {
|
||||
case: "success",
|
||||
value: create(PiBashExecSuccessSchema, { output: text, truncation }),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiBashError(error: string): PiBashExecResult {
|
||||
return create(PiBashExecResultSchema, {
|
||||
result: { case: "error", value: create(PiBashExecErrorSchema, { error }) },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* `PiEditExecSuccess` requires `diff` and `patch` alongside `output`. The local
|
||||
* `edit` tool reports them under `details`; when it does not, the strings stay
|
||||
* empty rather than being faked from the output text.
|
||||
*/
|
||||
export function buildPiEditResult(toolResult: ToolResultMessage): PiEditExecResult {
|
||||
const text = piOutputText(toolResult);
|
||||
if (toolResult.isError) return buildPiEditError(text || "Edit failed");
|
||||
const diff = bagValue(toolResult.details, "diff");
|
||||
const patch = bagValue(toolResult.details, "patch");
|
||||
return create(PiEditExecResultSchema, {
|
||||
result: {
|
||||
case: "success",
|
||||
value: create(PiEditExecSuccessSchema, {
|
||||
output: text,
|
||||
diff: typeof diff === "string" ? diff : "",
|
||||
patch: typeof patch === "string" ? patch : "",
|
||||
firstChangedLine: detailCount(toolResult, "firstChangedLine"),
|
||||
}),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiEditError(error: string): PiEditExecResult {
|
||||
return create(PiEditExecResultSchema, {
|
||||
result: { case: "error", value: create(PiEditExecErrorSchema, { error }) },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* A refusal is not an execution failure: `PiEditExecResult` models them as
|
||||
* separate variants, and answering a denied call with `error` reads as "the
|
||||
* edit ran and broke", which invites a retry of an operation that was never
|
||||
* permitted.
|
||||
*/
|
||||
export function buildPiEditRejected(reason: string): PiEditExecResult {
|
||||
return create(PiEditExecResultSchema, {
|
||||
result: { case: "rejected", value: create(PiEditExecRejectedSchema, { reason }) },
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiWriteResult(toolResult: ToolResultMessage): PiWriteExecResult {
|
||||
const text = piOutputText(toolResult);
|
||||
if (toolResult.isError) return buildPiWriteError(text || "Write failed");
|
||||
return create(PiWriteExecResultSchema, {
|
||||
result: { case: "success", value: create(PiWriteExecSuccessSchema, { output: text }) },
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiWriteError(error: string): PiWriteExecResult {
|
||||
return create(PiWriteExecResultSchema, {
|
||||
result: { case: "error", value: create(PiWriteExecErrorSchema, { error }) },
|
||||
});
|
||||
}
|
||||
|
||||
/** Same variant split as {@link buildPiEditRejected}. */
|
||||
export function buildPiWriteRejected(reason: string): PiWriteExecResult {
|
||||
return create(PiWriteExecResultSchema, {
|
||||
result: { case: "rejected", value: create(PiWriteExecRejectedSchema, { reason }) },
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiGrepResult(toolResult: ToolResultMessage): PiGrepExecResult {
|
||||
const text = piOutputText(toolResult);
|
||||
if (toolResult.isError) return buildPiGrepError(text || "Grep failed");
|
||||
const matchLimitReached = detailCount(toolResult, "perFileLimitReached");
|
||||
return create(PiGrepExecResultSchema, {
|
||||
result: {
|
||||
case: "success",
|
||||
value: create(PiGrepExecSuccessSchema, {
|
||||
output: text,
|
||||
truncation: piTruncation(toolResult) ?? grepInternalCapTruncation(toolResult, text, matchLimitReached),
|
||||
matchLimitReached,
|
||||
linesTruncated: bagValue(toolResult.details, "linesTruncated") === true,
|
||||
}),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* The one grep cap that reaches Cursor through no other field.
|
||||
*
|
||||
* `GrepTool` folds every cap into the flat `details.truncated`, but only
|
||||
* populates `details.truncation` when the output text itself was trimmed and
|
||||
* `perFileLimitReached` when a per-file or explicit total cap fired. Its native
|
||||
* backend's own ceiling (`INTERNAL_TOTAL_CAP`) sets neither, so a search that
|
||||
* hit it answered as an unqualified success over incomplete results — the one
|
||||
* failure mode a caller cannot detect and cannot retry around.
|
||||
*
|
||||
* Only consulted once the specific signals came back empty, so a cap that did
|
||||
* report itself is never restated. `totalLines` stays 0: the flat flag says
|
||||
* that results were dropped, not how many there were.
|
||||
*/
|
||||
function grepInternalCapTruncation(
|
||||
toolResult: ToolResultMessage,
|
||||
text: string,
|
||||
matchLimitReached: number | undefined,
|
||||
): PiTruncation | undefined {
|
||||
if (matchLimitReached !== undefined) return undefined;
|
||||
if (bagValue(toolResult.details, "truncated") !== true) return undefined;
|
||||
return create(PiTruncationSchema, {
|
||||
truncated: true,
|
||||
truncatedBy: "matches",
|
||||
totalLines: 0,
|
||||
outputLines: text ? text.split("\n").length : 0,
|
||||
outputBytes: Buffer.byteLength(text, "utf-8"),
|
||||
firstLineExceedsLimit: false,
|
||||
lastLinePartial: false,
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiGrepError(error: string): PiGrepExecResult {
|
||||
return create(PiGrepExecResultSchema, {
|
||||
result: { case: "error", value: create(PiGrepExecErrorSchema, { error }) },
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiFindResult(toolResult: ToolResultMessage): PiFindExecResult {
|
||||
const text = piOutputText(toolResult);
|
||||
if (toolResult.isError) return buildPiFindError(text || "Find failed");
|
||||
return create(PiFindExecResultSchema, {
|
||||
result: {
|
||||
case: "success",
|
||||
value: create(PiFindExecSuccessSchema, {
|
||||
output: text,
|
||||
truncation: piTruncation(toolResult),
|
||||
resultLimitReached: resultLimitReached(toolResult),
|
||||
}),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiFindError(error: string): PiFindExecResult {
|
||||
return create(PiFindExecResultSchema, {
|
||||
result: { case: "error", value: create(PiFindExecErrorSchema, { error }) },
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiLsResult(toolResult: ToolResultMessage): PiLsExecResult {
|
||||
const text = piOutputText(toolResult);
|
||||
if (toolResult.isError) return buildPiLsError(text || "Ls failed");
|
||||
return create(PiLsExecResultSchema, {
|
||||
result: {
|
||||
case: "success",
|
||||
value: create(PiLsExecSuccessSchema, {
|
||||
output: text,
|
||||
truncation: piTruncation(toolResult),
|
||||
entryLimitReached: resultLimitReached(toolResult),
|
||||
}),
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
export function buildPiLsError(error: string): PiLsExecResult {
|
||||
return create(PiLsExecResultSchema, {
|
||||
result: { case: "error", value: create(PiLsExecErrorSchema, { error }) },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Answer `mcpStateExecArgs` (frame 36) from the catalog already advertised in
|
||||
* `RequestContext.tools`.
|
||||
*
|
||||
* This client hosts no MCP servers of its own: every forwarded tool is a local
|
||||
* pi-agent tool published under a synthetic `providerIdentifier`. Regrouping
|
||||
* the same list keeps the server's view of "which servers exist and what do
|
||||
* they expose" consistent with what it was told at context time, instead of
|
||||
* claiming zero servers while tool calls for them keep arriving.
|
||||
*
|
||||
* `serverIdentifiers` filters the answer when the server asks about specific
|
||||
* servers. `kickOnly` is a restart request — there is nothing to restart, so it
|
||||
* is answered with the same state rather than an error.
|
||||
*/
|
||||
export function buildMcpStateResult(
|
||||
tools: McpToolDefinition[],
|
||||
serverIdentifiers: readonly string[],
|
||||
): McpStateExecResult {
|
||||
const byProvider = new Map<string, McpToolDefinition[]>();
|
||||
for (const tool of tools) {
|
||||
const identifier = tool.providerIdentifier;
|
||||
const existing = byProvider.get(identifier);
|
||||
if (existing) existing.push(tool);
|
||||
else byProvider.set(identifier, [tool]);
|
||||
}
|
||||
|
||||
const wanted = serverIdentifiers.length > 0 ? new Set(serverIdentifiers) : undefined;
|
||||
const servers = [];
|
||||
for (const [identifier, serverTools] of byProvider) {
|
||||
if (wanted && !wanted.has(identifier)) continue;
|
||||
servers.push(
|
||||
create(McpStateServerSchema, {
|
||||
serverName: identifier,
|
||||
serverIdentifier: identifier,
|
||||
tools: serverTools,
|
||||
status: "connected",
|
||||
}),
|
||||
);
|
||||
}
|
||||
|
||||
return create(McpStateExecResultSchema, {
|
||||
result: { case: "success", value: create(McpStateSuccessSchema, { servers }) },
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the neutral response for a hook query: the matching response case with
|
||||
* every field unset.
|
||||
*
|
||||
* This client runs no Cursor hooks, and every field of every response variant
|
||||
* is `optional` — so an empty response of the right case means "no hook had
|
||||
* anything to say", which is exactly true. It is NOT the unset-oneof fake
|
||||
* success: the case itself is set, only the payload is empty.
|
||||
*
|
||||
* `ExecuteHookRequest` and `ExecuteHookResponse` are parallel oneofs whose case
|
||||
* names line up, but the two unions are unrelated to the compiler: a `switch`
|
||||
* is what makes each pairing individually type-checked, and it forces a
|
||||
* deliberate branch when a future regen adds a request case.
|
||||
*
|
||||
* Returns `null` for a request case this build does not model, which the
|
||||
* dispatcher answers with `ExecClientThrow` rather than guessing a case.
|
||||
*/
|
||||
export function buildNeutralHookResult(request: ExecuteHookRequest | undefined): ExecuteHookResult | null {
|
||||
let response: ExecuteHookResponse;
|
||||
switch (request?.request.case) {
|
||||
case "preCompact":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "preCompact", value: create(PreCompactRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "subagentStart":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "subagentStart", value: create(SubagentStartRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "subagentStop":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "subagentStop", value: create(SubagentStopRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "preToolUse":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "preToolUse", value: create(PreToolUseRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "postToolUse":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "postToolUse", value: create(PostToolUseRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "postToolUseFailure":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "postToolUseFailure", value: create(PostToolUseFailureRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "beforeSubmitPrompt":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "beforeSubmitPrompt", value: create(BeforeSubmitPromptRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "afterAgentResponse":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "afterAgentResponse", value: create(AfterAgentResponseRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "afterAgentThought":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "afterAgentThought", value: create(AfterAgentThoughtRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
case "stop":
|
||||
response = create(ExecuteHookResponseSchema, {
|
||||
response: { case: "stop", value: create(StopRequestResponseSchema, {}) },
|
||||
});
|
||||
break;
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
return create(ExecuteHookResultSchema, { response });
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,19 @@
|
||||
import { createApiKeyLogin } from "./api-key-login";
|
||||
import type { OAuthLoginCallbacks } from "./oauth/types";
|
||||
import type { ProviderDefinition } from "./types";
|
||||
|
||||
export const loginExa = createApiKeyLogin({
|
||||
providerLabel: "Exa",
|
||||
authUrl: "https://dashboard.exa.ai/api-keys",
|
||||
instructions: "Create or copy your API key from the Exa dashboard.",
|
||||
promptMessage: "Paste your Exa API key",
|
||||
placeholder: "API key",
|
||||
validation: null,
|
||||
});
|
||||
|
||||
export const exaProvider = {
|
||||
id: "exa",
|
||||
name: "Exa",
|
||||
envKeys: "EXA_API_KEY",
|
||||
login: (cb: OAuthLoginCallbacks) => loginExa(cb),
|
||||
} as const satisfies ProviderDefinition;
|
||||
@@ -12,6 +12,7 @@ import { coreWeaveProvider } from "./coreweave";
|
||||
import { cursorProvider } from "./cursor";
|
||||
import { deepseekProvider } from "./deepseek";
|
||||
import { devinProvider } from "./devin";
|
||||
import { exaProvider } from "./exa";
|
||||
import { firepassProvider } from "./firepass";
|
||||
import { fireworksProvider } from "./fireworks";
|
||||
import { githubCopilotProvider } from "./github-copilot";
|
||||
@@ -93,6 +94,7 @@ const ALL = [
|
||||
googleAntigravityProvider,
|
||||
googleGeminiCliProvider,
|
||||
openaiCodexDeviceProvider,
|
||||
xaiProvider,
|
||||
xaiOauthProvider,
|
||||
gitlabDuoProvider,
|
||||
gitLabDuoWorkflowProvider,
|
||||
@@ -138,6 +140,7 @@ const ALL = [
|
||||
opencodeGoProvider,
|
||||
tavilyProvider,
|
||||
kagiProvider,
|
||||
exaProvider,
|
||||
parallelProvider,
|
||||
ollamaProvider,
|
||||
ollamaCloudProvider,
|
||||
@@ -147,7 +150,6 @@ const ALL = [
|
||||
openaiProvider,
|
||||
googleProvider,
|
||||
googleVertexProvider,
|
||||
xaiProvider,
|
||||
groqProvider,
|
||||
mistralProvider,
|
||||
minimaxProvider,
|
||||
|
||||
@@ -1,6 +1,22 @@
|
||||
import { createApiKeyLogin } from "./api-key-login";
|
||||
import type { OAuthLoginCallbacks } from "./oauth/types";
|
||||
import type { ProviderDefinition } from "./types";
|
||||
|
||||
export const loginXAI = createApiKeyLogin({
|
||||
providerLabel: "xAI",
|
||||
authUrl: "https://console.x.ai/team/default/api-keys",
|
||||
instructions: "Create or copy your API key from the xAI Console",
|
||||
promptMessage: "Paste your xAI API key",
|
||||
placeholder: "xai-...",
|
||||
validation: {
|
||||
kind: "models-endpoint",
|
||||
provider: "xAI",
|
||||
modelsUrl: "https://api.x.ai/v1/models",
|
||||
},
|
||||
});
|
||||
|
||||
export const xaiProvider = {
|
||||
id: "xai",
|
||||
name: "xAI",
|
||||
name: "xAI API",
|
||||
login: (cb: OAuthLoginCallbacks) => loginXAI(cb),
|
||||
} as const satisfies ProviderDefinition;
|
||||
|
||||
@@ -691,7 +691,6 @@ type KeyResolver = string | (() => string | undefined);
|
||||
const LEGACY_ENV_KEYS: Record<string, KeyResolver> = {
|
||||
// Non-provider / search-tool keys and API-name keys not modeled as registry provider defs.
|
||||
"azure-openai-responses": "AZURE_OPENAI_API_KEY",
|
||||
exa: "EXA_API_KEY",
|
||||
jina: "JINA_API_KEY",
|
||||
brave: "BRAVE_API_KEY",
|
||||
tinyfish: "TINYFISH_API_KEY",
|
||||
|
||||
@@ -11,6 +11,20 @@ import type {
|
||||
LsArgs,
|
||||
LsResult,
|
||||
McpResult,
|
||||
PiBashExecArgs,
|
||||
PiBashExecResult,
|
||||
PiEditExecArgs,
|
||||
PiEditExecResult,
|
||||
PiFindExecArgs,
|
||||
PiFindExecResult,
|
||||
PiGrepExecArgs,
|
||||
PiGrepExecResult,
|
||||
PiLsExecArgs,
|
||||
PiLsExecResult,
|
||||
PiReadExecArgs,
|
||||
PiReadExecResult,
|
||||
PiWriteExecArgs,
|
||||
PiWriteExecResult,
|
||||
ReadArgs,
|
||||
ReadResult,
|
||||
ShellArgs,
|
||||
@@ -938,6 +952,14 @@ export interface CursorMcpCall {
|
||||
toolCallId: string;
|
||||
args: Record<string, unknown>;
|
||||
rawArgs: Record<string, Uint8Array>;
|
||||
/**
|
||||
* The frame asks only whether this call would be permitted — it must not
|
||||
* run. The server sends it to resolve a smart-mode approval decision ahead
|
||||
* of the real invocation, and answers with the dedicated `approved`
|
||||
* variant, so executing here would fire a side-effecting tool the user has
|
||||
* not yet been asked about (and fire it twice once the real call arrives).
|
||||
*/
|
||||
approvalOnly?: boolean;
|
||||
}
|
||||
|
||||
export interface CursorTodoSnapshotItem {
|
||||
@@ -989,6 +1011,54 @@ export interface CursorShellStreamCallbacks {
|
||||
onStderr(data: string): void;
|
||||
}
|
||||
|
||||
/**
|
||||
* A modern Pi exec frame plus the call id the dispatcher minted for it.
|
||||
*
|
||||
* Unlike the legacy exec args (`ReadArgs`, `ShellArgs`, ...), the Pi frames
|
||||
* carry no `tool_call_id` field: on modern builds the id rides the streamed
|
||||
* `ToolCall` envelope (`ToolCall.tool_call_id = 57`) instead of each variant's
|
||||
* args. The exec channel has no access to that envelope, so the dispatcher
|
||||
* mints an id and hands it to the handler, keeping the synthesized transcript
|
||||
* block and its paired `toolResult` on the same key.
|
||||
*/
|
||||
export interface CursorPiCall<TArgs> {
|
||||
args: TArgs;
|
||||
toolCallId: string;
|
||||
}
|
||||
|
||||
/** One resource a host's MCP servers advertise. */
|
||||
export interface CursorMcpResource {
|
||||
uri: string;
|
||||
name?: string;
|
||||
description?: string;
|
||||
mimeType?: string;
|
||||
/** The server advertising it; Cursor addresses reads by this name. */
|
||||
server: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* The content of one resource read.
|
||||
*
|
||||
* `text` and `blob` are the wire's content oneof: exactly one is sent, with
|
||||
* `text` winning when a host supplies both. A download instead sets
|
||||
* `downloadPath` and no content at all — the model is told where the file
|
||||
* landed rather than being handed its bytes.
|
||||
*/
|
||||
export interface CursorMcpResourceContent {
|
||||
uri: string;
|
||||
name?: string;
|
||||
description?: string;
|
||||
mimeType?: string;
|
||||
text?: string;
|
||||
blob?: Uint8Array;
|
||||
/**
|
||||
* Where the host wrote the resource, workspace-relative, when the frame
|
||||
* asked for a download. Set this INSTEAD of `text`/`blob`: the wire
|
||||
* contract is that a download returns no content to the model.
|
||||
*/
|
||||
downloadPath?: string;
|
||||
}
|
||||
|
||||
export interface CursorExecHandlers {
|
||||
read?: (args: ReadArgs) => Promise<CursorExecHandlerResult<ReadResult>>;
|
||||
ls?: (args: LsArgs) => Promise<CursorExecHandlerResult<LsResult>>;
|
||||
@@ -1002,6 +1072,51 @@ export interface CursorExecHandlers {
|
||||
) => Promise<CursorExecHandlerResult<ShellResult>>;
|
||||
diagnostics?: (args: DiagnosticsArgs) => Promise<CursorExecHandlerResult<DiagnosticsResult>>;
|
||||
mcp?: (call: CursorMcpCall) => Promise<CursorExecHandlerResult<McpResult>>;
|
||||
/**
|
||||
* Answers "would this MCP call be permitted", without running it.
|
||||
*
|
||||
* A modern `mcpArgs` frame carrying `smart_mode_approval_only` asks for the
|
||||
* permission decision alone, ahead of the real invocation. Executing the
|
||||
* tool to answer it would fire a side effect the user never approved — and
|
||||
* fire it twice once the real call arrives.
|
||||
*
|
||||
* `true` only when the host's policy resolves to a definite allow. A pending
|
||||
* prompt is `false`: it can only be answered interactively at execution
|
||||
* time, and there is no "ask me later" reply in this frame's result. When no
|
||||
* handler is registered the provider refuses, since it cannot decide.
|
||||
*/
|
||||
mcpApprovalPreflight?: (call: CursorMcpCall) => Promise<boolean>;
|
||||
/**
|
||||
* Modern Cursor CLI Pi tool frames (`ExecServerMessage` 45-51). They are a
|
||||
* distinct frame family from the legacy `readArgs`/`shellArgs`/... set, not
|
||||
* an alias: different args, different result oneofs, and no `tool_call_id`.
|
||||
*/
|
||||
piRead?: (call: CursorPiCall<PiReadExecArgs>) => Promise<CursorExecHandlerResult<PiReadExecResult>>;
|
||||
piBash?: (call: CursorPiCall<PiBashExecArgs>) => Promise<CursorExecHandlerResult<PiBashExecResult>>;
|
||||
piEdit?: (call: CursorPiCall<PiEditExecArgs>) => Promise<CursorExecHandlerResult<PiEditExecResult>>;
|
||||
piWrite?: (call: CursorPiCall<PiWriteExecArgs>) => Promise<CursorExecHandlerResult<PiWriteExecResult>>;
|
||||
piGrep?: (call: CursorPiCall<PiGrepExecArgs>) => Promise<CursorExecHandlerResult<PiGrepExecResult>>;
|
||||
piFind?: (call: CursorPiCall<PiFindExecArgs>) => Promise<CursorExecHandlerResult<PiFindExecResult>>;
|
||||
piLs?: (call: CursorPiCall<PiLsExecArgs>) => Promise<CursorExecHandlerResult<PiLsExecResult>>;
|
||||
/**
|
||||
* The resources the host's MCP servers advertise, optionally filtered to one
|
||||
* server. Without a handler the provider answers an empty catalog, which
|
||||
* hides resources a host is in fact holding live connections to.
|
||||
*/
|
||||
listMcpResources?: (args: { server?: string }) => Promise<CursorMcpResource[]>;
|
||||
/**
|
||||
* Read one resource. `null` means the server or uri is genuinely unknown,
|
||||
* which the provider answers as `not_found`; throwing surfaces as `error`.
|
||||
*/
|
||||
readMcpResource?: (args: {
|
||||
server: string;
|
||||
uri: string;
|
||||
/**
|
||||
* When set, write the resource here (workspace-relative) and return
|
||||
* `downloadPath` instead of content.
|
||||
*/
|
||||
downloadPath?: string;
|
||||
}) => Promise<CursorMcpResourceContent | null>;
|
||||
/** Mirror Cursor's server-owned todo list into local session state. */
|
||||
todoSync?: CursorTodoSyncHandler;
|
||||
onToolResult?: CursorToolResultHandler;
|
||||
|
||||
@@ -0,0 +1,192 @@
|
||||
import { ProviderHttpError } from "../error";
|
||||
import type {
|
||||
UsageAmount,
|
||||
UsageFetchContext,
|
||||
UsageFetchParams,
|
||||
UsageLimit,
|
||||
UsageProvider,
|
||||
UsageReport,
|
||||
UsageStatus,
|
||||
UsageWindow,
|
||||
} from "../usage";
|
||||
import { isRecord } from "../utils";
|
||||
|
||||
const UMANS_PROVIDER = "umans";
|
||||
const DEFAULT_ENDPOINT = "https://api.code.umans.ai";
|
||||
const USAGE_PATH = "/v1/usage";
|
||||
const FIVE_HOURS_MS = 5 * 60 * 60 * 1000;
|
||||
|
||||
/** Umans `GET /v1/usage` response (subset; extras ignored). */
|
||||
interface UmansUsagePayload {
|
||||
plan?: { display_name?: string };
|
||||
limits?: {
|
||||
requests?: { limit?: number; hard_cap?: number | null; window_seconds?: number };
|
||||
concurrency?: { limit?: number; hard_cap?: number | null };
|
||||
};
|
||||
usage?: {
|
||||
requests_in_window?: number;
|
||||
remaining_requests?: number;
|
||||
concurrent_sessions?: number;
|
||||
tokens_in?: number;
|
||||
tokens_out?: number;
|
||||
priority?: { low?: boolean };
|
||||
};
|
||||
}
|
||||
|
||||
function normalizeBaseUrl(baseUrl?: string): string {
|
||||
if (!baseUrl?.trim()) return DEFAULT_ENDPOINT;
|
||||
const trimmed = baseUrl.trim();
|
||||
// Strip a trailing `/v1` (with optional surrounding slashes) so the usage
|
||||
// path doesn't double it, but preserve any preceding path prefix (e.g. a
|
||||
// path-mounted gateway like `https://gateway.example/team/umans/v1`).
|
||||
const withoutTrailingSlash = trimmed.replace(/\/+$/, "");
|
||||
return withoutTrailingSlash.replace(/\/v1$/i, "") || DEFAULT_ENDPOINT;
|
||||
}
|
||||
|
||||
function toFiniteNumber(value: unknown): number | undefined {
|
||||
if (typeof value !== "number" || !Number.isFinite(value)) return undefined;
|
||||
return value;
|
||||
}
|
||||
|
||||
function resolveStatus(usedFraction: number | undefined): UsageStatus | undefined {
|
||||
if (usedFraction === undefined) return undefined;
|
||||
if (usedFraction >= 1) return "exhausted";
|
||||
if (usedFraction >= 0.9) return "warning";
|
||||
return "ok";
|
||||
}
|
||||
|
||||
function buildAmount(args: {
|
||||
used: number | undefined;
|
||||
limit: number | undefined;
|
||||
remaining: number | undefined;
|
||||
unit: UsageAmount["unit"];
|
||||
}): UsageAmount {
|
||||
const used = args.used;
|
||||
const limit = args.limit;
|
||||
const usedFraction = used !== undefined && limit !== undefined && limit > 0 ? Math.min(used / limit, 1) : undefined;
|
||||
const remainingFraction = usedFraction !== undefined ? Math.max(1 - usedFraction, 0) : undefined;
|
||||
return {
|
||||
used,
|
||||
limit,
|
||||
remaining: args.remaining,
|
||||
usedFraction,
|
||||
remainingFraction,
|
||||
unit: args.unit,
|
||||
};
|
||||
}
|
||||
|
||||
function buildRequestsLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null {
|
||||
const limit = toFiniteNumber(payload.limits?.requests?.limit);
|
||||
const windowSeconds = toFiniteNumber(payload.limits?.requests?.window_seconds);
|
||||
const used = toFiniteNumber(payload.usage?.requests_in_window);
|
||||
const remaining = toFiniteNumber(payload.usage?.remaining_requests);
|
||||
if (limit === undefined && used === undefined) return null;
|
||||
const amount = buildAmount({ used, limit, remaining, unit: "requests" });
|
||||
// Rolling window: each request ages out `window_seconds` after it fired, so
|
||||
// there is no single reset timestamp. Surface the window size + label only.
|
||||
// `window.id` is `"5h"` to match the status-line usage segment's window-id
|
||||
// contract (it only recognizes `"5h"`/`"7d"`); `label` stays human-readable.
|
||||
const window: UsageWindow = {
|
||||
id: "5h",
|
||||
label: "rolling 5h",
|
||||
durationMs: windowSeconds ? windowSeconds * 1000 : FIVE_HOURS_MS,
|
||||
};
|
||||
return {
|
||||
id: "umans:requests",
|
||||
label: "Requests (rolling 5h)",
|
||||
scope: { provider, windowId: window.id, shared: true },
|
||||
window,
|
||||
amount,
|
||||
status: resolveStatus(amount.usedFraction),
|
||||
};
|
||||
}
|
||||
|
||||
function buildConcurrencyLimit(payload: UmansUsagePayload, provider: string): UsageLimit | null {
|
||||
const limit = toFiniteNumber(payload.limits?.concurrency?.limit);
|
||||
const used = toFiniteNumber(payload.usage?.concurrent_sessions);
|
||||
if (limit === undefined && used === undefined) return null;
|
||||
const amount = buildAmount({ used, limit, remaining: undefined, unit: "requests" });
|
||||
return {
|
||||
id: "umans:concurrency",
|
||||
label: "Concurrency",
|
||||
// Concurrency is instantaneous, not windowed.
|
||||
scope: { provider, windowId: "concurrency" },
|
||||
amount,
|
||||
status: resolveStatus(amount.usedFraction),
|
||||
};
|
||||
}
|
||||
|
||||
async function fetchUmansUsage(params: UsageFetchParams, ctx: UsageFetchContext): Promise<UsageReport | null> {
|
||||
if (params.provider !== UMANS_PROVIDER) return null;
|
||||
const credential = params.credential;
|
||||
if (credential.type !== "api_key" || !credential.apiKey) return null;
|
||||
|
||||
const baseUrl = normalizeBaseUrl(params.baseUrl);
|
||||
const url = `${baseUrl}${USAGE_PATH}`;
|
||||
const headers: Record<string, string> = {
|
||||
authorization: `Bearer ${credential.apiKey}`,
|
||||
accept: "application/json",
|
||||
};
|
||||
|
||||
let payload: UmansUsagePayload | null = null;
|
||||
try {
|
||||
const response = await ctx.fetch(url, { headers, signal: params.signal });
|
||||
if (!response.ok) {
|
||||
// Auth failures (401/403) must throw so checkCredentials flags the bad
|
||||
// key as ok:false rather than ok:null (unknown). Other non-ok statuses
|
||||
// are transient — return null so the probe reports "no data".
|
||||
if (response.status === 401 || response.status === 403) {
|
||||
throw new ProviderHttpError(
|
||||
`Umans usage endpoint returned ${response.status} ${response.statusText}`.trim(),
|
||||
response.status,
|
||||
);
|
||||
}
|
||||
ctx.logger?.warn("Umans usage fetch failed", { status: response.status, statusText: response.statusText });
|
||||
return null;
|
||||
}
|
||||
const json = (await response.json()) as unknown;
|
||||
if (!isRecord(json)) {
|
||||
ctx.logger?.warn("Umans usage response was not a JSON object");
|
||||
return null;
|
||||
}
|
||||
payload = json as unknown as UmansUsagePayload;
|
||||
} catch (error) {
|
||||
// Re-throw auth errors so the credential-health probe can surface them.
|
||||
if (error instanceof ProviderHttpError) throw error;
|
||||
ctx.logger?.warn("Umans usage fetch error", { error: String(error) });
|
||||
return null;
|
||||
}
|
||||
|
||||
const limits: UsageLimit[] = [];
|
||||
const requests = buildRequestsLimit(payload, params.provider);
|
||||
if (requests) limits.push(requests);
|
||||
const concurrency = buildConcurrencyLimit(payload, params.provider);
|
||||
if (concurrency) limits.push(concurrency);
|
||||
if (limits.length === 0) return null;
|
||||
|
||||
const notes: string[] = [];
|
||||
if (payload.usage?.priority?.low === true) {
|
||||
notes.push("Requests deprioritized after a rate-limit burst.");
|
||||
}
|
||||
|
||||
return {
|
||||
provider: params.provider,
|
||||
fetchedAt: Date.now(),
|
||||
limits,
|
||||
notes: notes.length > 0 ? notes : undefined,
|
||||
metadata: {
|
||||
plan: payload.plan?.display_name,
|
||||
accountId: credential.accountId,
|
||||
email: credential.email,
|
||||
endpoint: url,
|
||||
},
|
||||
raw: payload as Record<string, unknown>,
|
||||
};
|
||||
}
|
||||
|
||||
export const umansUsageProvider: UsageProvider = {
|
||||
id: UMANS_PROVIDER,
|
||||
fetchUsage: fetchUmansUsage,
|
||||
supports: params => params.provider === UMANS_PROVIDER && params.credential.type === "api_key",
|
||||
validatesCredentials: true,
|
||||
};
|
||||
@@ -25,6 +25,18 @@ export const kStreamingBlockIndex = Symbol("provider.block.index");
|
||||
/** Stores the last parsed argument prefix length for throttled streaming JSON parsing. */
|
||||
export const kStreamingLastParseLen = Symbol("provider.block.lastParseLen");
|
||||
|
||||
/**
|
||||
* The Cursor interaction envelope's `call_id` for a streamed tool-call block.
|
||||
*
|
||||
* Tracked separately from the block's own `id` because they are NOT the same
|
||||
* key: MCP and Pi blocks are filed under the id inside the call's `args`, which
|
||||
* is what the exec channel pairs its result under, while every streamed
|
||||
* `ToolCall*Update` correlates on the envelope's `call_id`. Matching
|
||||
* completions against the block id would mis-route every call whose args carry
|
||||
* their own id.
|
||||
*/
|
||||
export const kStreamingEnvelopeId = Symbol("provider.block.envelopeId");
|
||||
|
||||
/** Marks streamed tool-call arguments that already received an authoritative done payload. */
|
||||
export const kStreamingArgumentsDone = Symbol("provider.block.argumentsDone");
|
||||
|
||||
|
||||
@@ -55,6 +55,7 @@ export interface NormalizeSchemaOptions {
|
||||
inferTypeForBareEnum: boolean;
|
||||
foldOneOfIntoAnyOf: boolean;
|
||||
dropNonScalarEnum: boolean;
|
||||
stringEnumsOnly?: boolean;
|
||||
rejectResidualIncompatibilities?: ReadonlyArray<ResidualSchemaIncompatibility>;
|
||||
validateAndFallback?: { fallback: unknown };
|
||||
}
|
||||
@@ -106,6 +107,15 @@ const SUBSCHEMA_VALUE_KEYS: Record<string, true> = {
|
||||
contentSchema: true,
|
||||
};
|
||||
|
||||
/**
|
||||
* Keywords whose value is either a boolean keyword value or an object
|
||||
* subschema. Object values must be walked, while bare booleans stay literal.
|
||||
*/
|
||||
const BOOLEAN_OR_SCHEMA_VALUE_KEYS: Record<string, true> = {
|
||||
additionalProperties: true,
|
||||
unevaluatedProperties: true,
|
||||
};
|
||||
|
||||
/** Keywords whose value is an array of subschemas. */
|
||||
const SUBSCHEMA_ARRAY_KEYS: Record<string, true> = {
|
||||
anyOf: true,
|
||||
@@ -124,6 +134,61 @@ const SUBSCHEMA_MAP_KEYS: Record<string, true> = {
|
||||
definitions: true,
|
||||
};
|
||||
|
||||
type SchemaChildKind = "schema" | "map";
|
||||
|
||||
/** Classify only JSON Schema-valued children; instance payloads remain opaque. */
|
||||
function classifySchemaChild(key: string, value: unknown, insideSchemaMap: boolean): SchemaChildKind | undefined {
|
||||
if (insideSchemaMap) return "schema";
|
||||
const normalizedKey = SNAKE_TO_CAMEL_RENAMES.get(key) ?? key;
|
||||
if (Object.hasOwn(SUBSCHEMA_MAP_KEYS, normalizedKey)) return "map";
|
||||
if (Object.hasOwn(SUBSCHEMA_VALUE_KEYS, normalizedKey) || Object.hasOwn(SUBSCHEMA_ARRAY_KEYS, normalizedKey)) {
|
||||
return "schema";
|
||||
}
|
||||
if (Object.hasOwn(BOOLEAN_OR_SCHEMA_VALUE_KEYS, normalizedKey) && isJsonObject(value)) return "schema";
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function hasUnrepresentableGoogleEnumConstraint(
|
||||
value: unknown,
|
||||
insideSchemaMap = false,
|
||||
seen = new Set<object>(),
|
||||
): boolean {
|
||||
if (Array.isArray(value)) {
|
||||
if (seen.has(value)) return false;
|
||||
seen.add(value);
|
||||
return value.some(entry => hasUnrepresentableGoogleEnumConstraint(entry, false, seen));
|
||||
}
|
||||
if (!isJsonObject(value)) return false;
|
||||
if (seen.has(value)) return false;
|
||||
seen.add(value);
|
||||
|
||||
if (insideSchemaMap) {
|
||||
for (const key in value) {
|
||||
if (Object.hasOwn(value, key) && hasUnrepresentableGoogleEnumConstraint(value[key], false, seen)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
if (
|
||||
Array.isArray(value.enum) &&
|
||||
(value.enum.length === 0 || value.enum.some(enumValue => typeof enumValue !== "string"))
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
if (Object.hasOwn(value, "const") && typeof value.const !== "string") return true;
|
||||
|
||||
for (const key in value) {
|
||||
if (!Object.hasOwn(value, key)) continue;
|
||||
const childKind = classifySchemaChild(key, value[key], false);
|
||||
if (childKind && hasUnrepresentableGoogleEnumConstraint(value[key], childKind === "map", seen)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
const CLOUD_CODE_ASSIST_CLAUDE_FALLBACK_SCHEMA = {
|
||||
type: "object",
|
||||
properties: {},
|
||||
@@ -361,14 +426,22 @@ function normalizeSchemaObjectNode(value: JsonObject, options: NormalizeSchemaWa
|
||||
continue;
|
||||
}
|
||||
if (options.stripNullableKeyword && key === "nullable") continue;
|
||||
result[key] = normalizeSchemaNode(entry, {
|
||||
...options,
|
||||
insideSchemaMap: !options.insideSchemaMap && Object.hasOwn(SUBSCHEMA_MAP_KEYS, key),
|
||||
booleanIsSubschema:
|
||||
options.insideSchemaMap ||
|
||||
Object.hasOwn(SUBSCHEMA_VALUE_KEYS, key) ||
|
||||
Object.hasOwn(SUBSCHEMA_ARRAY_KEYS, key),
|
||||
});
|
||||
if (
|
||||
options.stringEnumsOnly &&
|
||||
!options.insideSchemaMap &&
|
||||
key === "not" &&
|
||||
hasUnrepresentableGoogleEnumConstraint(entry)
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
const childKind = classifySchemaChild(key, entry, options.insideSchemaMap);
|
||||
result[key] = childKind
|
||||
? normalizeSchemaNode(entry, {
|
||||
...options,
|
||||
insideSchemaMap: childKind === "map",
|
||||
booleanIsSubschema: childKind === "schema",
|
||||
})
|
||||
: entry;
|
||||
}
|
||||
applyDescriptionSpill(result, spill, options);
|
||||
return applyNodePostProcessing(result, options);
|
||||
@@ -387,14 +460,22 @@ function normalizeSchemaObjectNode(value: JsonObject, options: NormalizeSchemaWa
|
||||
constValue = entry;
|
||||
continue;
|
||||
}
|
||||
result[key] = normalizeSchemaNode(entry, {
|
||||
...options,
|
||||
insideSchemaMap: !options.insideSchemaMap && Object.hasOwn(SUBSCHEMA_MAP_KEYS, key),
|
||||
booleanIsSubschema:
|
||||
options.insideSchemaMap ||
|
||||
Object.hasOwn(SUBSCHEMA_VALUE_KEYS, key) ||
|
||||
Object.hasOwn(SUBSCHEMA_ARRAY_KEYS, key),
|
||||
});
|
||||
if (
|
||||
options.stringEnumsOnly &&
|
||||
!options.insideSchemaMap &&
|
||||
key === "not" &&
|
||||
hasUnrepresentableGoogleEnumConstraint(entry)
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
const childKind = classifySchemaChild(key, entry, options.insideSchemaMap);
|
||||
result[key] = childKind
|
||||
? normalizeSchemaNode(entry, {
|
||||
...options,
|
||||
insideSchemaMap: childKind === "map",
|
||||
booleanIsSubschema: childKind === "schema",
|
||||
})
|
||||
: entry;
|
||||
}
|
||||
|
||||
if (options.normalizeTypeArrayToNullable && Array.isArray(result.type)) {
|
||||
@@ -464,6 +545,7 @@ function applyNodePostProcessing(schema: JsonObject, options: NormalizeSchemaWal
|
||||
}
|
||||
if (options.foldOneOfIntoAnyOf) current = foldOneOfIntoAnyOf(current);
|
||||
if (options.dropNonScalarEnum) current = dropNonScalarEnumForMfjs(current);
|
||||
if (options.stringEnumsOnly && options.booleanIsSubschema) current = dropNonStringEnumForGoogle(current);
|
||||
return current;
|
||||
}
|
||||
|
||||
@@ -484,6 +566,13 @@ function dropNonScalarEnumForMfjs(schema: JsonObject): JsonObject {
|
||||
return copySchemaWithout(schema, "enum");
|
||||
}
|
||||
|
||||
/** Google's Schema enum field accepts string values only; omit unsupported enums without dropping the node's type. */
|
||||
function dropNonStringEnumForGoogle(schema: JsonObject): JsonObject {
|
||||
if (!Array.isArray(schema.enum)) return schema;
|
||||
const isStringEnum = schema.enum.length > 0 && schema.enum.every(value => typeof value === "string");
|
||||
return isStringEnum ? schema : copySchemaWithout(schema, "enum");
|
||||
}
|
||||
|
||||
/** Copy all keys from a schema except the specified combiner key. */
|
||||
export function copySchemaWithout(schema: JsonObject, combiner: string): JsonObject {
|
||||
const { [combiner]: _, ...rest } = schema;
|
||||
@@ -732,16 +821,25 @@ function collapseSameTypeCombinerVariants(schema: JsonObject, combiner: "anyOf"
|
||||
* create new anyOf in merged subtrees after child normalization already ran.
|
||||
*/
|
||||
export function stripResidualCombiners(value: unknown, epoch: number = epochNext()): unknown {
|
||||
return stripResidualCombinersNode(value, epoch, false);
|
||||
}
|
||||
|
||||
function stripResidualCombinersNode(value: unknown, epoch: number, insideSchemaMap: boolean): unknown {
|
||||
if (Array.isArray(value)) {
|
||||
if (!once(value, epoch)) return [];
|
||||
return value.map(entry => stripResidualCombiners(entry, epoch));
|
||||
return value.map(entry => stripResidualCombinersNode(entry, epoch, false));
|
||||
}
|
||||
if (!isJsonObject(value)) return value;
|
||||
if (!once(value, epoch)) return {};
|
||||
const result: JsonObject = {};
|
||||
for (const key in value) {
|
||||
if (Object.hasOwn(value, key)) result[key] = stripResidualCombiners(value[key], epoch);
|
||||
if (!Object.hasOwn(value, key)) continue;
|
||||
const entry = value[key];
|
||||
const childKind = classifySchemaChild(key, entry, insideSchemaMap);
|
||||
result[key] = childKind ? stripResidualCombinersNode(entry, epoch, childKind === "map") : entry;
|
||||
}
|
||||
if (insideSchemaMap) return result;
|
||||
|
||||
let current: JsonObject = result;
|
||||
let changed = true;
|
||||
while (changed) {
|
||||
@@ -840,6 +938,7 @@ function normalizeNullablePropertiesForCloudCodeAssist(
|
||||
value: unknown,
|
||||
isPropertySchema = false,
|
||||
epoch: number = epochNext(),
|
||||
insideSchemaMap = false,
|
||||
): NullableNormalizationResult {
|
||||
if (Array.isArray(value)) {
|
||||
if (!once(value, epoch)) {
|
||||
@@ -859,9 +958,14 @@ function normalizeNullablePropertiesForCloudCodeAssist(
|
||||
|
||||
const normalized: JsonObject = {};
|
||||
for (const key in value) {
|
||||
if (Object.hasOwn(value, key))
|
||||
normalized[key] = normalizeNullablePropertiesForCloudCodeAssist(value[key], false, epoch).schema;
|
||||
if (!Object.hasOwn(value, key)) continue;
|
||||
const entry = value[key];
|
||||
const childKind = classifySchemaChild(key, entry, insideSchemaMap);
|
||||
normalized[key] = childKind
|
||||
? normalizeNullablePropertiesForCloudCodeAssist(entry, false, epoch, childKind === "map").schema
|
||||
: entry;
|
||||
}
|
||||
if (insideSchemaMap) return { schema: normalized, nullable: false };
|
||||
|
||||
if (isJsonObject(normalized.properties)) {
|
||||
const properties = normalized.properties;
|
||||
@@ -933,7 +1037,7 @@ function hasResidualSchemaIncompatibilities(
|
||||
): boolean {
|
||||
if (Array.isArray(value)) {
|
||||
if (!once(value, epoch)) return false;
|
||||
return value.some(entry => hasResidualSchemaIncompatibilities(entry, checks, epoch, insideSchemaMap));
|
||||
return value.some(entry => hasResidualSchemaIncompatibilities(entry, checks, epoch, false));
|
||||
}
|
||||
if (!isJsonObject(value)) {
|
||||
return false;
|
||||
@@ -942,25 +1046,22 @@ function hasResidualSchemaIncompatibilities(
|
||||
return false;
|
||||
}
|
||||
|
||||
if (checks.typeArray && Array.isArray(value.type)) return true;
|
||||
if (checks.typeNull && value.type === "null") return true;
|
||||
if (checks.nullable && Object.hasOwn(value, "nullable")) return true;
|
||||
if (!insideSchemaMap && checks.not && Object.hasOwn(value, "not")) return true;
|
||||
if (!insideSchemaMap && checks.combiners) {
|
||||
for (const combiner of CCA_FORBIDDEN_COMBINERS) {
|
||||
if (Array.isArray(value[combiner])) return true;
|
||||
if (!insideSchemaMap) {
|
||||
if (checks.typeArray && Array.isArray(value.type)) return true;
|
||||
if (checks.typeNull && value.type === "null") return true;
|
||||
if (checks.nullable && Object.hasOwn(value, "nullable")) return true;
|
||||
if (checks.not && Object.hasOwn(value, "not")) return true;
|
||||
if (checks.combiners) {
|
||||
for (const combiner of CCA_FORBIDDEN_COMBINERS) {
|
||||
if (Array.isArray(value[combiner])) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const k in value) {
|
||||
if (!Object.hasOwn(value, k)) continue;
|
||||
if (
|
||||
hasResidualSchemaIncompatibilities(
|
||||
value[k],
|
||||
checks,
|
||||
epoch,
|
||||
!insideSchemaMap && Object.hasOwn(SUBSCHEMA_MAP_KEYS, k),
|
||||
)
|
||||
) {
|
||||
for (const key in value) {
|
||||
if (!Object.hasOwn(value, key)) continue;
|
||||
const entry = value[key];
|
||||
const childKind = classifySchemaChild(key, entry, insideSchemaMap);
|
||||
if (childKind && hasResidualSchemaIncompatibilities(entry, checks, epoch, childKind === "map")) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -1012,6 +1113,7 @@ export function normalizeSchemaForGoogle(value: unknown): unknown {
|
||||
extractNullableFromUnions: false,
|
||||
inferTypeForBareEnum: true,
|
||||
dropNonScalarEnum: false,
|
||||
stringEnumsOnly: true,
|
||||
foldOneOfIntoAnyOf: false,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -28,6 +28,10 @@ const anthropicErrorBody = JSON.stringify({
|
||||
type: "error",
|
||||
error: { type: "invalid_request_error", message: "The compiled grammar is too large." },
|
||||
});
|
||||
const anthropicOverloadedErrorBody = JSON.stringify({
|
||||
type: "error",
|
||||
error: { type: "overloaded_error", message: "Overloaded" },
|
||||
});
|
||||
|
||||
describe("AnthropicMessagesClient error mapping", () => {
|
||||
it("maps non-2xx responses to AnthropicApiError with status and body in message", async () => {
|
||||
@@ -134,6 +138,19 @@ describe("AnthropicMessagesClient retries", () => {
|
||||
expect(error).toBeInstanceOf(AIError.AnthropicApiError);
|
||||
expect(calls.length).toBe(3); // initial attempt + 2 retries
|
||||
});
|
||||
|
||||
it("disables the cap when maxRetryDelayMs is negative", async () => {
|
||||
const { calls, fetch } = createFetchMock([
|
||||
new Response("overloaded", { status: 429, headers: { "retry-after-ms": "1" } }),
|
||||
new Response("{}", { status: 200 }),
|
||||
]);
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const response = await client.messages.create(params, { maxRetryDelayMs: -1 }).asResponse();
|
||||
|
||||
expect(response.status).toBe(200);
|
||||
expect(calls.length).toBe(2);
|
||||
});
|
||||
});
|
||||
|
||||
describe("AnthropicMessagesClient timeout and abort", () => {
|
||||
@@ -212,3 +229,449 @@ describe("AnthropicMessagesClient request assembly", () => {
|
||||
expect(calls[0].url).toBe("https://api.anthropic.com/v1/messages");
|
||||
});
|
||||
});
|
||||
|
||||
describe("AnthropicMessagesClient retry-after cap", () => {
|
||||
it("uses the documented 60-second default cap when callers omit one", async () => {
|
||||
const { calls, fetch } = createFetchMock([
|
||||
new Response(anthropicOverloadedErrorBody, { status: 429, headers: { "retry-after": "120" } }),
|
||||
]);
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
|
||||
expect(error).toBeInstanceOf(AIError.AnthropicApiError);
|
||||
expect(error.status).toBe(429);
|
||||
expect(calls.length).toBe(1);
|
||||
});
|
||||
|
||||
it("declines a retry and preserves original status/body/headers when retry-after exceeds maxRetryDelayMs", async () => {
|
||||
const errorHeaders = { "retry-after": "120", "request-id": "req_cap" };
|
||||
const { calls, fetch } = createFetchMock([
|
||||
new Response(anthropicOverloadedErrorBody, { status: 429, headers: errorHeaders }),
|
||||
]);
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const error = await client.messages
|
||||
.create(params, { maxRetryDelayMs: 60_000 })
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error).toBeInstanceOf(AIError.AnthropicApiError);
|
||||
expect(error.status).toBe(429);
|
||||
expect(error.message).toContain("overloaded");
|
||||
expect(error.headers.get("request-id")).toBe("req_cap");
|
||||
expect(calls.length).toBe(1);
|
||||
});
|
||||
|
||||
it("cancels and releases an open error-body reader when the caller aborts", async () => {
|
||||
const controller = new AbortController();
|
||||
const encoder = new TextEncoder();
|
||||
let readBlocked = false;
|
||||
let bodyCancelled = false;
|
||||
let response: Response | undefined;
|
||||
const openBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(encoder.encode("overloaded"));
|
||||
},
|
||||
pull() {
|
||||
readBlocked = true;
|
||||
return Promise.withResolvers<void>().promise;
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const fetch: FetchImpl = async () => {
|
||||
response = new Response(openBody, {
|
||||
status: 429,
|
||||
headers: { "retry-after": "120", "request-id": "req_abort" },
|
||||
});
|
||||
return response;
|
||||
};
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
const pending = client.messages
|
||||
.create(params, { signal: controller.signal, maxRetryDelayMs: 60_000 })
|
||||
.asResponse();
|
||||
for (let i = 0; i < 1000 && !readBlocked; i++) await Promise.resolve();
|
||||
if (!readBlocked) throw new Error("Anthropic error-body read did not block");
|
||||
|
||||
controller.abort();
|
||||
const error = await pending.catch(err => err as Error);
|
||||
if (!(error instanceof Error)) throw new Error("Expected request abort error");
|
||||
for (let i = 0; i < 1000 && !bodyCancelled; i++) await Promise.resolve();
|
||||
if (!bodyCancelled) throw new Error("Anthropic error-body reader was not cancelled");
|
||||
|
||||
expect(error).toBeInstanceOf(AIError.AbortError);
|
||||
expect(error.message).toBe("Request was aborted.");
|
||||
expect(response?.body?.locked).toBe(false);
|
||||
});
|
||||
|
||||
it("bounds a never-ending error body without a caller signal", async () => {
|
||||
const encoder = new TextEncoder();
|
||||
let readBlocked = false;
|
||||
let bodyCancelled = false;
|
||||
let response: Response | undefined;
|
||||
const openBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(encoder.encode("overloaded"));
|
||||
},
|
||||
pull() {
|
||||
readBlocked = true;
|
||||
return new Promise<void>(() => {});
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const fetch: FetchImpl = async () => {
|
||||
response = new Response(openBody, {
|
||||
status: 529,
|
||||
headers: { "retry-after": "120", "request-id": "req_body_timeout" },
|
||||
});
|
||||
return response;
|
||||
};
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 5, fetch });
|
||||
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(5);
|
||||
try {
|
||||
const error = await client.messages
|
||||
.create(params, { maxRetryDelayMs: 60_000 })
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(readBlocked).toBe(true);
|
||||
expect(error.status).toBe(529);
|
||||
expect(error.message).toBe("529 overloaded");
|
||||
expect(error.headers.get("request-id")).toBe("req_body_timeout");
|
||||
expect(bodyCancelled).toBe(true);
|
||||
expect(response?.body?.locked).toBe(false);
|
||||
} finally {
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(undefined);
|
||||
}
|
||||
});
|
||||
|
||||
it("preserves bounded partial error details when the body stream errors", async () => {
|
||||
const encoder = new TextEncoder();
|
||||
let response: Response | undefined;
|
||||
const partialBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(encoder.encode("partial detail"));
|
||||
},
|
||||
pull(streamController) {
|
||||
streamController.error(new Error("socket closed"));
|
||||
},
|
||||
});
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => {
|
||||
response = new Response(partialBody, { status: 502, headers: { "request-id": "req_partial" } });
|
||||
return response;
|
||||
},
|
||||
});
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error.status).toBe(502);
|
||||
expect(error.message).toBe("502 partial detail");
|
||||
expect(error.headers.get("request-id")).toBe("req_partial");
|
||||
expect(response?.body?.locked).toBe(false);
|
||||
});
|
||||
|
||||
it("flushes an incomplete UTF-8 prefix at clean error-body EOF", async () => {
|
||||
let response: Response | undefined;
|
||||
const incompleteBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(new Uint8Array([0xe2, 0x82]));
|
||||
streamController.close();
|
||||
},
|
||||
});
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => {
|
||||
response = new Response(incompleteBody, { status: 500 });
|
||||
return response;
|
||||
},
|
||||
});
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error.message).toBe("500 \uFFFD");
|
||||
expect(response?.body?.locked).toBe(false);
|
||||
});
|
||||
|
||||
it("keeps complete error text without flushing an incomplete UTF-8 prefix after a read rejection", async () => {
|
||||
const completeText = new TextEncoder().encode("complete text");
|
||||
const completeTextWithIncompletePrefix = new Uint8Array(completeText.byteLength + 2);
|
||||
completeTextWithIncompletePrefix.set(completeText);
|
||||
completeTextWithIncompletePrefix.set([0xe2, 0x82], completeText.byteLength);
|
||||
let response: Response | undefined;
|
||||
let pullErrored = false;
|
||||
const rejectedBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(completeTextWithIncompletePrefix);
|
||||
},
|
||||
pull(streamController) {
|
||||
pullErrored = true;
|
||||
streamController.error(new Error("socket closed"));
|
||||
},
|
||||
});
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => {
|
||||
response = new Response(rejectedBody, { status: 502 });
|
||||
return response;
|
||||
},
|
||||
});
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error.message).toBe("502 complete text");
|
||||
expect(error.message).not.toContain("\uFFFD");
|
||||
expect(response?.body?.locked).toBe(false);
|
||||
expect(pullErrored).toBe(true);
|
||||
});
|
||||
|
||||
it("does not flush an incomplete UTF-8 prefix when the error body times out", async () => {
|
||||
const readBlocked = Promise.withResolvers<void>();
|
||||
let didBlockRead = false;
|
||||
let bodyCancelled = false;
|
||||
const incompleteBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(new Uint8Array([0xe2, 0x82]));
|
||||
},
|
||||
pull() {
|
||||
didBlockRead = true;
|
||||
return readBlocked.promise;
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => new Response(incompleteBody, { status: 500 }),
|
||||
});
|
||||
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(5);
|
||||
try {
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error.message).toBe("500 status code (no body)");
|
||||
expect(error.message).not.toContain("\uFFFD");
|
||||
expect(didBlockRead).toBe(true);
|
||||
expect(bodyCancelled).toBe(true);
|
||||
} finally {
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(undefined);
|
||||
}
|
||||
});
|
||||
|
||||
it("bounds, marks, and cancels a continuously ready oversized error body", async () => {
|
||||
const chunk = new TextEncoder().encode("x".repeat(1024));
|
||||
let bodyCancelled = false;
|
||||
let response: Response | undefined;
|
||||
const oversizedBody = new ReadableStream<Uint8Array>({
|
||||
pull(streamController) {
|
||||
streamController.enqueue(chunk);
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const fetch: FetchImpl = async () => {
|
||||
response = new Response(oversizedBody, { status: 400 });
|
||||
return response;
|
||||
};
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 0, fetch });
|
||||
|
||||
const startedAt = performance.now();
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
for (let i = 0; i < 1000 && !bodyCancelled; i++) await Promise.resolve();
|
||||
if (!bodyCancelled) throw new Error("Anthropic error-body reader was not cancelled");
|
||||
|
||||
expect(performance.now() - startedAt).toBeLessThan(1_000);
|
||||
expect(error.message).toBe(`400 ${"x".repeat(64 * 1024)}\n[Response body truncated after 64 KiB]`);
|
||||
expect(response?.body?.locked).toBe(false);
|
||||
});
|
||||
|
||||
it("preserves an exact 64 KiB error body without marking or cancelling it", async () => {
|
||||
const body = new Uint8Array(64 * 1024).fill(120);
|
||||
let bodyCancelled = false;
|
||||
const exactBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(body);
|
||||
streamController.close();
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => new Response(exactBody, { status: 400 }),
|
||||
});
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error.message).toBe(`400 ${"x".repeat(64 * 1024)}`);
|
||||
expect(bodyCancelled).toBe(false);
|
||||
});
|
||||
|
||||
it("marks and cancels only after observing the 64 KiB-plus-one error-body byte", async () => {
|
||||
const firstChunk = new Uint8Array(64 * 1024).fill(120);
|
||||
const overflowByte = new Uint8Array([120]);
|
||||
let bodyCancelled = false;
|
||||
const overflowingBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(firstChunk);
|
||||
streamController.enqueue(overflowByte);
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => new Response(overflowingBody, { status: 400 }),
|
||||
});
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error.message).toBe(`400 ${"x".repeat(64 * 1024)}\n[Response body truncated after 64 KiB]`);
|
||||
expect(bodyCancelled).toBe(true);
|
||||
});
|
||||
|
||||
it("does not append a replacement character for UTF-8 split by the truncation boundary", async () => {
|
||||
const body = new Uint8Array(64 * 1024 + 2).fill(120);
|
||||
body.set([0xe2, 0x82, 0xac], 64 * 1024 - 1);
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => new Response(body, { status: 400 }),
|
||||
});
|
||||
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(error.message).toBe(`400 ${"x".repeat(64 * 1024 - 1)}\n[Response body truncated after 64 KiB]`);
|
||||
expect(error.message).not.toContain("\uFFFD");
|
||||
});
|
||||
|
||||
it("bounds continuously-ready empty error-body chunks without accumulating read reactions", async () => {
|
||||
let pulls = 0;
|
||||
let bodyCancelled = false;
|
||||
const emptyBody = new ReadableStream<Uint8Array>({
|
||||
pull(streamController) {
|
||||
pulls += 1;
|
||||
streamController.enqueue(new Uint8Array());
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch: async () => new Response(emptyBody, { status: 500 }),
|
||||
});
|
||||
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(5);
|
||||
try {
|
||||
const startedAt = performance.now();
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
|
||||
expect(performance.now() - startedAt).toBeLessThan(1_000);
|
||||
expect(pulls).toBeGreaterThan(0);
|
||||
expect(pulls).toBeLessThan(100_000);
|
||||
expect(bodyCancelled).toBe(true);
|
||||
expect(error.message).toBe("500 status code (no body)");
|
||||
} finally {
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(undefined);
|
||||
}
|
||||
});
|
||||
|
||||
it("checks the deadline before an always-ready error body can starve timer callbacks", async () => {
|
||||
let bodyCancelled = false;
|
||||
let response: Response | undefined;
|
||||
const alwaysReadyBody = new ReadableStream<Uint8Array>({
|
||||
start(streamController) {
|
||||
streamController.enqueue(new Uint8Array([120]));
|
||||
},
|
||||
pull(streamController) {
|
||||
streamController.enqueue(new Uint8Array([120]));
|
||||
},
|
||||
cancel() {
|
||||
bodyCancelled = true;
|
||||
},
|
||||
});
|
||||
const fetch: FetchImpl = async () => {
|
||||
response = new Response(alwaysReadyBody, { status: 500 });
|
||||
return response;
|
||||
};
|
||||
const client = new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 0, fetch });
|
||||
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(0);
|
||||
try {
|
||||
const startedAt = performance.now();
|
||||
const error = await client.messages
|
||||
.create(params)
|
||||
.asResponse()
|
||||
.catch(err => err as AIError.AnthropicApiError);
|
||||
if (!(error instanceof AIError.AnthropicApiError)) throw new Error("Expected AnthropicApiError");
|
||||
for (let i = 0; i < 1000 && !bodyCancelled; i++) await Promise.resolve();
|
||||
if (!bodyCancelled) throw new Error("Anthropic error-body reader was not cancelled");
|
||||
|
||||
expect(performance.now() - startedAt).toBeLessThan(1_000);
|
||||
expect(error.message).toBe("500 status code (no body)");
|
||||
expect(response?.body?.locked).toBe(false);
|
||||
} finally {
|
||||
AIError.__anthropicApiErrorForTesting.setBodyReadTimeoutMs(undefined);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,6 +1,10 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { isFastModeUnsupported } from "@oh-my-pi/pi-ai/error";
|
||||
import { clearAnthropicFastModeFallback, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import {
|
||||
clearAnthropicFastModeFallback,
|
||||
isAnthropicFastModeFallbackDisabled,
|
||||
streamAnthropic,
|
||||
} from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type { Context, Model, ProviderSessionState, ServiceTier } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
@@ -88,6 +92,49 @@ describe("Anthropic priority service tier → speed='fast'", () => {
|
||||
expect(payload.speed).toBeUndefined();
|
||||
}
|
||||
});
|
||||
describe("provider-session fallback scope", () => {
|
||||
function stateKey(model: Model<"anthropic-messages">): string {
|
||||
return `anthropic-messages:${model.baseUrl}\u0000${model.id}`;
|
||||
}
|
||||
|
||||
it("inspects the exact model and endpoint fallback without bleeding to other requests", async () => {
|
||||
const model = makeAnthropicModel("claude-opus-4-7");
|
||||
const otherModel = makeAnthropicModel("claude-opus-4-6");
|
||||
const otherEndpoint = buildModel({
|
||||
...model,
|
||||
id: model.id,
|
||||
name: `${model.id} gateway`,
|
||||
baseUrl: "https://gateway.example/v1",
|
||||
});
|
||||
const state = new Map<string, ProviderSessionState>();
|
||||
state.set(stateKey(model), {
|
||||
strictToolsDisabled: false,
|
||||
fastModeDisabled: true,
|
||||
replayUnsignedThinkingDisabled: false,
|
||||
close: () => {},
|
||||
} as ProviderSessionState);
|
||||
|
||||
const matching = (await capturePayload(model, {
|
||||
serviceTier: "priority",
|
||||
providerSessionState: state,
|
||||
})) as Record<string, unknown>;
|
||||
const differentModel = (await capturePayload(otherModel, {
|
||||
serviceTier: "priority",
|
||||
providerSessionState: state,
|
||||
})) as Record<string, unknown>;
|
||||
const differentEndpoint = (await capturePayload(otherEndpoint, {
|
||||
serviceTier: "priority",
|
||||
providerSessionState: state,
|
||||
})) as Record<string, unknown>;
|
||||
|
||||
expect(isAnthropicFastModeFallbackDisabled(state, model)).toBe(true);
|
||||
expect(isAnthropicFastModeFallbackDisabled(state, otherModel)).toBe(false);
|
||||
expect(isAnthropicFastModeFallbackDisabled(state, otherEndpoint)).toBe(false);
|
||||
expect(matching.speed).toBeUndefined();
|
||||
expect(differentModel.speed).toBe("fast");
|
||||
expect(differentEndpoint.speed).toBe("fast");
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe("clearAnthropicFastModeFallback", () => {
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as AIError from "@oh-my-pi/pi-ai/error";
|
||||
import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type { AnthropicMessagesClientLike } from "@oh-my-pi/pi-ai/providers/anthropic-client";
|
||||
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { AnthropicMessagesClient, type AnthropicMessagesClientLike } from "@oh-my-pi/pi-ai/providers/anthropic-client";
|
||||
import type { Context, FetchImpl, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import { waitForDelayOrAbort } from "./helpers";
|
||||
|
||||
@@ -90,6 +90,36 @@ function createSuccessfulAnthropicEvents(text: string): MockAnthropicEvent[] {
|
||||
];
|
||||
}
|
||||
|
||||
function createAnthropicSseResponse(text: string): Response {
|
||||
const body = createSuccessfulAnthropicEvents(text)
|
||||
.map(event => `event: ${event.type}\ndata: ${JSON.stringify(event)}\n\n`)
|
||||
.join("");
|
||||
return new Response(body, {
|
||||
status: 200,
|
||||
headers: { "content-type": "text/event-stream", "request-id": "req_retry_success" },
|
||||
});
|
||||
}
|
||||
|
||||
function createResponseClient(responses: Response[]): {
|
||||
calls: { count: number };
|
||||
client: AnthropicMessagesClientLike;
|
||||
} {
|
||||
const calls = { count: 0 };
|
||||
const fetch: FetchImpl = async () => {
|
||||
const response = responses[Math.min(calls.count++, responses.length - 1)];
|
||||
if (!response) throw new Error("Expected an Anthropic mock response");
|
||||
return new Response(response.body, {
|
||||
status: response.status,
|
||||
statusText: response.statusText,
|
||||
headers: response.headers,
|
||||
});
|
||||
};
|
||||
return {
|
||||
calls,
|
||||
client: new AnthropicMessagesClient({ apiKey: "sk-test", maxRetries: 0, fetch }),
|
||||
};
|
||||
}
|
||||
|
||||
function createAnthropicMockStream({
|
||||
signal,
|
||||
connectDelayMs = 0,
|
||||
@@ -533,3 +563,195 @@ describe("anthropic provider retry delays", () => {
|
||||
expect(JSON.parse(JSON.stringify(result.content))).toEqual([{ type: "text", text: "recovered from 502" }]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("anthropic retry-after cap (maxRetryDelayMs)", () => {
|
||||
it("surfaces the original HTTP error without a second attempt when retry-after exceeds the default 60s cap", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"rate_limit_error","message":"Too many requests"}}', {
|
||||
status: 429,
|
||||
headers: { "retry-after": "120" },
|
||||
}),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, { client, providerRetryWait }).result();
|
||||
|
||||
expect(calls.count).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(429);
|
||||
expect(result.errorMessage).toContain("rate_limit_error");
|
||||
});
|
||||
|
||||
it("surfaces the original HTTP error when retry-after exceeds an explicit cap", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after-ms": "10000" },
|
||||
}),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 5_000,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(529);
|
||||
});
|
||||
|
||||
it("disables the cap when maxRetryDelayMs is negative", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after": "1" },
|
||||
}),
|
||||
createAnthropicSseResponse("after unbounded wait"),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: -1,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(1_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("disables the cap when maxRetryDelayMs is 0 and waits the full server hint", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after": "120" },
|
||||
}),
|
||||
createAnthropicSseResponse("after long wait"),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 0,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(120_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("retries when the HTTP retry-after hint is under the cap", async () => {
|
||||
const { calls, client } = createResponseClient([
|
||||
new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 529,
|
||||
headers: { "retry-after": "30" },
|
||||
}),
|
||||
createAnthropicSseResponse("after backoff"),
|
||||
]);
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client,
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 60_000,
|
||||
}).result();
|
||||
|
||||
expect(calls.count).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(30_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
});
|
||||
|
||||
it("honors retry headers from structurally compatible injected SDK errors", async () => {
|
||||
let attempt = 0;
|
||||
const error = Object.assign(new Error("529 overloaded"), {
|
||||
status: 529,
|
||||
headers: new Headers({ "retry-after-ms": "10000" }),
|
||||
});
|
||||
const create = ((_body: unknown) => {
|
||||
attempt += 1;
|
||||
return createRejectedAnthropicRequest(error) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client: { messages: { create } },
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 5_000,
|
||||
}).result();
|
||||
|
||||
expect(attempt).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(529);
|
||||
});
|
||||
|
||||
it("honors record-valued retry headers from injected SDK errors", async () => {
|
||||
let attempt = 0;
|
||||
const error = Object.assign(new Error("529 overloaded"), {
|
||||
status: 529,
|
||||
headers: { "Retry-After-Ms": "10000" },
|
||||
});
|
||||
const create = ((_body: unknown) => {
|
||||
attempt += 1;
|
||||
return createRejectedAnthropicRequest(error) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client: { messages: { create } },
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 5_000,
|
||||
}).result();
|
||||
|
||||
expect(attempt).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(529);
|
||||
});
|
||||
|
||||
it("honors nested response retry headers from injected SDK errors", async () => {
|
||||
let attempt = 0;
|
||||
const error = Object.assign(new Error("529 overloaded"), {
|
||||
response: { status: 529, headers: { "retry-after-ms": "10000" } },
|
||||
});
|
||||
const create = ((_body: unknown) => {
|
||||
attempt += 1;
|
||||
return createRejectedAnthropicRequest(error) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, {
|
||||
client: { messages: { create } },
|
||||
providerRetryWait,
|
||||
maxRetryDelayMs: 5_000,
|
||||
}).result();
|
||||
|
||||
expect(attempt).toBe(1);
|
||||
expect(providerRetryWait).not.toHaveBeenCalled();
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(529);
|
||||
});
|
||||
|
||||
it("passes maxRetryDelayMs to internally constructed Anthropic clients", async () => {
|
||||
let calls = 0;
|
||||
const fetch: FetchImpl = async () => {
|
||||
calls += 1;
|
||||
return new Response('{"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}', {
|
||||
status: 429,
|
||||
headers: { "retry-after-ms": "10" },
|
||||
});
|
||||
};
|
||||
|
||||
const result = await streamAnthropic(model, context, { fetch, maxRetryDelayMs: 5 }).result();
|
||||
|
||||
expect(calls).toBe(1);
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorStatus).toBe(429);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -689,6 +689,7 @@ function newBlockState(): BlockState {
|
||||
get currentToolCall() {
|
||||
return toolCall;
|
||||
},
|
||||
openToolCalls: new Map(),
|
||||
resolvedMcpToolCallIds: new Set(),
|
||||
firstTokenTime: undefined,
|
||||
setTextBlock: b => {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -44,4 +44,16 @@ describe("cursor buildMcpToolDefinitions", () => {
|
||||
const names = buildMcpToolDefinitions([tool("read"), tool("write"), tool("bash")]).map(def => def.name);
|
||||
expect(names).toEqual([]);
|
||||
});
|
||||
|
||||
it("advertises lsp, which Cursor has no native equivalent for", () => {
|
||||
// `lsp` is deliberately absent from the native-filtered set: Cursor's own
|
||||
// tools cover none of definition/references/rename, so filtering it out
|
||||
// leaves the model with no way to reach them at all.
|
||||
const defs = buildMcpToolDefinitions([tool("read"), tool("bash"), tool("lsp")]);
|
||||
|
||||
const lspDef = defs.find(def => def.name === "lsp");
|
||||
expect(lspDef).toBeDefined();
|
||||
expect(lspDef?.providerIdentifier).toBe("pi-agent");
|
||||
expect(lspDef?.toolName).toBe("lsp");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -58,6 +58,7 @@ function newHarness(): Harness {
|
||||
get currentToolCall() {
|
||||
return toolCall;
|
||||
},
|
||||
openToolCalls: new Map(),
|
||||
resolvedMcpToolCallIds: new Set(),
|
||||
firstTokenTime: undefined,
|
||||
setTextBlock: b => {
|
||||
|
||||
@@ -2,7 +2,7 @@ import { afterEach, describe, expect, it } from "bun:test";
|
||||
import * as http2 from "node:http2";
|
||||
import { create, toBinary } from "@bufbuild/protobuf";
|
||||
import { streamCursor } from "@oh-my-pi/pi-ai/providers/cursor";
|
||||
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import type { Context, CursorToolResultHandler, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
import {
|
||||
AgentServerMessageSchema,
|
||||
@@ -10,7 +10,11 @@ import {
|
||||
InteractionUpdateSchema,
|
||||
ReadArgsSchema,
|
||||
TextDeltaUpdateSchema,
|
||||
ToolCallSchema,
|
||||
ToolCallStartedUpdateSchema,
|
||||
TurnEndedUpdateSchema,
|
||||
UpdateTodosArgsSchema,
|
||||
UpdateTodosToolCallSchema,
|
||||
} from "@oh-my-pi/pi-catalog/discovery/cursor-gen/agent_pb";
|
||||
|
||||
const CONNECT_END_STREAM_FLAG = 0b00000010;
|
||||
@@ -23,7 +27,8 @@ type Scenario =
|
||||
| { kind: "hang-after-turn" }
|
||||
| { kind: "exec-in-final-chunk"; responseFinished: PromiseWithResolvers<void> }
|
||||
| { kind: "exec-then-transport-error"; responseFinished: PromiseWithResolvers<void> }
|
||||
| { kind: "exec-then-hang" };
|
||||
| { kind: "exec-then-hang" }
|
||||
| { kind: "todo-start-then-death" };
|
||||
|
||||
let server: http2.Http2Server | undefined;
|
||||
const sessions = new Set<http2.Http2Session>();
|
||||
@@ -104,6 +109,36 @@ function execAndTurnEndedFrame(): Buffer {
|
||||
return Buffer.concat([execRequestFrame(), turnEndedFrame()]);
|
||||
}
|
||||
|
||||
/**
|
||||
* A native `update_todos` call announcement. Cursor runs these server-side, so
|
||||
* the block is stamped resolved at start and only its `toolCallCompleted`
|
||||
* frame pairs a result — nothing downstream synthesizes one.
|
||||
*/
|
||||
function todoStartFrame(): Buffer {
|
||||
const message = create(AgentServerMessageSchema, {
|
||||
message: {
|
||||
case: "interactionUpdate",
|
||||
value: create(InteractionUpdateSchema, {
|
||||
message: {
|
||||
case: "toolCallStarted",
|
||||
value: create(ToolCallStartedUpdateSchema, {
|
||||
callId: "todo-envelope",
|
||||
toolCall: create(ToolCallSchema, {
|
||||
tool: {
|
||||
case: "updateTodosToolCall",
|
||||
value: create(UpdateTodosToolCallSchema, {
|
||||
args: create(UpdateTodosArgsSchema, { todos: [] }),
|
||||
}),
|
||||
},
|
||||
}),
|
||||
}),
|
||||
},
|
||||
}),
|
||||
},
|
||||
});
|
||||
return frameConnectMessage(toBinary(AgentServerMessageSchema, message));
|
||||
}
|
||||
|
||||
async function startServer(): Promise<string> {
|
||||
server = http2.createServer();
|
||||
server.on("session", session => {
|
||||
@@ -150,6 +185,16 @@ async function startServer(): Promise<string> {
|
||||
return;
|
||||
}
|
||||
|
||||
if (scenario.kind === "todo-start-then-death") {
|
||||
// The server announces a native todo call, then the stream dies
|
||||
// without `turnEnded` and without the call's completion frame. This
|
||||
// is the real interrupted-call shape: `settleH2` rejects, so the
|
||||
// success-path flush never runs.
|
||||
stream.write(todoStartFrame());
|
||||
stream.end();
|
||||
return;
|
||||
}
|
||||
|
||||
if (scenario.kind === "exec-in-final-chunk") {
|
||||
const { responseFinished } = scenario;
|
||||
// Resolves once the server has flushed the whole response, so the test
|
||||
@@ -226,8 +271,15 @@ const context: Context = {
|
||||
messages: [{ role: "user", content: "terminal lifecycle", timestamp: 1 }],
|
||||
};
|
||||
|
||||
async function collectStream(model: Model<"cursor-agent">, options?: { signal?: AbortSignal }) {
|
||||
const stream = streamCursor(model, context, { apiKey: "test-token", signal: options?.signal });
|
||||
async function collectStream(
|
||||
model: Model<"cursor-agent">,
|
||||
options?: { signal?: AbortSignal; onToolResult?: CursorToolResultHandler },
|
||||
) {
|
||||
const stream = streamCursor(model, context, {
|
||||
apiKey: "test-token",
|
||||
signal: options?.signal,
|
||||
onToolResult: options?.onToolResult,
|
||||
});
|
||||
const eventTypes: string[] = [];
|
||||
for await (const event of stream) {
|
||||
eventTypes.push(event.type);
|
||||
@@ -303,6 +355,37 @@ describe("Cursor terminal lifecycle after turnEnded", () => {
|
||||
expect(result.errorMessage).toContain("Cursor stream ended before turnEnded");
|
||||
});
|
||||
|
||||
it("pairs and closes a server-owned call the dying stream left open", async () => {
|
||||
// The failure this guards: a native todo block is stamped resolved at
|
||||
// start, so `agent-loop.ts` synthesizes no placeholder for it and only
|
||||
// its completion frame pairs a result. When the transport dies first the
|
||||
// call went unpaired and its card stayed animating — and
|
||||
// `buildSessionContext` strips a dangling call, so the interaction
|
||||
// vanished from every rebuilt transcript.
|
||||
//
|
||||
// This must run against the real terminal-error path: `settleH2` rejects
|
||||
// on a stream that ends before `turnEnded`, so the success path's flush
|
||||
// is never reached.
|
||||
scenario = { kind: "todo-start-then-death" };
|
||||
const baseUrl = await startServer();
|
||||
const paired: ToolResultMessage[] = [];
|
||||
const { eventTypes, result } = await collectStream(makeModel(baseUrl), {
|
||||
onToolResult: toolResult => void paired.push(toolResult),
|
||||
});
|
||||
|
||||
expect(eventTypes.at(-1)).toBe("error");
|
||||
expect(result.stopReason).toBe("error");
|
||||
|
||||
const call = result.content.find(block => block.type === "toolCall");
|
||||
if (!call) throw new Error("expected the announced todo call in the output");
|
||||
// Closed, so no live card is left animating.
|
||||
expect(eventTypes).toContain("toolcall_end");
|
||||
// Paired, so replay keeps the interaction.
|
||||
expect(paired).toHaveLength(1);
|
||||
expect(paired[0].toolCallId).toBe(call.id);
|
||||
expect(paired[0].isError).toBe(true);
|
||||
});
|
||||
|
||||
it("aborts without emitting done when the signal fires", async () => {
|
||||
scenario = { kind: "hang-after-turn" };
|
||||
const baseUrl = await startServer();
|
||||
|
||||
@@ -89,6 +89,7 @@ function newHarness(): Harness {
|
||||
get currentToolCall() {
|
||||
return toolCall;
|
||||
},
|
||||
openToolCalls: new Map(),
|
||||
resolvedMcpToolCallIds: new Set(),
|
||||
firstTokenTime: undefined,
|
||||
setTextBlock: b => {
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { loginExa } from "@oh-my-pi/pi-ai/registry/exa";
|
||||
|
||||
describe("exa login", () => {
|
||||
it("opens Exa API-key settings and returns a trimmed key without validation requests", async () => {
|
||||
let authUrl: string | undefined;
|
||||
let authInstructions: string | undefined;
|
||||
let promptMessage: string | undefined;
|
||||
let promptPlaceholder: string | undefined;
|
||||
|
||||
const apiKey = await loginExa({
|
||||
onAuth: info => {
|
||||
authUrl = info.url;
|
||||
authInstructions = info.instructions;
|
||||
},
|
||||
onPrompt: async prompt => {
|
||||
promptMessage = prompt.message;
|
||||
promptPlaceholder = prompt.placeholder;
|
||||
return " exa-test-key ";
|
||||
},
|
||||
fetch: () => {
|
||||
throw new Error("Exa login must not make a network request");
|
||||
},
|
||||
});
|
||||
|
||||
expect(authUrl).toBe("https://dashboard.exa.ai/api-keys");
|
||||
expect(authInstructions).toBe("Create or copy your API key from the Exa dashboard.");
|
||||
expect(promptMessage).toBe("Paste your Exa API key");
|
||||
expect(promptPlaceholder).toBe("API key");
|
||||
expect(apiKey).toBe("exa-test-key");
|
||||
});
|
||||
});
|
||||
@@ -585,14 +585,26 @@ describe("normalizeSchemaForGoogle parity with python-genai process_schema", ()
|
||||
expect(sanitized).toEqual({ type: "string", enum: ["FOO"] });
|
||||
});
|
||||
|
||||
// Mirrors python-genai test_schema.py::test_process_schema_forbids_non_string_const
|
||||
// We deviate intentionally: rather than raise on non-string const we accept
|
||||
// the value as a singleton enum. Google's Schema proto accepts numeric enums
|
||||
// and we prefer permissive normalization over surfacing a transformer-level error.
|
||||
it("accepts non-string const as a singleton enum (intentional deviation from upstream raise)", () => {
|
||||
const sanitized = normalizeSchemaForGoogle({ type: "integer", const: 123 }) as Record<string, unknown>;
|
||||
expect(sanitized.enum).toEqual([123]);
|
||||
expect(sanitized.type).toBe("integer");
|
||||
// Mirrors python-genai test_schema.py::test_process_schema_forbids_non_string_const.
|
||||
// Google enum fields accept strings only, so normalization drops the numeric
|
||||
// singleton enum while preserving the integer type constraint.
|
||||
it("omits a non-string const enum while preserving its type", () => {
|
||||
const sanitized = normalizeSchemaForGoogle({ type: "integer", const: 123 });
|
||||
expect(sanitized).toEqual({ type: "integer" });
|
||||
});
|
||||
|
||||
it("drops negations whose non-string enums cannot be represented", () => {
|
||||
const sanitized = normalizeSchemaForGoogle({ not: { enum: [1] } });
|
||||
expect(sanitized).toEqual({});
|
||||
expect(
|
||||
normalizeSchemaForGoogle({
|
||||
not: { type: "object", properties: { value: { enum: [1] } } },
|
||||
}),
|
||||
).toEqual({});
|
||||
});
|
||||
|
||||
it("drops negations containing snake-case combiners with non-string enums", () => {
|
||||
expect(normalizeSchemaForGoogle({ not: { any_of: [{ const: 1 }] } })).toEqual({});
|
||||
});
|
||||
|
||||
// Mirrors python-genai test_schema.py::test_process_schema_order_properties_propagates_into_defs
|
||||
|
||||
@@ -47,7 +47,7 @@ describe("provider registry auth surface", () => {
|
||||
expect(getEnvApiKey("umans")).toBe("umans-env");
|
||||
Bun.env.LLAMA_CPP_API_KEY = "llama-env";
|
||||
expect(getEnvApiKey("llama.cpp")).toBe("llama-env");
|
||||
// Legacy search-tool key preserved (not a registry provider def).
|
||||
// Exa is derived from the provider registry's `envKeys` definition.
|
||||
expect(getEnvApiKey("exa")).toBe("exa-env");
|
||||
});
|
||||
|
||||
@@ -65,6 +65,7 @@ describe("provider registry auth surface", () => {
|
||||
const ids = getOAuthProviders().map(provider => provider.id);
|
||||
expect(ids).toContain("zenmux");
|
||||
expect(ids).toContain("kagi");
|
||||
expect(ids).toContain("exa");
|
||||
expect(ids).toContain("umans");
|
||||
expect(ids).toContain("llama.cpp");
|
||||
// openai has no interactive login flow.
|
||||
|
||||
@@ -203,7 +203,58 @@ describe("upgradeJsonSchemaTo202012", () => {
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe("normalizeSchemaForGoogle", () => {
|
||||
it("sets object type when converting an object const to an enum entry", () => {
|
||||
it("preserves string enums and removes enums Google cannot represent", () => {
|
||||
const sanitized = normalizeSchemaForGoogle({
|
||||
type: "object",
|
||||
properties: {
|
||||
valid: { type: "string", enum: ["draft", "published"] },
|
||||
numeric: { type: "number", enum: [1, 2] },
|
||||
mixed: { enum: ["draft", 1] },
|
||||
},
|
||||
}) as { properties: Record<string, unknown> };
|
||||
|
||||
expect(sanitized.properties).toEqual({
|
||||
valid: { type: "string", enum: ["draft", "published"] },
|
||||
numeric: { type: "number" },
|
||||
mixed: {},
|
||||
});
|
||||
});
|
||||
|
||||
it("preserves enum keys inside object-valued defaults", () => {
|
||||
expect(
|
||||
normalizeSchemaForGoogle({
|
||||
type: "object",
|
||||
default: { enum: [1], value: 2 },
|
||||
}),
|
||||
).toEqual({
|
||||
type: "object",
|
||||
default: { enum: [1], value: 2 },
|
||||
properties: {},
|
||||
});
|
||||
});
|
||||
|
||||
it("keeps CCA incompatibility passes out of literal defaults", () => {
|
||||
const literal = {
|
||||
nullable: true,
|
||||
allOf: [{ type: "object" }],
|
||||
oneOf: [{ type: "string" }, { type: "number" }],
|
||||
};
|
||||
expect(
|
||||
normalizeSchemaForCCA({
|
||||
type: "object",
|
||||
properties: {
|
||||
value: { oneOf: [{ type: "string" }, { type: "string" }] },
|
||||
},
|
||||
default: literal,
|
||||
}),
|
||||
).toEqual({
|
||||
type: "object",
|
||||
properties: { value: { type: "string" } },
|
||||
default: literal,
|
||||
});
|
||||
});
|
||||
|
||||
it("sets object type while removing an object-valued enum converted from const", () => {
|
||||
const sanitized = normalizeSchemaForGoogle({
|
||||
const: { a: 1 },
|
||||
});
|
||||
@@ -211,11 +262,10 @@ describe("normalizeSchemaForGoogle", () => {
|
||||
expect(sanitized).toEqual({
|
||||
type: "object",
|
||||
properties: {},
|
||||
enum: [{ a: 1 }],
|
||||
});
|
||||
});
|
||||
|
||||
it("deduplicates a deep-equal object const against an existing enum entry", () => {
|
||||
it("removes an object-valued enum after deduplicating a deep-equal const", () => {
|
||||
const sanitized = normalizeSchemaForGoogle({
|
||||
type: "object",
|
||||
enum: [{ a: 1 }],
|
||||
@@ -225,11 +275,10 @@ describe("normalizeSchemaForGoogle", () => {
|
||||
expect(sanitized).toEqual({
|
||||
type: "object",
|
||||
properties: {},
|
||||
enum: [{ a: 1 }],
|
||||
});
|
||||
});
|
||||
|
||||
it("does not stamp a wrong scalar type when const variants span multiple primitive types", () => {
|
||||
it("removes an enum when const variants span multiple primitive types", () => {
|
||||
const sanitized = normalizeSchemaForGoogle({
|
||||
anyOf: [
|
||||
{ const: "A", type: "string" },
|
||||
@@ -238,18 +287,18 @@ describe("normalizeSchemaForGoogle", () => {
|
||||
],
|
||||
}) as Record<string, unknown>;
|
||||
|
||||
expect(sanitized.enum).toEqual(["A", 1, true]);
|
||||
expect(sanitized.enum).toBeUndefined();
|
||||
expect(sanitized.type).toBeUndefined();
|
||||
});
|
||||
|
||||
it("collapses inferred null type to nullable when const is null", () => {
|
||||
it("collapses inferred null type to nullable while removing its enum", () => {
|
||||
// After python-genai parity (handle_null_fields), bare `type: 'null'` is
|
||||
// folded into `nullable: true` so the schema is OpenAPI-compatible.
|
||||
const sanitized = normalizeSchemaForGoogle({ const: null }) as Record<string, unknown>;
|
||||
|
||||
expect(sanitized.type).toBeUndefined();
|
||||
expect(sanitized.nullable).toBe(true);
|
||||
expect(sanitized.enum).toEqual([null]);
|
||||
expect(sanitized.enum).toBeUndefined();
|
||||
});
|
||||
|
||||
it("coerces a boolean subschema literally named additionalProperties inside properties", () => {
|
||||
@@ -1277,6 +1326,21 @@ describe("normalizeSchemaForMoonshot", () => {
|
||||
expect(props.limit).toEqual({ type: "integer", default: 10 });
|
||||
});
|
||||
|
||||
it("normalizes schema-valued additionalProperties without walking literal payload objects", () => {
|
||||
const literal = { oneOf: [{ const: "literal-a" }, { const: "literal-b" }] };
|
||||
const normalized = normalizeSchemaForMoonshot({
|
||||
type: "object",
|
||||
additionalProperties: { oneOf: [{ const: 1 }, { const: 2 }] },
|
||||
default: literal,
|
||||
});
|
||||
|
||||
expect(normalized).toEqual({
|
||||
type: "object",
|
||||
additionalProperties: { type: "number", enum: [1, 2] },
|
||||
default: literal,
|
||||
});
|
||||
});
|
||||
|
||||
it("coerces boolean subschemas to MFJS object forms without changing boolean keywords", () => {
|
||||
expect(
|
||||
normalizeSchemaForMoonshot({
|
||||
|
||||
@@ -0,0 +1,225 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import type { FetchImpl } from "@oh-my-pi/pi-ai/types";
|
||||
import { umansUsageProvider } from "../src/usage/umans";
|
||||
|
||||
const DEFAULT_BASE_URL = "https://api.code.umans.ai";
|
||||
|
||||
function umansPayload(overrides: Record<string, unknown> = {}): Record<string, unknown> {
|
||||
return {
|
||||
plan: { display_name: "Code Max" },
|
||||
limits: {
|
||||
requests: { limit: 200, hard_cap: 400, burst_pct: 1.0, window_seconds: 18000 },
|
||||
concurrency: { limit: 4, hard_cap: 8, burst_pct: 1.0 },
|
||||
},
|
||||
usage: {
|
||||
requests_in_window: 48,
|
||||
remaining_requests: 152,
|
||||
concurrent_sessions: 1,
|
||||
tokens_in: 1_200_000,
|
||||
tokens_out: 340_000,
|
||||
priority: { low: false, boxed_until: null, reason: null },
|
||||
},
|
||||
...overrides,
|
||||
};
|
||||
}
|
||||
|
||||
function fakeFetch(payload: unknown, status = 200): FetchImpl {
|
||||
const fn = async () =>
|
||||
new Response(JSON.stringify(payload), {
|
||||
status,
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
return fn as unknown as typeof fetch;
|
||||
}
|
||||
|
||||
function fetchRecorder(
|
||||
calls: Array<{ url: string; headers: Record<string, string> }>,
|
||||
payload: unknown,
|
||||
status = 200,
|
||||
): FetchImpl {
|
||||
const fn = async (input: string | URL | Request, init?: RequestInit) => {
|
||||
calls.push({
|
||||
url: String(input),
|
||||
headers: (init?.headers as Record<string, string>) ?? {},
|
||||
});
|
||||
return new Response(JSON.stringify(payload), {
|
||||
status,
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
};
|
||||
return fn as unknown as typeof fetch;
|
||||
}
|
||||
|
||||
describe("umans usage provider", () => {
|
||||
it("parses the rolling 5h request window into a UsageLimit with used/remaining/fraction", async () => {
|
||||
const report = await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test", accountId: "acct-1", email: "u@example.com" },
|
||||
},
|
||||
{ fetch: fakeFetch(umansPayload()) },
|
||||
);
|
||||
expect(report).not.toBeNull();
|
||||
const requests = report?.limits.find(l => l.id === "umans:requests");
|
||||
expect(requests).toBeDefined();
|
||||
expect(requests?.amount.used).toBe(48);
|
||||
expect(requests?.amount.limit).toBe(200);
|
||||
expect(requests?.amount.remaining).toBe(152);
|
||||
expect(requests?.amount.usedFraction).toBeCloseTo(0.24, 5);
|
||||
expect(requests?.amount.remainingFraction).toBeCloseTo(0.76, 5);
|
||||
expect(requests?.amount.unit).toBe("requests");
|
||||
// Rolling window: no fabricated reset timestamp.
|
||||
expect(requests?.window?.resetsAt).toBeUndefined();
|
||||
expect(requests?.window?.durationMs).toBe(18000_000);
|
||||
expect(requests?.window?.label).toBe("rolling 5h");
|
||||
});
|
||||
|
||||
it("emits a concurrency limit from limits.concurrency", async () => {
|
||||
const report = await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
},
|
||||
{ fetch: fakeFetch(umansPayload()) },
|
||||
);
|
||||
const concurrency = report?.limits.find(l => l.id === "umans:concurrency");
|
||||
expect(concurrency).toBeDefined();
|
||||
expect(concurrency?.amount.used).toBe(1);
|
||||
expect(concurrency?.amount.limit).toBe(4);
|
||||
expect(concurrency?.amount.unit).toBe("requests");
|
||||
});
|
||||
|
||||
it("sends Authorization: Bearer <key> to the default base URL", async () => {
|
||||
const calls: Array<{ url: string; headers: Record<string, string> }> = [];
|
||||
await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
},
|
||||
{ fetch: fetchRecorder(calls, umansPayload()) },
|
||||
);
|
||||
expect(calls).toHaveLength(1);
|
||||
expect(calls[0]?.url).toBe(`${DEFAULT_BASE_URL}/v1/usage`);
|
||||
expect(calls[0]?.headers.authorization).toBe("Bearer sk-test");
|
||||
});
|
||||
|
||||
it("honors a custom baseUrl from params", async () => {
|
||||
const calls: Array<{ url: string; headers: Record<string, string> }> = [];
|
||||
await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
baseUrl: "https://custom.umans.example",
|
||||
},
|
||||
{ fetch: fetchRecorder(calls, umansPayload()) },
|
||||
);
|
||||
expect(calls[0]?.url).toBe("https://custom.umans.example/v1/usage");
|
||||
});
|
||||
|
||||
it("strips a trailing /v1 from a custom baseUrl", async () => {
|
||||
const calls: Array<{ url: string; headers: Record<string, string> }> = [];
|
||||
await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
baseUrl: "https://api.code.umans.ai/v1",
|
||||
},
|
||||
{ fetch: fetchRecorder(calls, umansPayload()) },
|
||||
);
|
||||
expect(calls[0]?.url).toBe("https://api.code.umans.ai/v1/usage");
|
||||
});
|
||||
|
||||
it("preserves a path-mounted gateway prefix while stripping /v1", async () => {
|
||||
const calls: Array<{ url: string; headers: Record<string, string> }> = [];
|
||||
await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
baseUrl: "https://gateway.example/team/umans/v1",
|
||||
},
|
||||
{ fetch: fetchRecorder(calls, umansPayload()) },
|
||||
);
|
||||
expect(calls[0]?.url).toBe("https://gateway.example/team/umans/v1/usage");
|
||||
});
|
||||
|
||||
it("surfaces priority.low as a provider note", async () => {
|
||||
const payload = umansPayload({
|
||||
usage: {
|
||||
requests_in_window: 250,
|
||||
remaining_requests: 0,
|
||||
concurrent_sessions: 1,
|
||||
tokens_in: 0,
|
||||
tokens_out: 0,
|
||||
priority: { low: true, boxed_until: "2026-06-27T12:00:00Z", reason: "burst" },
|
||||
},
|
||||
});
|
||||
const report = await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
},
|
||||
{ fetch: fakeFetch(payload) },
|
||||
);
|
||||
expect(report?.notes).toContain("Requests deprioritized after a rate-limit burst.");
|
||||
});
|
||||
|
||||
it("throws on a 401 auth failure so checkCredentials flags the bad key", async () => {
|
||||
await expect(
|
||||
umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
},
|
||||
{ fetch: fakeFetch({ message: "unauthorized" }, 401) },
|
||||
),
|
||||
).rejects.toThrow(/401/);
|
||||
});
|
||||
|
||||
it("throws on a 403 auth failure so checkCredentials flags the bad key", async () => {
|
||||
await expect(
|
||||
umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
},
|
||||
{ fetch: fakeFetch({ message: "forbidden" }, 403) },
|
||||
),
|
||||
).rejects.toThrow(/403/);
|
||||
});
|
||||
|
||||
it("returns null on a transient non-auth HTTP failure (500)", async () => {
|
||||
const report = await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test" },
|
||||
},
|
||||
{ fetch: fakeFetch({ message: "internal server error" }, 500) },
|
||||
);
|
||||
expect(report).toBeNull();
|
||||
});
|
||||
|
||||
it("returns null when supports() is called for a different provider or credential type", () => {
|
||||
expect(umansUsageProvider.supports?.({ provider: "zai", credential: { type: "api_key", apiKey: "x" } })).toBe(
|
||||
false,
|
||||
);
|
||||
expect(
|
||||
umansUsageProvider.supports?.({ provider: "umans", credential: { type: "oauth", accessToken: "x" } }),
|
||||
).toBe(false);
|
||||
expect(umansUsageProvider.supports?.({ provider: "umans", credential: { type: "api_key", apiKey: "x" } })).toBe(
|
||||
true,
|
||||
);
|
||||
});
|
||||
|
||||
it("includes plan display name and account identity in metadata", async () => {
|
||||
const report = await umansUsageProvider.fetchUsage(
|
||||
{
|
||||
provider: "umans",
|
||||
credential: { type: "api_key", apiKey: "sk-test", accountId: "acct-42", email: "dev@example.com" },
|
||||
},
|
||||
{ fetch: fakeFetch(umansPayload({ plan: { display_name: "Code Pro" } })) },
|
||||
);
|
||||
expect(report?.metadata?.plan).toBe("Code Pro");
|
||||
expect(report?.metadata?.accountId).toBe("acct-42");
|
||||
expect(report?.metadata?.email).toBe("dev@example.com");
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,98 @@
|
||||
import { Database } from "bun:sqlite";
|
||||
import { afterEach, describe, expect, test, vi } from "bun:test";
|
||||
import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai/auth-storage";
|
||||
import { getOAuthProviders } from "@oh-my-pi/pi-ai/registry/oauth";
|
||||
import { getEnvApiKey } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { FetchImpl } from "@oh-my-pi/pi-ai/types";
|
||||
|
||||
const originalXaiApiKey = Bun.env.XAI_API_KEY;
|
||||
|
||||
afterEach(() => {
|
||||
if (originalXaiApiKey === undefined) {
|
||||
delete Bun.env.XAI_API_KEY;
|
||||
} else {
|
||||
Bun.env.XAI_API_KEY = originalXaiApiKey;
|
||||
}
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
describe("xAI API login wiring", () => {
|
||||
test("registers xAI API in the OAuth provider selector", () => {
|
||||
const provider = getOAuthProviders().find(item => item.id === "xai");
|
||||
expect(provider).toBeDefined();
|
||||
expect(provider?.name).toBe("xAI API");
|
||||
expect(provider?.available).toBe(true);
|
||||
});
|
||||
|
||||
test("resolves XAI_API_KEY from environment", () => {
|
||||
Bun.env.XAI_API_KEY = "xai-env-key";
|
||||
expect(getEnvApiKey("xai")).toBe("xai-env-key");
|
||||
});
|
||||
|
||||
test("AuthStorage.login('xai') validates against /models and stores the pasted key", async () => {
|
||||
const fetchCalls: Array<{ url: string; init: RequestInit | undefined }> = [];
|
||||
const fetchMock: FetchImpl = vi.fn(async (input: string | URL | Request, init?: RequestInit) => {
|
||||
let url: string;
|
||||
if (typeof input === "string") {
|
||||
url = input;
|
||||
} else if (input instanceof URL) {
|
||||
url = input.toString();
|
||||
} else {
|
||||
url = input.url;
|
||||
}
|
||||
fetchCalls.push({ url, init });
|
||||
if (url === "https://api.x.ai/v1/models") {
|
||||
return new Response(JSON.stringify({ data: [{ id: "grok-4" }] }), {
|
||||
status: 200,
|
||||
headers: { "Content-Type": "application/json" },
|
||||
});
|
||||
}
|
||||
throw new Error(`unexpected fetch: ${url}`);
|
||||
});
|
||||
|
||||
const store = new SqliteAuthCredentialStore(new Database(":memory:"));
|
||||
const storage = new AuthStorage(store);
|
||||
await storage.reload();
|
||||
|
||||
await storage.login("xai", {
|
||||
onAuth: () => {},
|
||||
onPrompt: async () => "xai-validated",
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
const credential = await storage.get("xai");
|
||||
expect(credential).toEqual({ type: "api_key", key: "xai-validated", source: "login" });
|
||||
|
||||
const modelsCall = fetchCalls.find(call => call.url.endsWith("/v1/models"));
|
||||
expect(modelsCall).toBeDefined();
|
||||
const headers = new Headers(modelsCall?.init?.headers);
|
||||
expect(headers.get("Authorization")).toBe("Bearer xai-validated");
|
||||
|
||||
store.close();
|
||||
});
|
||||
|
||||
test("AuthStorage.login('xai') rejects keys that fail /models validation", async () => {
|
||||
const fetchMock: FetchImpl = vi.fn(
|
||||
async () =>
|
||||
new Response("Unauthorized", {
|
||||
status: 401,
|
||||
headers: { "Content-Type": "text/plain" },
|
||||
}),
|
||||
);
|
||||
|
||||
const store = new SqliteAuthCredentialStore(new Database(":memory:"));
|
||||
const storage = new AuthStorage(store);
|
||||
await storage.reload();
|
||||
|
||||
await expect(
|
||||
storage.login("xai", {
|
||||
onAuth: () => {},
|
||||
onPrompt: async () => "xai-bogus",
|
||||
fetch: fetchMock,
|
||||
}),
|
||||
).rejects.toThrow(/xAI API key validation failed \(401\)/);
|
||||
|
||||
expect(await storage.get("xai")).toBeUndefined();
|
||||
store.close();
|
||||
});
|
||||
});
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Regenerated the Cursor agent protobufs (`discovery/cursor-gen/agent_pb.ts`) against the modern `agent.proto`, adding the message and enum families current Cursor CLI builds emit: Pi tool exec frames, hook queries and responses, subagents, allowlist prechecks, MCP state, smart-mode classification, canvas diagnostics, conversation search, agent-store conflicts and git diff. Purely additive — no existing exported symbol changed shape.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed an issue where LM Studio first turns failed with a 400 Invalid tool_choice error when a named tool was forced, by using the supported tool_choice: "required" string selector.
|
||||
|
||||
@@ -30,7 +30,8 @@
|
||||
"test": "bun test --parallel",
|
||||
"fix": "biome check --write --unsafe .",
|
||||
"fmt": "biome format --write .",
|
||||
"gen:models": "bun scripts/generate-models.ts"
|
||||
"gen:models": "bun scripts/generate-models.ts",
|
||||
"gen:cursor-proto": "protoc --plugin=protoc-gen-es=../../node_modules/.bin/protoc-gen-es --es_out=src/discovery/cursor-gen --es_opt=target=ts -I ../ai/src/providers/cursor/proto ../ai/src/providers/cursor/proto/agent.proto"
|
||||
},
|
||||
"dependencies": {
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -7,6 +7,12 @@
|
||||
- Added server-name autocomplete for `/mcp` commands (`enable`, `disable`, `test`, `remove`, `reconnect`, `reauth`, `unauth`) using configured and runtime-discovered MCP servers.
|
||||
- Added `--from-claude` and `--from-codex` session imports, also available from `/resume @claude` and `/resume @codex`.
|
||||
- Added an opt-in OMP-native software-security workflow (`security.enabled`, default off) with immutable scan plans, exact-account Codex subscription affinity, native task-worker review, canonical findings/coverage/SARIF publication, project-scoped history, explicit dispositions, producer-differential comparison, and the read-only `security://` resource namespace. Generic SARIF and official Codex Security bundles normalize into the same OMP-owned store.
|
||||
- Added `--from-claude` and `--from-codex` session imports (including compaction state for Codex), also available from `/resume @claude` and `/resume @codex`.
|
||||
- Added interactive Exa API-key onboarding through `/login exa`, opening the official key dashboard and saving pasted keys for authenticated web search while preserving `EXA_API_KEY` and explicit-selection public MCP fallback behavior ([#1798](https://github.com/can1357/oh-my-pi/issues/1798)).
|
||||
- Added `ExtensionContext.getAsyncJobSnapshot()` so extensions can read the owning session's async-job state without relying on process-global job-manager identity
|
||||
- Added opt-in `tui.codexResetFireworks` celebrations for unscheduled Codex weekly usage resets and newly banked saved resets, shown in a theme-aware top-third modal until Escape ([#6858](https://github.com/can1357/oh-my-pi/pull/6858) by [@joshrzemien](https://github.com/joshrzemien)).
|
||||
- The Cursor exec bridge serves the seven modern Pi tool frames, mapping each to its local equivalent: `pi_read`/`pi_ls` → `read`, `pi_bash` → `bash`, `pi_edit` → `edit`, `pi_write` → `write`, `pi_grep` → `grep`, and `pi_find` → `glob`. The frames are a separate wire family from the legacy args, not aliases, so each mapping is a real translation — `pi_grep`'s `ignore_case` is the inverse of the local tool's case-sensitivity flag, `pi_find` searches filenames rather than contents, and `pi_edit`'s replacements are renamed to the local snake_case pairs.
|
||||
- `providers.autoThinkingMaxEffort` (`xhigh` | `max`, default `xhigh`) raises the ceiling of the `auto` thinking classifier. `max` became a first-class effort tier after the classifier prompt was written, so `auto` could never reach it on models that expose the tier — only the `ultrathink` keyword could. Opting in adds `max` to the classifier's vocabulary, gated on the target model actually supporting it; the default keeps today's prompt byte-for-byte. The ceiling is enforced inside the effort clamp rather than on the classifier's answer, so a sparse ladder cannot snap an excluded request back up, and the Low floor is still resolved against the model's own ladder. The on-device 3-bucket classifier stays capped at `xhigh` regardless of the setting. The ceiling governs what `auto` resolves: a ladder with nothing underneath it yields no auto level, and a `thinking.requiresEffort` model still gets its lowest supported effort from the transport.
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -15,9 +21,15 @@
|
||||
- Optimized tool guidance for bash, grep, and glob to be more concise while clarifying shell boundaries and search timeouts.
|
||||
- Optimized models configuration resource probing to run in a single child process, reducing startup contention.
|
||||
- Reserved `security://` from RPC host URI shadowing so vendor adapters cannot replace OMP's canonical security-analysis namespace.
|
||||
- Startup release notes now default to a compact change-count summary. Use `startup.changelogMode` (`summary` | `expanded` | `hidden`) to control them; legacy `collapseChangelog` choices migrate automatically ([#6771](https://github.com/can1357/oh-my-pi/issues/6771)).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed task tool blocks duplicating their per-agent progress rows into terminal scrollback on every update: live task frames now pin the transcript live region so mid-run rows are never recorded as frozen snapshots, and a detached background task freezes its progress the moment any of its rows commit to scrollback instead of mutating committed history.
|
||||
- Fixed Codex reset fireworks comparing different quota tiers or plans, preventing false celebrations when usage reports switch between Spark and base weekly limits.
|
||||
- Fixed Cursor ranged-read results losing the full file byte size after applying the requested window.
|
||||
- Fixed empty Codex final-stop recovery discarding an earlier commentary message when both messages shared response metadata.
|
||||
- Fixed Advisor availability with providers that refuse echoed reasoning by retrying once with primary thinking stripped and surfacing persistent refusals immediately.
|
||||
- Fixed `/tan` agents being unable to read parent-session `local://` attachments by correctly resolving local protocol options against the parent session's artifacts.
|
||||
- Fixed Codex web search silently returning plain completions when the hosted web search tool was skipped.
|
||||
- Fixed TUI collaboration guest loader not starting when joining or reconnecting mid-turn.
|
||||
@@ -36,6 +48,44 @@
|
||||
- Fixed file corruption and snapshot mismatches when writing files through the ACP client bridge by verifying the final on-disk content after client-side post-save formatting.
|
||||
- Fixed `omp ttsr test` silently evaluating source files as prose when their extensions were missing from the allowlist, and expanded the allowlist to support .NET, Shell, SQL, Zig, Dart, Scala, Elixir, and Protobuf files.
|
||||
- Fixed automatic light/dark theme switching in direct WezTerm sessions on macOS when DEC Mode 2031 is unsupported, and improved theme-change color responsiveness.
|
||||
- Fixed configured `retry.maxDelayMs` not being forwarded into Anthropic retry handling, so over-budget server retry delays fail fast.
|
||||
- Added tokens-per-second throughput to RPC `get_state` responses for non-TUI clients.
|
||||
- Added the RPC `set_fast_mode` command and typed TypeScript/Python client methods for live fast-mode control.
|
||||
- Added `fastModeEnabled` and `fastModeActive` to RPC `get_state` responses.
|
||||
- Fixed RPC fast-mode state reporting after direct Anthropic rejects `speed: "fast"`, while allowing explicit re-enable requests to retry priority service.
|
||||
- Added opt-in subagent access to `checkpoint`, `rewind`, `learn`, and `manage_skill` when explicitly listed in an agent definition's `tools:` frontmatter. Listing one of `checkpoint`/`rewind` auto-includes the other. Settings (`checkpoint.enabled`, `autolearn.enabled`) remain master toggles.
|
||||
- Added a `browser.cdpUrl` setting that points browser automation at an already-running CDP endpoint by default, so `app.cdp_url` no longer has to be repeated on every call. Explicit `app` options still take precedence.
|
||||
- Native compaction preserves provider-native success and non-authentication failure semantics while retaining authenticated cross-provider fallback when the native provider rejects credentials.
|
||||
- Fixed the Cursor Pi exec bridge silently dropping frame arguments. `pi_read`'s `offset`/`limit` were ignored, so a ranged read returned the whole file; `pi_grep`'s `literal` was ignored, so a fixed-string search ran as a regex and matched the wrong lines; and the path/glob join produced a `./`-prefixed spec. Ranges are now composed onto `read`'s `:N+K` inline selector, literal patterns are escaped, and the join uses `node:path`. These are `optional int32` fields, so a present `0` is honored rather than folded into a default: `pi_read` with `limit: 0` answers with empty output instead of the entire file, and `pi_find` with `limit: 0` clamps to 1 the way the reference client does.
|
||||
- `pi_grep`'s `context` and `limit` are honored. Neither is expressible in the model-facing `grep` schema — context width comes from `grep.contextBefore`/`grep.contextAfter` fixed at tool construction — so the bridge builds a per-call `grep` for frames that supply them. `GrepTool` accepts these as constructor options; the model-facing schema is unchanged, and a frame that supplies neither keeps the shared instance and the session's defaults.
|
||||
- `pi_ls`'s `limit` is still not mapped, now deliberately: it caps directory *entries*, while the local `read` tool renders a depth-2 tree with per-directory caps and elision rows and applies a selector as a *rendered line* slice. Mapping it to `:1+K` would cap a different unit while appearing honored.
|
||||
- The legacy pi shim's regex-literal escaper and path/glob join were verbatim copies of the modern bridge's. Both paths now call the shared helpers, so the two Pi translations cannot drift.
|
||||
- Fixed every Cursor `pi_edit` frame failing instead of editing. Two independent causes: the session drops `edit` from the tool registry for Cursor so the model uses full-file `write`, but that registry is also the exec bridge's tool source, so the native frame — which the server sends regardless of the advertised catalog — found no tool; and the retained instance followed the session's configured edit mode, while `PiEditExecArgs` carries `old_text`/`new_text` pairs that only `replace` accepts (the default `hashline` takes a single `input` string). The bridge now resolves a `replace`-mode instance through its fallback resolver, still wrapped for approval.
|
||||
- Fixed a `pi_grep` frame carrying `context` or `limit` escaping the approval gate. Honoring those fields needs a per-call `grep`, and the per-call instance was built raw while every registry tool is wrapped, so such calls bypassed `tools.approval.grep` and the exec-tier check for SSH-targeted paths. Both bridge callsites now build it through one shared factory that applies the same wrapper.
|
||||
- Fixed Cursor advisors ignoring `pi_grep`'s `context` and `limit`. Only the primary session supplied the per-call `grep` factory, so advisor frames silently fell back to session defaults. Advisors now receive the same factory, gated on the advisor actually having been granted `grep`.
|
||||
- Fixed Cursor advisors failing every `pi_edit`. The advisor roster handed the bridge the `edit` instance built for the advisor's own loop, which follows the configured `edit.mode` (`hashline` by default) and rejects the frame's `old_text`/`new_text` pairs — the same mode mismatch the primary bridge already fixed, on the path it missed. The exec map now substitutes a `replace`-mode instance, gated on the advisor actually having been granted `edit`, while the advisor's own loop keeps the tool it was given.
|
||||
- Fixed `pi_bash` killing commands that explicitly asked for no deadline. `timeout` is `optional int32` and `bash` documents `0` as "disables the command deadline", but a truthiness check folded a supplied `0` into unset, applying the 300s default instead. A present `0` now passes through; negatives, which have no local meaning and would otherwise clamp to the 1s floor, still fall back to the default.
|
||||
- Fixed the Cursor exec bridge granting `edit` and `grep` to sessions that withheld them. Both bridge-only tools are constructed rather than looked up, and `executeTool` prefers a constructed override over the registry, so a restricted tool set (`toolNames` without them, or `restrictToolNames`) still got a working `pi_edit`/`pi_grep` — native frames arrive regardless of the advertised catalog. Both are now gated on the session having actually granted the tool, matching the `delete` frame's existing check (issue #5680).
|
||||
- Fixed Cursor advisor bridge tools bypassing approval settings. The advisor's `pi_edit`/`pi_grep` instances are approval-wrapped, but the wrapper reads `tools.approvalMode`, per-tool `tools.approval.<tool>` policies and `autoApprove` only from the execute-time tool context — which the advisor bridge never supplied, so every native advisor frame resolved as `yolo` with empty policies and ran past a configured `ask` or `deny`. Advisors now receive the same context store as the primary bridge.
|
||||
- Fixed Cursor's `list_mcp_resources`/`read_mcp_resource` frames answering as though the client hosted no MCP servers. The bridge hardcoded an empty catalog and `not_found`, so resources from servers the session held live connections to were invisible to the model even while the same session read them through `mcp://`. Both frames now answer from the session's `MCPManager` — awaiting a server's background resource discovery rather than reading the not-yet-populated cache and reporting "advertises nothing" — and a lookup failure surfaces as an error rather than an empty catalog, which would read as "asked, none exist". A read carrying `download_path` writes the resource to that path and answers with the path alone, per the wire contract, instead of putting the payload back in the model's context. That path arrives from the server while the general-purpose resolver deliberately honors absolute paths and `..`, so downloads are confined to the workspace: the resolved target and its deepest existing ancestor must stay inside it, and a target that is itself a symlink is refused. The write then opens `O_NOFOLLOW` and refuses a non-regular or hard-linked file before truncating, so the final component cannot be swapped for a link or an inode shared outside after the check. A parent directory replaced by a symlink mid-write is still followed; closing that needs `openat`/dirfd walking, which this does not attempt.
|
||||
- Fixed the Cursor native `delete` frame bypassing approval settings. Unlike every other frame it removes the file directly instead of running a registry tool, so no approval wrapper sat in front of it — the bridge's `allowDirectFileMutation` grant answers whether a mutating tool was granted, which is a different question from whether the user's policy allows the call. A configured `tools.approval.delete: deny`, or an `always-ask` session that this channel cannot prompt in, now refuses the frame and keeps the file.
|
||||
- Fixed Cursor download-mode resource reads bypassing the session's mutation restrictions. A `read_mcp_resource` frame carrying `download_path` creates and overwrites workspace files without running a registry tool — the same hole the native `delete` frame had — so a session that withheld `write`/`edit`, or one whose `write` tier is `deny`/`always-ask`, still had files written. Both frames now share one grant (`allowDirectFileMutation`, renamed from `allowNativeDelete` now that it gates more than deletion) and one `write`-tier policy check, and the download refuses before the read so a blocked call does not fetch the resource either. The primary session derives that grant before it rewrites its registry: Cursor moves `edit` out of the tool map and `write` may be auto-registered later, so reading the map at bridge-construction time would have misjudged both.
|
||||
- Fixed `pi_ls` never reporting that a listing was clipped. The bridge read the entry cap from a flat `details.resultLimitReached`, which `glob` sets but `read` — the tool serving `pi_ls` — does not: it records the cap through `OutputMeta` at `details.meta.limits.resultLimit.reached`. Every capped listing therefore reached Cursor with `entry_limit_reached` unset, reading as complete. Both shapes are now checked, the same way the truncation translation already handles its two producers.
|
||||
- Fixed a mixed-content MCP resource read reaching Cursor mislabelled. The mime type was taken from the first content item while the payload came from whichever item supplied it, so an image blob followed by a text note sent the text as `image/png`. Each branch now reports the type of the part it actually sends.
|
||||
- Fixed `pi_read`'s `offset`/`limit` returning more lines than the frame asked for. The range is composed onto the local `read` tool's inline selector, and a plain `:N+K` deliberately pads with one leading and three trailing context lines — helpful when a human reads a snippet, wrong for a caller that named an exact range: offset 5/limit 20 handed Cursor lines 4-27. Ranged Pi reads now compose `:raw:N+K`, which slices exactly the requested lines.
|
||||
- Fixed `pi_grep` returning fewer matches than it asked for when they spread across many files. The local `grep` windows results to the first 20 files and tells the caller to paginate with `skip`, but `PiGrepExecArgs` has no `skip` field — so a frame asking for 100 matches over 25 one-match files got 20, `match_limit_reached` unset, and advice it could not act on: output silently short and labelled complete. A search carrying a total match cap now reads enough files to satisfy it (cap+1, so a result landing exactly on the cap is distinguishable from a clipped one) and reports the cap when it actually bites.
|
||||
- Fixed every native `pi_edit` failing after a session switched onto Cursor. The replace-mode `edit` instance the frame needs was built only for sessions *created* on Cursor, and the tool roster is not rebuilt on a model switch — so a session that started elsewhere kept its configured-mode `edit` in the registry, which the bridge resolves before its fallback, and the frame's `old_text`/`new_text` pairs failed validation against a `hashline` schema. The instance is now built from the `edit` grant regardless of the session's initial provider (lazily, so a session that never reaches Cursor never constructs one) and `pi_edit` asks for it explicitly through a dedicated accessor. A session that was never granted `edit` is still refused.
|
||||
- Fixed the Cursor bridge's tool resolver being able to execute an unadvertised `edit`. That resolver doubles as the agent loop's fallback for any call outside the advertised set, so serving `edit` from it meant a hallucinated call — or one naming a tool the session deselected after startup — could run a replace-mode edit the model was never offered. It is device-only again; `pi_edit` uses its own accessor.
|
||||
- Fixed the legacy Cursor `read` frame ignoring the `offset`/`limit` modern builds paginate with. Only the Pi variant composed a range, so every page of a legacy read returned the whole file (or its own truncation) and a model walking a large file never advanced past the first window. Both frames now translate a range through the same helper, and the answer sets `range_applied` to describe whether a window was actually composed.
|
||||
- Fixed the legacy Cursor `grep` frame ignoring its pagination `offset`. The local `grep` paginates by file through `skip` and advertises exactly that in its own "use skip=N" advice, so an unforwarded offset re-ran the identical search and answered page one for every page. The answer now reports the offset it applied in `offset_applied`.
|
||||
- Fixed a paginated Cursor `read` or `grep` frame being recorded as an unpaginated one. The executed call and the transcript block are built separately, so forwarding the frame's range and page fixed only the execution: the block still showed a bare path and an unskipped search, which is what a reloaded session replays and what the next turn reasons from — a slice of a file presented as the whole thing, and results from a later window presented as page one. Both are now synthesized from the same translation that runs them, including a `limit: 0` read, which is recorded as the zero lines it returns rather than a whole-file read.
|
||||
- Fixed Cursor advisors answering every MCP resource frame as though the client hosted no servers. Only the primary bridge received the `MCPManager`-backed resource adapter, so an advisor's `list_mcp_resources` reported an empty catalog and its `read_mcp_resource` a `not_found` even though the advisor shares the session's live connections. Advisors now receive the same adapter; it is not gated on a tool grant, since reading what a server advertises is a different permission from calling one of its tools.
|
||||
- Fixed advisor tools bypassing the approval gate. They are built straight from the builtin table, outside the loop that wraps every registry tool, and both the advisor's own agent loop and its Cursor exec bridge (`pi_write`, `pi_bash`) run those instances directly — so an advisor granted `write` or `bash` executed them regardless of a configured `ask` or `deny`. They now carry the same `ExtensionToolWrapper` as every other tool.
|
||||
- Added `mcp_notification` extension event and multi-listener `MCPManager.addNotificationListener` API. The runtime already received MCP server-initiated JSON-RPC notifications at the transport layer but had no path to forward them to extensions; every notification (including server-custom methods) is now delivered as `{ server, method, params }` after the manager's own list/update handling. For known list-change methods (`notifications/tools/list_changed`, `notifications/resources/list_changed`, `notifications/prompts/list_changed`) the internal refresh promise is awaited before fanout, so a listener acting on `tools/list_changed` sees fresh `getTools()`. Notifications received before any listener attaches are buffered (bounded FIFO, cap 100, drop-oldest — matches `IrcBus`'s `MAILBOX_CAP`) and drained into the first subscriber, so startup-time frames aren't lost even if the extension binds after MCP discovery. Extensions can use this to bridge push-capable MCP servers (e.g. peer messaging) into session behavior by injecting a mid-turn steer via `pi.sendMessage` / `pi.sendUserMessage`.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the dangling `MCPManager.setOnNotification` single-slot setter, which had no callers in the runtime. Replaced by `MCPManager.addNotificationListener` — multi-listener, per-listener error isolation, returns an unsubscribe function.
|
||||
|
||||
## [17.1.8] - 2026-07-28
|
||||
|
||||
|
||||
@@ -78,6 +78,8 @@ export interface AdvisorRuntimeHost {
|
||||
* recovery (credential switch, fallback chain) declined. Cleared only by
|
||||
* an explicit reset (`/new`, config rebuild, session restart). */
|
||||
notifyQuotaExhausted?(): void;
|
||||
/** Stable identity for the live advisor model. Used to restore full transcript rendering after a model switch. */
|
||||
getModelIdentity?(): string;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -230,7 +232,6 @@ const MAX_COALESCE_ROUNDS = 3;
|
||||
const MAX_QUARANTINE_RETRIES = 2;
|
||||
|
||||
const ADVISOR_RENDER_OPTIONS = {
|
||||
includeThinking: true,
|
||||
includeToolIntent: true,
|
||||
watchedRoles: true,
|
||||
expandPrimaryContext: true,
|
||||
@@ -295,6 +296,9 @@ export class AdvisorRuntime {
|
||||
#failureNotified = false;
|
||||
/** Consecutive quarantined turns since the last success/reset (issue #6661). */
|
||||
#consecutiveQuarantines = 0;
|
||||
/** Whether primary reasoning is included in advisor deltas for the current model. */
|
||||
#includeThinking = true;
|
||||
#modelIdentity: string | undefined;
|
||||
/** Completed 3-failure backlog-drop cycles since the last success/reset. */
|
||||
#droppedBacklogs = 0;
|
||||
/**
|
||||
@@ -563,13 +567,23 @@ export class AdvisorRuntime {
|
||||
this.#wakeAllWaiters();
|
||||
}
|
||||
|
||||
#syncModelIdentity(): void {
|
||||
const identity = this.host.getModelIdentity?.();
|
||||
if (identity === undefined || identity === this.#modelIdentity) return;
|
||||
this.#modelIdentity = identity;
|
||||
this.#includeThinking = true;
|
||||
}
|
||||
|
||||
#formatRawDelta(rawMessages: AgentMessage[], wip = false): string | null {
|
||||
const delta = rawMessages
|
||||
.filter(message => !(message.role === "custom" && message.customType === "advisor"))
|
||||
.map(message => this.#dedupContextMessage(message));
|
||||
if (delta.length === 0) return null;
|
||||
const obfuscator = this.host.obfuscator;
|
||||
let md = formatSessionHistoryMarkdown(delta, ADVISOR_RENDER_OPTIONS);
|
||||
let md = formatSessionHistoryMarkdown(delta, {
|
||||
...ADVISOR_RENDER_OPTIONS,
|
||||
includeThinking: this.#includeThinking,
|
||||
});
|
||||
if (!md.trim()) return null;
|
||||
if (obfuscator?.hasSecrets()) {
|
||||
let discoveredNewRegexSecretValue = false;
|
||||
@@ -603,7 +617,7 @@ export class AdvisorRuntime {
|
||||
? obfuscateAdvisorMessage(obfuscator, message, this.#advisorRegexSecretValues)
|
||||
: message,
|
||||
),
|
||||
ADVISOR_RENDER_OPTIONS,
|
||||
{ ...ADVISOR_RENDER_OPTIONS, includeThinking: this.#includeThinking },
|
||||
);
|
||||
md = obfuscator.obfuscate(md, this.#advisorRegexSecretValues);
|
||||
}
|
||||
@@ -842,7 +856,9 @@ export class AdvisorRuntime {
|
||||
if (this.#busy || this.#sessionTransitionPaused) return;
|
||||
this.#busy = true;
|
||||
try {
|
||||
this.#syncModelIdentity();
|
||||
while (!this.disposed && !this.#sessionTransitionPaused && this.#pending.length) {
|
||||
this.#syncModelIdentity();
|
||||
let popped: PendingDelta[];
|
||||
if (this.#pending[0]?.overflowRecovery) {
|
||||
const recovery = this.#pending.shift();
|
||||
@@ -944,6 +960,9 @@ export class AdvisorRuntime {
|
||||
this.#wakeAllWaiters();
|
||||
const failedMessages = this.agent.state.messages.slice(messageSnapshot);
|
||||
const terminalFailure = this.#terminalAssistantFailure(messageSnapshot);
|
||||
const classifierRefusal =
|
||||
(terminalFailure !== undefined && isClassifierRefusal(terminalFailure)) ||
|
||||
AIError.is(AIError.classify(err), AIError.Flag.ContentBlocked);
|
||||
const terminalFailureId =
|
||||
terminalFailure === undefined ? undefined : AIError.classifyMessage(terminalFailure);
|
||||
const contextOverflow =
|
||||
@@ -959,6 +978,29 @@ export class AdvisorRuntime {
|
||||
AIError.is(terminalFailureId, AIError.Flag.ContextOverflow);
|
||||
this.#rollbackFailedTurn(messageSnapshot);
|
||||
logger.debug("advisor turn failed", { err: String(err) });
|
||||
if (classifierRefusal) {
|
||||
if (this.#includeThinking) {
|
||||
this.#includeThinking = false;
|
||||
const strippedBatch = this.#formatRawDelta(rawMessages, wip);
|
||||
if (strippedBatch) {
|
||||
this.#pending.unshift({
|
||||
text: strippedBatch,
|
||||
rawMessages,
|
||||
renderRevision: this.#renderRevision,
|
||||
turns: finalTurns,
|
||||
wip,
|
||||
overflowRecovery: recoveringOverflow || undefined,
|
||||
});
|
||||
logger.debug("advisor refusal recovered by stripping primary reasoning");
|
||||
continue;
|
||||
}
|
||||
}
|
||||
this.#notifyFailureOnce(err);
|
||||
this.#clearSeenContext();
|
||||
this.#backlog = Math.max(0, this.#backlog - finalTurns);
|
||||
this.#notifyWaiters();
|
||||
continue;
|
||||
}
|
||||
let recovered = false;
|
||||
try {
|
||||
recovered =
|
||||
@@ -1112,6 +1154,14 @@ export class AdvisorRuntime {
|
||||
}
|
||||
}
|
||||
|
||||
/** Mirrors turn recovery's refusal classification and retains AIError's provider-neutral content-block fallback. */
|
||||
function isClassifierRefusal(message: AssistantMessage): boolean {
|
||||
if (message.stopReason !== "error") return false;
|
||||
const stopType = message.stopDetails?.type;
|
||||
if (stopType === "refusal" || stopType === "sensitive") return true;
|
||||
return AIError.is(AIError.classifyMessage(message), AIError.Flag.ContentBlocked);
|
||||
}
|
||||
|
||||
/**
|
||||
* The only malformed advisor turn shape: the prompt resolved but produced no
|
||||
* assistant response at all. Everything an assistant message carries — advice,
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
* {@link Effort}, clamped into the active model's supported range (never below
|
||||
* {@link Effort.Low}). Two backends, selected by `providers.autoThinkingModel`:
|
||||
*
|
||||
* - `online` (default): a smol model classifies into `low|medium|high|xhigh`.
|
||||
* - `online` (default): a smol model classifies into `low|medium|high|xhigh`,
|
||||
* plus `max` when the target model exposes that tier.
|
||||
* - a local key: an on-device memory model classifies into the coarser
|
||||
* `trivial|moderate|hard` scheme (3-class is more reliable than 4-way ordinal
|
||||
* on sub-2B models), mapped to `low|high|xhigh`.
|
||||
@@ -14,6 +15,7 @@
|
||||
* the caller falls back to a concrete level and continues the turn.
|
||||
*/
|
||||
import { type AssistantMessage, completeSimple, Effort, type Model } from "@oh-my-pi/pi-ai";
|
||||
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
|
||||
import { prompt } from "@oh-my-pi/pi-utils";
|
||||
|
||||
import type { ModelRegistry } from "../config/model-registry";
|
||||
@@ -30,7 +32,31 @@ import {
|
||||
} from "../tiny/models";
|
||||
import { tinyModelClient } from "../tiny/title-client";
|
||||
|
||||
const DIFFICULTY_SYSTEM_PROMPT = prompt.render(difficultySystemPrompt);
|
||||
/**
|
||||
* Rendered classifier prompts, keyed by whether `max` is offered as a label.
|
||||
* Two variants only, so both are memoized on first use.
|
||||
*/
|
||||
const DIFFICULTY_SYSTEM_PROMPTS: Partial<Record<"max" | "xhigh", string>> = {};
|
||||
|
||||
/**
|
||||
* Highest effort this turn's classification may resolve to: the configured
|
||||
* ceiling, further limited by what the target model actually exposes. The
|
||||
* default keeps `auto` one tier below the top, so only an explicit
|
||||
* `ultrathink` reaches {@link Effort.Max}.
|
||||
*/
|
||||
function autoEffortCeiling(deps: ClassifyDifficultyDeps): Effort {
|
||||
if (deps.settings.get("providers.autoThinkingMaxEffort") !== Effort.Max) return Effort.XHigh;
|
||||
return getSupportedEfforts(deps.model).includes(Effort.Max) ? Effort.Max : Effort.XHigh;
|
||||
}
|
||||
|
||||
function difficultySystemPromptFor(ceiling: Effort): string {
|
||||
const key = ceiling === Effort.Max ? "max" : "xhigh";
|
||||
const cached = DIFFICULTY_SYSTEM_PROMPTS[key];
|
||||
if (cached !== undefined) return cached;
|
||||
const rendered = prompt.render(difficultySystemPrompt, { allowMax: key === "max" });
|
||||
DIFFICULTY_SYSTEM_PROMPTS[key] = rendered;
|
||||
return rendered;
|
||||
}
|
||||
|
||||
/** Local classifiers occasionally need more room for chat-template boilerplate. */
|
||||
const LOCAL_ANSWER_MAX_TOKENS = 16;
|
||||
@@ -64,14 +90,18 @@ export async function classifyDifficulty(
|
||||
): Promise<Effort | undefined> {
|
||||
const backend = deps.settings.get("providers.autoThinkingModel");
|
||||
const input = preprocessTinyMessage(promptText);
|
||||
const effort =
|
||||
backend === ONLINE_AUTO_THINKING_MODEL_KEY
|
||||
? await classifyOnline(input, deps)
|
||||
: await classifyLocal(input, backend, deps);
|
||||
return clampAutoThinkingEffort(deps.model, effort);
|
||||
const online = backend === ONLINE_AUTO_THINKING_MODEL_KEY;
|
||||
// The 3-bucket local classifier cannot select `max`, so its ceiling stays at
|
||||
// XHigh whatever the setting says — otherwise a sparse ladder would snap its
|
||||
// `hard` bucket up to a tier it never chose.
|
||||
const ceiling = online ? autoEffortCeiling(deps) : Effort.XHigh;
|
||||
const effort = online ? await classifyOnline(input, deps, ceiling) : await classifyLocal(input, backend, deps);
|
||||
// The ceiling goes into the clamp itself: capping the request alone is not
|
||||
// enough, because a sparse ladder snaps an excluded request back up.
|
||||
return clampAutoThinkingEffort(deps.model, effort, ceiling);
|
||||
}
|
||||
|
||||
async function classifyOnline(input: string, deps: ClassifyDifficultyDeps): Promise<Effort> {
|
||||
async function classifyOnline(input: string, deps: ClassifyDifficultyDeps, ceiling: Effort): Promise<Effort> {
|
||||
const resolved = resolveRoleSelection(["tiny", "smol"], deps.settings, deps.registry.getAvailable());
|
||||
const model = resolved?.model;
|
||||
if (!model) {
|
||||
@@ -88,7 +118,7 @@ async function classifyOnline(input: string, deps: ClassifyDifficultyDeps): Prom
|
||||
const response = await completeSimple(
|
||||
model,
|
||||
{
|
||||
systemPrompt: [DIFFICULTY_SYSTEM_PROMPT],
|
||||
systemPrompt: [difficultySystemPromptFor(ceiling)],
|
||||
messages: [{ role: "user", content: input, timestamp: Date.now() }],
|
||||
},
|
||||
{
|
||||
@@ -134,7 +164,13 @@ async function classifyLocal(input: string, modelKey: string, deps: ClassifyDiff
|
||||
return effort;
|
||||
}
|
||||
|
||||
/** Map the online 4-way level keyword to an {@link Effort}; earliest match wins. */
|
||||
/**
|
||||
* Map an online level keyword to an {@link Effort}; earliest match wins.
|
||||
*
|
||||
* `max` is only offered to the classifier when the target model exposes that
|
||||
* tier, but it is always parsed: an unsupported `max` is snapped back down by
|
||||
* {@link clampAutoThinkingEffort} rather than failing the turn.
|
||||
*/
|
||||
export function parseDifficultyLevel(text: string): Effort | undefined {
|
||||
const lower = text.toLowerCase();
|
||||
const candidates: Array<[number, Effort]> = [];
|
||||
@@ -142,6 +178,8 @@ export function parseDifficultyLevel(text: string): Effort | undefined {
|
||||
// inside "xhigh" (no word boundary between `x` and `h`), so the two never collide.
|
||||
const xhigh = lower.search(/x[\s_-]?high/);
|
||||
if (xhigh >= 0) candidates.push([xhigh, Effort.XHigh]);
|
||||
const max = lower.search(/\bmax\b/);
|
||||
if (max >= 0) candidates.push([max, Effort.Max]);
|
||||
const high = lower.search(/\bhigh\b/);
|
||||
if (high >= 0) candidates.push([high, Effort.High]);
|
||||
const medium = lower.search(/\bmed(?:ium)?\b/);
|
||||
|
||||
@@ -927,6 +927,18 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
"tui.codexResetFireworks": {
|
||||
type: "boolean",
|
||||
default: false,
|
||||
ui: {
|
||||
tab: "appearance",
|
||||
group: "Display",
|
||||
label: "Codex Reset Fireworks",
|
||||
description:
|
||||
"Celebrate unscheduled Codex weekly usage resets and newly banked saved resets with a top-third fireworks overlay that remains until Escape",
|
||||
},
|
||||
},
|
||||
|
||||
"tui.titleState": {
|
||||
type: "boolean",
|
||||
default: true,
|
||||
@@ -1811,14 +1823,32 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
collapseChangelog: {
|
||||
type: "boolean",
|
||||
default: false,
|
||||
"startup.changelogMode": {
|
||||
type: "enum",
|
||||
values: ["summary", "expanded", "hidden"] as const,
|
||||
default: "summary",
|
||||
ui: {
|
||||
tab: "interaction",
|
||||
group: "Startup & Updates",
|
||||
label: "Collapse Changelog",
|
||||
description: "Show condensed changelog after updates",
|
||||
label: "Startup Changelog",
|
||||
description: "Choose whether update notes start as a summary, full details, or stay hidden",
|
||||
options: [
|
||||
{
|
||||
value: "summary",
|
||||
label: "Summary",
|
||||
description: "Show release and change counts with a /changelog hint",
|
||||
},
|
||||
{
|
||||
value: "expanded",
|
||||
label: "Expanded",
|
||||
description: "Show the recent release notes in full",
|
||||
},
|
||||
{
|
||||
value: "hidden",
|
||||
label: "Hidden",
|
||||
description: "Do not show release notes on startup",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
@@ -4066,6 +4096,18 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
"browser.cdpUrl": {
|
||||
type: "string",
|
||||
default: undefined,
|
||||
ui: {
|
||||
tab: "tools",
|
||||
group: "Grep & Browser",
|
||||
label: "Browser CDP URL",
|
||||
description:
|
||||
"Default HTTP CDP discovery endpoint (for example http://127.0.0.1:9222) to attach to instead of launching a browser. Explicit app.cdp_url or app.path on the tool call take precedence.",
|
||||
},
|
||||
},
|
||||
|
||||
"browser.headless": {
|
||||
type: "boolean",
|
||||
default: true,
|
||||
@@ -5076,6 +5118,23 @@ export const SETTINGS_SCHEMA = {
|
||||
options: AUTO_THINKING_MODEL_OPTIONS,
|
||||
},
|
||||
},
|
||||
"providers.autoThinkingMaxEffort": {
|
||||
type: "enum",
|
||||
values: ["xhigh", "max"] as const,
|
||||
default: "xhigh",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Thinking",
|
||||
label: "Auto Thinking Ceiling",
|
||||
description:
|
||||
"Highest effort the `auto` classifier may resolve. `xhigh` keeps the classifier one tier below the top, so only an explicit `ultrathink` reaches `max`; `max` lets a turn the classifier judges exceptional bill the top tier on models that expose it.",
|
||||
condition: "autoThinkingActive",
|
||||
options: [
|
||||
{ value: "xhigh", label: "xhigh", description: "Classifier stops at xhigh (default)" },
|
||||
{ value: "max", label: "max", description: "Classifier may resolve max where the model supports it" },
|
||||
],
|
||||
},
|
||||
},
|
||||
"features.unexpectedStopDetection": {
|
||||
type: "boolean",
|
||||
default: false,
|
||||
|
||||
@@ -1346,6 +1346,31 @@ export class Settings {
|
||||
}
|
||||
delete raw.lastChangelogVersion;
|
||||
|
||||
// collapseChangelog (boolean) -> startup.changelogMode (enum). Preserve
|
||||
// every explicit legacy choice while giving new installs the schema's
|
||||
// "summary" default: true -> summary, false -> expanded. A separately
|
||||
// configured new mode always wins.
|
||||
const startupObj = isRecord(raw.startup) ? (raw.startup as Record<string, unknown>) : undefined;
|
||||
const legacyCollapseChangelog = typeof raw.collapseChangelog === "boolean" ? raw.collapseChangelog : undefined;
|
||||
const flatChangelogMode = raw["startup.changelogMode"];
|
||||
const normalizedFlatChangelogMode =
|
||||
flatChangelogMode === "summary" || flatChangelogMode === "expanded" || flatChangelogMode === "hidden"
|
||||
? flatChangelogMode
|
||||
: undefined;
|
||||
if (legacyCollapseChangelog !== undefined || normalizedFlatChangelogMode !== undefined) {
|
||||
if (!startupObj) {
|
||||
raw.startup = {};
|
||||
}
|
||||
const target = raw.startup as Record<string, unknown>;
|
||||
if (target.changelogMode === undefined) {
|
||||
target.changelogMode =
|
||||
normalizedFlatChangelogMode ??
|
||||
(legacyCollapseChangelog !== undefined ? (legacyCollapseChangelog ? "summary" : "expanded") : undefined);
|
||||
}
|
||||
}
|
||||
delete raw.collapseChangelog;
|
||||
delete raw["startup.changelogMode"];
|
||||
|
||||
// ask.timeout: ms -> seconds (if value > 1000, it's old ms format)
|
||||
if (raw.ask && typeof (raw.ask as Record<string, unknown>).timeout === "number") {
|
||||
const oldValue = (raw.ask as Record<string, unknown>).timeout as number;
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
/**
|
||||
* Per-call tools the Cursor exec bridge needs but the model-facing registry
|
||||
* cannot supply.
|
||||
*
|
||||
* Both bridge callsites — the primary session and the advisor roster — build
|
||||
* the same instances, and both must apply the session's approval wrapper. A
|
||||
* raw tool here silently escapes the gate every registry call goes through, so
|
||||
* the construction lives in one place rather than being repeated per callsite.
|
||||
*/
|
||||
|
||||
import type { AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import { EditTool } from "./edit";
|
||||
import type { ExtensionRunner } from "./extensibility/extensions";
|
||||
import { ExtensionToolWrapper } from "./extensibility/extensions";
|
||||
import type { GrepToolOptions, Tool, ToolSession } from "./tools";
|
||||
import { GrepTool } from "./tools";
|
||||
|
||||
/**
|
||||
* Build the bridge's `createGrepTool` factory for one tool session.
|
||||
*
|
||||
* A `pi_grep` frame carries its own context width and total match cap. Neither
|
||||
* is expressible in the model-facing `grep` schema — context comes from
|
||||
* `grep.contextBefore`/`grep.contextAfter`, fixed when the shared instance is
|
||||
* constructed — so honoring them needs a fresh tool per call.
|
||||
*
|
||||
* The result is wrapped exactly like a registry tool: the approval gate runs on
|
||||
* every call site, and a per-call instance is no exception.
|
||||
*/
|
||||
export function createBridgeGrepFactory(
|
||||
session: ToolSession,
|
||||
extensionRunner: ExtensionRunner,
|
||||
): (options: GrepToolOptions) => AgentTool {
|
||||
return options => {
|
||||
const grepTool: Tool = new GrepTool(session, options);
|
||||
return new ExtensionToolWrapper(grepTool, extensionRunner);
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the `replace`-mode `edit` the bridge answers `pi_edit` with.
|
||||
*
|
||||
* `PiEditExecArgs` carries `old_text`/`new_text` pairs, which is exactly
|
||||
* `replace`'s schema and nothing else's. The session's own instance follows the
|
||||
* configured `edit.mode` — `hashline` by default, whose schema is a single
|
||||
* `input` string — so a frame handed that instance fails validation instead of
|
||||
* editing the file.
|
||||
*
|
||||
* Callers MUST gate this on the session having actually granted `edit`: the
|
||||
* tool is constructed rather than looked up, so building one unconditionally
|
||||
* hands a restricted agent a mutating tool it was denied (issue #5680).
|
||||
*/
|
||||
export function createBridgeEditTool(session: ToolSession, extensionRunner: ExtensionRunner): AgentTool {
|
||||
const editTool: Tool = new EditTool(session, "replace");
|
||||
return new ExtensionToolWrapper(editTool, extensionRunner);
|
||||
}
|
||||
|
||||
/**
|
||||
* The tool map the exec bridge should run, given the map a caller granted.
|
||||
*
|
||||
* `pi_edit` needs a `replace`-mode instance, but only when `edit` was granted:
|
||||
* the tool is constructed rather than looked up, so substituting
|
||||
* unconditionally would hand a restricted roster a mutating tool it was denied
|
||||
* (issue #5680). The granted map is never mutated: an unsubstituted result is a
|
||||
* copy, so a caller without an `edit` grant cannot accidentally gain one.
|
||||
*
|
||||
* The advisor roster passes its granted map here; the primary session keeps its
|
||||
* instance out of the registry entirely (Cursor does not advertise `edit`) and
|
||||
* serves it through the bridge's `getEditReplaceTool` accessor instead — not
|
||||
* the `getTool` fallback, which doubles as the agent loop's resolver for
|
||||
* unadvertised calls and must stay device-only.
|
||||
*/
|
||||
export function bridgeToolMap(
|
||||
granted: ReadonlyMap<string, AgentTool>,
|
||||
createEditTool: (() => AgentTool | undefined) | undefined,
|
||||
): Map<string, AgentTool> {
|
||||
const bridged = new Map(granted);
|
||||
if (!granted.has("edit") || !createEditTool) return bridged;
|
||||
const bridgeEdit = createEditTool();
|
||||
if (bridgeEdit) bridged.set("edit", bridgeEdit);
|
||||
return bridged;
|
||||
}
|
||||
@@ -1,5 +1,6 @@
|
||||
import { randomUUID } from "node:crypto";
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import type {
|
||||
AgentEvent,
|
||||
AgentTool,
|
||||
@@ -9,35 +10,91 @@ import type {
|
||||
} from "@oh-my-pi/pi-agent-core";
|
||||
import type {
|
||||
CursorMcpCall,
|
||||
CursorMcpResource,
|
||||
CursorMcpResourceContent,
|
||||
CursorShellStreamCallbacks,
|
||||
CursorTodoSnapshot,
|
||||
CursorExecHandlers as ICursorExecHandlers,
|
||||
ToolResultMessage,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
piEscapeRegexLiteral,
|
||||
piGrepSkip,
|
||||
piJoinPath,
|
||||
piLimit,
|
||||
piLsPath,
|
||||
piReadPath,
|
||||
piTimeout,
|
||||
} from "@oh-my-pi/pi-ai/providers/cursor/exec-modern";
|
||||
import { sanitizeText } from "@oh-my-pi/pi-utils";
|
||||
import { resolveToCwd } from "./tools/path-utils";
|
||||
import type { MCPResourceReadResult } from "./mcp/types";
|
||||
import type { ApprovalMode } from "./tools/approval";
|
||||
import { resolveApproval } from "./tools/approval";
|
||||
import { confineToWorkspace, resolveToCwd } from "./tools/path-utils";
|
||||
import type { TodoPhase, TodoStatus } from "./tools/todo";
|
||||
|
||||
/** Phase used for Cursor-owned tasks with no local phase grouping. */
|
||||
const CURSOR_TODO_PHASE = "Tasks";
|
||||
|
||||
/**
|
||||
* A tool instance the bridge can run, matching the erased shape the session's
|
||||
* tool registry stores. The concrete tools have narrower `execute` parameter
|
||||
* types than the default `AgentTool`, which only unify through this alias.
|
||||
*/
|
||||
type CursorBridgeTool = AgentTool<any, any, any>;
|
||||
|
||||
/**
|
||||
* The live MCP connections Cursor's resource frames are answered from.
|
||||
*
|
||||
* Named so every construction site can hand over the same adapter; a session
|
||||
* and its advisors share one set of connections.
|
||||
*/
|
||||
export interface CursorMcpResourceAdapter {
|
||||
serverNames(): string[];
|
||||
getServerResources(
|
||||
name: string,
|
||||
): Promise<{ resources: { uri: string; name?: string; description?: string; mimeType?: string }[] } | undefined>;
|
||||
readServerResource(name: string, uri: string): Promise<MCPResourceReadResult | undefined>;
|
||||
}
|
||||
|
||||
interface CursorExecBridgeOptions {
|
||||
cwd: string;
|
||||
getCwd?: () => string;
|
||||
tools: Map<string, AgentTool>;
|
||||
/** Resolves execution overrides (mounted-device permission wrappers) before the canonical map. */
|
||||
getExecutableTool?: (name: string) => AgentTool | undefined;
|
||||
/**
|
||||
* The `replace`-mode `edit` instance `pi_edit` must run, when the session
|
||||
* granted `edit` at all.
|
||||
*
|
||||
* `PiEditExecArgs` is that mode's schema verbatim, and the session's own
|
||||
* `edit` may be in any mode — `hashline` by default — whose schema rejects
|
||||
* `old_text`/`new_text` outright. {@link tools} therefore cannot be trusted
|
||||
* for this one frame: a session that starts on another provider keeps its
|
||||
* configured instance in the map (only Cursor sessions move `edit` out), and
|
||||
* switching to Cursor later does not rebuild the roster.
|
||||
*/
|
||||
getEditReplaceTool?: () => CursorBridgeTool | undefined;
|
||||
getToolContext?: () => AgentToolContext | undefined;
|
||||
emitEvent?: (event: AgentEvent) => void;
|
||||
/**
|
||||
* Whether the Cursor native `delete` frame may remove files. Unlike every
|
||||
* other exec handler, `executeDelete` mutates the filesystem directly instead
|
||||
* of consulting {@link tools}, so a background read-only advisor could delete
|
||||
* workspace files it was never granted a mutating tool for (issue #5680
|
||||
* review). Defaults to allowed to preserve the primary agent's behavior;
|
||||
* callers with a restricted tool set (advisors) opt out.
|
||||
* Whether frames that mutate the filesystem WITHOUT running a registry tool
|
||||
* may do so: the native `delete` frame, and a `read_mcp_resource` carrying
|
||||
* `download_path`. Both write or remove workspace files directly instead of
|
||||
* consulting {@link tools}, so a background read-only advisor could touch
|
||||
* files it was never granted a mutating tool for (issue #5680 review).
|
||||
*
|
||||
* This is a grant, not a policy: it answers "did the session hand this
|
||||
* channel a file-writing tool", which callers derive from their own roster
|
||||
* before any bridge-specific rewriting. The primary Cursor session moves
|
||||
* `edit` out of {@link tools} and serves it through
|
||||
* {@link getEditReplaceTool}, so reading the map here would deny an
|
||||
* edit-only session. Defaults to allowed
|
||||
* to preserve the primary agent's behavior; callers with a restricted tool
|
||||
* set (advisors) opt out. The user's approval policy is resolved separately,
|
||||
* per call.
|
||||
*/
|
||||
allowNativeDelete?: boolean;
|
||||
allowDirectFileMutation?: boolean;
|
||||
/**
|
||||
* Mirror Cursor's server-owned todo list into local session state. Cursor
|
||||
* resolves `update_todos` / `read_todos` remotely, so without this bridge
|
||||
@@ -50,6 +107,99 @@ interface CursorExecBridgeOptions {
|
||||
* Cursor emits no local `todo` toolResult, so nothing else records it.
|
||||
*/
|
||||
persistTodoPhases?: (phases: TodoPhase[]) => void;
|
||||
/**
|
||||
* Build a `grep` tool honoring a frame's own context width and match cap.
|
||||
*
|
||||
* The modern `pi_grep` frame carries both, and the shared `grep` instance
|
||||
* is fixed to the session settings at construction — so without this the
|
||||
* two fields are silently dropped. Callers that cannot supply it keep the
|
||||
* shared instance and the session's defaults.
|
||||
*
|
||||
* The returned tool is executed as-is. Callers whose registry tools carry an
|
||||
* approval wrapper MUST apply the same wrapper here, or a frame supplying
|
||||
* either field silently escapes the approval gate that every other call
|
||||
* goes through.
|
||||
*/
|
||||
createGrepTool?(options: { context?: number; totalMatchLimit?: number }): CursorBridgeTool | undefined;
|
||||
/**
|
||||
* The session's live MCP connections, for Cursor's resource frames.
|
||||
*
|
||||
* `list_mcp_resources` / `read_mcp_resource` ask what this client's servers
|
||||
* advertise. Without this the bridge answers an empty catalog and
|
||||
* `not_found`, hiding resources the session is in fact connected to.
|
||||
*
|
||||
* `getServerResources` is async because a server's catalog loads in the
|
||||
* background after its tools register: a frame arriving in that window would
|
||||
* otherwise read the not-yet-populated cache and report an empty catalog,
|
||||
* which is indistinguishable from a server that advertises nothing.
|
||||
*/
|
||||
mcpResources?: CursorMcpResourceAdapter;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write a downloaded resource without following a link at the target.
|
||||
*
|
||||
* The containment check and the write are separate syscalls, so a link planted
|
||||
* at the target in between would redirect the bytes — the check cannot close
|
||||
* that window on its own. `O_NOFOLLOW` decides it atomically for symlinks.
|
||||
*
|
||||
* A hard link needs a second check: it is a regular file that passes both the
|
||||
* containment check and `O_NOFOLLOW` while sharing its inode with a file
|
||||
* anywhere else on the volume, so truncating it overwrites that file too.
|
||||
* `nlink > 1` on the OPEN handle is the test — statting the path first would
|
||||
* reintroduce the race the open just closed. Same reasoning, and the same
|
||||
* refusal, as `autolearn/managed-skills.ts`.
|
||||
*
|
||||
* Truncation therefore happens after that check rather than through `O_TRUNC`,
|
||||
* which would have already destroyed the contents by the time it ran.
|
||||
*
|
||||
* Scope: `O_NOFOLLOW` applies to the FINAL component only. A parent directory
|
||||
* swapped for an outward symlink between the check and this open is still
|
||||
* followed; refusing that needs an `openat`/dirfd walk of every segment, which
|
||||
* this does not attempt — an attacker who can rewrite the workspace tree
|
||||
* mid-download is already inside the boundary this guard defends.
|
||||
*
|
||||
* Parent directories are created first, since the frame may name a path whose
|
||||
* directories do not exist yet.
|
||||
*
|
||||
* `O_NONBLOCK` is what keeps the guard below reachable. A write-only open of a
|
||||
* FIFO blocks until a reader arrives, so a `download_path` naming one would
|
||||
* hang the turn forever WITHOUT the non-regular check ever running — the open
|
||||
* itself never returns. Non-blocking turns that into `ENXIO` when no reader is
|
||||
* attached, and hands back a descriptor the `isFile()` check refuses when one
|
||||
* is. The flag has no effect on regular files, which is every legitimate
|
||||
* target.
|
||||
*/
|
||||
async function writeWithoutFollowingLinks(absolutePath: string, payload: string | Buffer): Promise<void> {
|
||||
await fs.promises.mkdir(path.dirname(absolutePath), { recursive: true });
|
||||
const handle = await fs.promises
|
||||
.open(
|
||||
absolutePath,
|
||||
fs.constants.O_WRONLY | fs.constants.O_CREAT | fs.constants.O_NOFOLLOW | fs.constants.O_NONBLOCK,
|
||||
)
|
||||
.catch((error: NodeJS.ErrnoException) => {
|
||||
// A readerless FIFO. Reported as the refusal it is, rather than the
|
||||
// bare "no such device or address" the errno spells out.
|
||||
if (error.code === "ENXIO") {
|
||||
throw new Error(`Refusing to download onto a special file: ${absolutePath}`);
|
||||
}
|
||||
throw error;
|
||||
});
|
||||
try {
|
||||
const stat = await handle.stat();
|
||||
if (!stat.isFile()) {
|
||||
throw new Error(`Refusing to download onto a non-regular file: ${absolutePath}`);
|
||||
}
|
||||
if (stat.nlink > 1) {
|
||||
throw new Error(
|
||||
`Refusing to download onto a file with ${stat.nlink} hard links, which would overwrite its other names: ${absolutePath}`,
|
||||
);
|
||||
}
|
||||
await handle.truncate(0);
|
||||
await handle.writeFile(payload);
|
||||
} finally {
|
||||
await handle.close();
|
||||
}
|
||||
}
|
||||
|
||||
function createToolResultMessage(
|
||||
@@ -81,8 +231,9 @@ async function executeTool(
|
||||
toolName: string,
|
||||
toolCallId: string,
|
||||
args: Record<string, unknown>,
|
||||
overrideTool?: CursorBridgeTool,
|
||||
): Promise<ToolResultMessage> {
|
||||
const tool = options.getExecutableTool?.(toolName) ?? options.tools.get(toolName);
|
||||
const tool = overrideTool ?? options.getExecutableTool?.(toolName) ?? options.tools.get(toolName);
|
||||
if (!tool) {
|
||||
const result = buildToolErrorResult(`Tool "${toolName}" not available`);
|
||||
return createToolResultMessage(toolCallId, toolName, result, true);
|
||||
@@ -133,14 +284,49 @@ async function executeTool(
|
||||
return createToolResultMessage(toolCallId, toolName, result, isError);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the user's policy for a frame that mutates the filesystem directly.
|
||||
*
|
||||
* The native `delete` and `read_mcp_resource` download frames both bypass the
|
||||
* registry, so no approval wrapper sits in front of them. `write` is the tier
|
||||
* a file creation or removal belongs to. Returns `null` when the call may
|
||||
* proceed, or the refusal text to answer with.
|
||||
*/
|
||||
function refuseByWritePolicy(options: CursorExecBridgeOptions, toolName: string, pathArg: string): string | null {
|
||||
const context = options.getToolContext?.();
|
||||
const settings = context?.settings;
|
||||
const approvalMode: ApprovalMode =
|
||||
context?.autoApprove === true ? "yolo" : (settings?.get("tools.approvalMode") ?? "yolo");
|
||||
const approval = resolveApproval(
|
||||
{ name: toolName, approval: "write" },
|
||||
{ path: pathArg },
|
||||
approvalMode,
|
||||
(settings?.get("tools.approval") ?? {}) as Record<string, unknown>,
|
||||
);
|
||||
if (approval.policy === "allow") return null;
|
||||
return approval.policy === "deny"
|
||||
? `Tool "${toolName}" is blocked by user policy.`
|
||||
: `Tool "${toolName}" requires approval, which this channel cannot request.`;
|
||||
}
|
||||
|
||||
async function executeDelete(options: CursorExecBridgeOptions, pathArg: string, toolCallId: string) {
|
||||
const toolName = "delete";
|
||||
|
||||
if (options.allowNativeDelete === false) {
|
||||
if (options.allowDirectFileMutation === false) {
|
||||
const result = buildToolErrorResult(`Tool "${toolName}" not available`);
|
||||
return createToolResultMessage(toolCallId, toolName, result, true);
|
||||
}
|
||||
|
||||
// Unlike every other frame, this one mutates the filesystem directly instead
|
||||
// of running a registry tool, so no approval wrapper sits in front of it.
|
||||
// `allowDirectFileMutation` answers "was a mutating tool granted", which is a
|
||||
// different question from "does the user's policy allow this call" — without
|
||||
// this, a configured `deny` or an `always-ask` session still lost the file.
|
||||
const refusal = refuseByWritePolicy(options, toolName, pathArg);
|
||||
if (refusal) {
|
||||
return createToolResultMessage(toolCallId, toolName, buildToolErrorResult(refusal), true);
|
||||
}
|
||||
|
||||
options.emitEvent?.({ type: "tool_execution_start", toolCallId, toolName, args: { path: pathArg } });
|
||||
|
||||
const absolutePath = resolveToCwd(pathArg, options.getCwd?.() ?? options.cwd);
|
||||
@@ -241,10 +427,21 @@ function buildTodoSyncResult(
|
||||
export class CursorExecHandlers implements ICursorExecHandlers {
|
||||
constructor(private options: CursorExecBridgeOptions) {}
|
||||
|
||||
/**
|
||||
* Modern Cursor builds paginate the legacy `read` frame with
|
||||
* `offset`/`limit`, exactly as `pi_read` does. Dropping them returns the
|
||||
* whole file (or its own truncation) for every page, so a model walking a
|
||||
* large file never advances. Composed with the same helper, so both frames
|
||||
* translate a range identically.
|
||||
*/
|
||||
async read(args: Parameters<NonNullable<ICursorExecHandlers["read"]>>[0]) {
|
||||
const toolCallId = decodeToolCallId(args.toolCallId);
|
||||
const toolResultMessage = await executeTool(this.options, "read", toolCallId, { path: args.path });
|
||||
return toolResultMessage;
|
||||
const composed = piReadPath(args.path, args.offset, args.limit);
|
||||
// A present `limit: 0` asks for zero lines; no selector expresses that.
|
||||
if (composed === null) {
|
||||
return createToolResultMessage(toolCallId, "read", { content: [{ type: "text", text: "" }] }, false);
|
||||
}
|
||||
return await executeTool(this.options, "read", toolCallId, { path: composed });
|
||||
}
|
||||
|
||||
async ls(args: Parameters<NonNullable<ICursorExecHandlers["ls"]>>[0]) {
|
||||
@@ -254,6 +451,12 @@ export class CursorExecHandlers implements ICursorExecHandlers {
|
||||
return toolResultMessage;
|
||||
}
|
||||
|
||||
/**
|
||||
* Modern Cursor builds paginate this frame with `offset`. The local `grep`
|
||||
* paginates by file through `skip`, which is the same unit its own
|
||||
* "use skip=N for the next page" advice counts in — so an unforwarded
|
||||
* offset re-runs the identical search and returns page one forever.
|
||||
*/
|
||||
async grep(args: Parameters<NonNullable<ICursorExecHandlers["grep"]>>[0]) {
|
||||
const toolCallId = decodeToolCallId(args.toolCallId);
|
||||
const searchPath = args.glob ? `${args.path || "."}/${args.glob}` : args.path || ".";
|
||||
@@ -261,6 +464,7 @@ export class CursorExecHandlers implements ICursorExecHandlers {
|
||||
pattern: args.pattern,
|
||||
path: searchPath,
|
||||
case: args.caseInsensitive === true ? false : undefined,
|
||||
skip: piGrepSkip(args.offset),
|
||||
});
|
||||
return toolResultMessage;
|
||||
}
|
||||
@@ -401,6 +605,227 @@ export class CursorExecHandlers implements ICursorExecHandlers {
|
||||
});
|
||||
return toolResultMessage;
|
||||
}
|
||||
/**
|
||||
* Modern Cursor CLI Pi tool frames (`ExecServerMessage` 45-51).
|
||||
*
|
||||
* These are a separate frame family from the legacy `read`/`shell`/... set,
|
||||
* not aliases: different args, different result oneofs, and no `tool_call_id`
|
||||
* (the provider mints one and passes it in `call.toolCallId`). Each maps onto
|
||||
* the local tool with matching semantics, so the same approval, sandboxing
|
||||
* and event plumbing applies as for a model-issued call.
|
||||
*/
|
||||
/**
|
||||
* `offset`/`limit` are a 1-indexed start line plus a line count (verified
|
||||
* against the reference `LocalPiReadExecutor`), which is exactly the local
|
||||
* `read` tool's `:N+K` inline selector — the tool takes no range kwargs, so
|
||||
* the range has to be composed onto the path or ranged reads silently
|
||||
* return the whole file.
|
||||
*/
|
||||
async piRead(call: Parameters<NonNullable<ICursorExecHandlers["piRead"]>>[0]) {
|
||||
const { path: readPath, offset, limit } = call.args;
|
||||
const composed = piReadPath(readPath, offset, limit);
|
||||
// A present `limit: 0` asks for zero lines. The reference slices an empty
|
||||
// string for it; no `read` selector expresses that, so answer directly
|
||||
// rather than falling back to a whole-file read.
|
||||
if (composed === null) {
|
||||
return createToolResultMessage(call.toolCallId, "read", { content: [{ type: "text", text: "" }] }, false);
|
||||
}
|
||||
return await executeTool(this.options, "read", call.toolCallId, { path: composed });
|
||||
}
|
||||
|
||||
async piBash(call: Parameters<NonNullable<ICursorExecHandlers["piBash"]>>[0]) {
|
||||
return await executeTool(this.options, "bash", call.toolCallId, {
|
||||
command: call.args.command,
|
||||
timeout: piTimeout(call.args.timeout),
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* `PiEditExecArgs` is the local `edit` tool's replace mode verbatim: a path
|
||||
* plus `old_text`/`new_text` pairs. The tool's schema is snake_case, so the
|
||||
* proto's camelCase accessors are mapped back on the way in.
|
||||
*
|
||||
* The replace-mode instance is requested explicitly rather than resolved
|
||||
* from {@link CursorExecBridgeOptions.tools}: the registry's `edit` is in
|
||||
* the session's configured mode, whose schema rejects these arguments.
|
||||
*/
|
||||
async piEdit(call: Parameters<NonNullable<ICursorExecHandlers["piEdit"]>>[0]) {
|
||||
return await executeTool(
|
||||
this.options,
|
||||
"edit",
|
||||
call.toolCallId,
|
||||
{
|
||||
path: call.args.path,
|
||||
edits: call.args.edits.map(edit => ({ old_text: edit.oldText, new_text: edit.newText })),
|
||||
},
|
||||
this.options.getEditReplaceTool?.(),
|
||||
);
|
||||
}
|
||||
|
||||
async piWrite(call: Parameters<NonNullable<ICursorExecHandlers["piWrite"]>>[0]) {
|
||||
return await executeTool(this.options, "write", call.toolCallId, {
|
||||
path: call.args.path,
|
||||
content: call.args.content,
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* `literal` makes the pattern a fixed string; the local tool is regex-only,
|
||||
* so the pattern is escaped on the way in (same translation the legacy pi
|
||||
* shim does).
|
||||
*
|
||||
* `context` and `limit` are not expressible in the model-facing schema —
|
||||
* context width comes from settings fixed at tool construction — so the
|
||||
* frame's values are honored by building a per-call `grep` through
|
||||
* {@link CursorExecBridgeOptions.createGrepTool}. Both are `optional int32`,
|
||||
* so a present `0` context means "no context lines", not "use the default".
|
||||
* Without the factory the shared instance runs with session defaults.
|
||||
*/
|
||||
async piGrep(call: Parameters<NonNullable<ICursorExecHandlers["piGrep"]>>[0]) {
|
||||
const { pattern, path, glob, ignoreCase, literal, context, limit } = call.args;
|
||||
const scoped =
|
||||
context !== undefined || limit !== undefined
|
||||
? this.options.createGrepTool?.({ context, totalMatchLimit: piLimit(limit) })
|
||||
: undefined;
|
||||
// Same arg mapping as the legacy `grep` handler: the local tool takes one
|
||||
// path spec, and its `case` flag is case-SENSITIVITY, the inverse of the
|
||||
// frame's `ignore_case`.
|
||||
return await executeTool(
|
||||
this.options,
|
||||
"grep",
|
||||
call.toolCallId,
|
||||
{
|
||||
pattern: literal === true ? piEscapeRegexLiteral(pattern) : pattern,
|
||||
path: glob ? piJoinPath(path, glob) : path || ".",
|
||||
case: ignoreCase === true ? false : undefined,
|
||||
},
|
||||
scoped,
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* `pi_find` is a filename search, which is the local `glob` tool — not
|
||||
* `grep`. Its `pattern` is a glob, joined onto `path` because `glob` takes a
|
||||
* single combined path spec.
|
||||
*
|
||||
* `limit` is `optional int32`, so `0` is present rather than unset; the
|
||||
* reference clamps it with `Math.max(1, limit ?? 1000)`, and an unset limit
|
||||
* leaves the local tool's own default in place.
|
||||
*/
|
||||
async piFind(call: Parameters<NonNullable<ICursorExecHandlers["piFind"]>>[0]) {
|
||||
const { pattern, path, limit } = call.args;
|
||||
return await executeTool(this.options, "glob", call.toolCallId, {
|
||||
path: piJoinPath(path, pattern),
|
||||
limit: piLimit(limit),
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Redirected to `read`, which lists directories — same as the legacy `ls`.
|
||||
* The frame's entry `limit` is not mapped; see {@link piLsPath}.
|
||||
*/
|
||||
async piLs(call: Parameters<NonNullable<ICursorExecHandlers["piLs"]>>[0]) {
|
||||
return await executeTool(this.options, "read", call.toolCallId, { path: piLsPath(call.args.path) });
|
||||
}
|
||||
|
||||
/**
|
||||
* The resources this client's MCP servers advertise.
|
||||
*
|
||||
* Cursor addresses a later read by `server`, so every entry carries the name
|
||||
* it came from. An absent `server` filter means "all of them".
|
||||
*/
|
||||
async listMcpResources({ server }: { server?: string }): Promise<CursorMcpResource[]> {
|
||||
const mcp = this.options.mcpResources;
|
||||
if (!mcp) return [];
|
||||
const names = server ? [server] : mcp.serverNames();
|
||||
// Concurrently: each name may block on that server's first catalog load,
|
||||
// and a slow server should not delay the rest of the listing.
|
||||
const catalogs = await Promise.all(names.map(async name => [name, await mcp.getServerResources(name)] as const));
|
||||
const listed: CursorMcpResource[] = [];
|
||||
for (const [name, catalog] of catalogs) {
|
||||
for (const resource of catalog?.resources ?? []) {
|
||||
listed.push({
|
||||
uri: resource.uri,
|
||||
name: resource.name,
|
||||
description: resource.description,
|
||||
mimeType: resource.mimeType,
|
||||
server: name,
|
||||
});
|
||||
}
|
||||
}
|
||||
return listed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Read one resource, or `null` when the server or uri is unknown.
|
||||
*
|
||||
* MCP returns a list of content items; the wire carries exactly one text or
|
||||
* blob. Text items are joined, since a multi-part text resource is one
|
||||
* document; otherwise the first blob stands in. `blob` arrives base64 and
|
||||
* the wire wants bytes.
|
||||
*
|
||||
* A `downloadPath` frame is a different contract: write the bytes to that
|
||||
* workspace-relative path and answer with the path alone, so a large binary
|
||||
* lands on disk instead of in the model's context. That makes it a workspace
|
||||
* mutation reached without a registry tool, so it is gated exactly like the
|
||||
* native `delete` frame — on the session actually granting a file-writing
|
||||
* tool, and on the user's `write`-tier policy. The gate runs before the read
|
||||
* so a refused download never fetches the resource either.
|
||||
*/
|
||||
async readMcpResource({
|
||||
server,
|
||||
uri,
|
||||
downloadPath,
|
||||
}: {
|
||||
server: string;
|
||||
uri: string;
|
||||
downloadPath?: string;
|
||||
}): Promise<CursorMcpResourceContent | null> {
|
||||
if (downloadPath) {
|
||||
if (this.options.allowDirectFileMutation === false) {
|
||||
throw new Error('Tool "write" not available: this session cannot download resources to disk.');
|
||||
}
|
||||
const refusal = refuseByWritePolicy(this.options, "write", downloadPath);
|
||||
if (refusal) throw new Error(refusal);
|
||||
}
|
||||
const mcp = this.options.mcpResources;
|
||||
if (!mcp) return null;
|
||||
const read = await mcp.readServerResource(server, uri);
|
||||
if (!read) return null;
|
||||
// The mime type must describe the bytes actually sent, not whatever item
|
||||
// happened to be first: an image blob followed by a text note would
|
||||
// otherwise label the text `image/png` and mislead the model about what
|
||||
// it is holding. Each branch below takes the type from its own producer.
|
||||
const textItems = read.contents.filter(item => item.text !== undefined);
|
||||
const texts = textItems.map(item => item.text as string);
|
||||
const blobItem = read.contents.find(item => item.blob !== undefined);
|
||||
const blob = blobItem?.blob;
|
||||
const textMimeType = textItems[0]?.mimeType;
|
||||
const blobMimeType = blobItem?.mimeType;
|
||||
|
||||
if (downloadPath) {
|
||||
// Text resources download as their own bytes; a blob decodes first.
|
||||
const payload =
|
||||
texts.length > 0 ? texts.join("\n") : blob !== undefined ? Buffer.from(blob, "base64") : undefined;
|
||||
if (payload === undefined) return null;
|
||||
// The path is workspace-relative BY CONTRACT, but it arrives from the
|
||||
// server, and `resolveToCwd` deliberately honors absolute paths and
|
||||
// `..` for user-authored tool input. Taking it at its word would let a
|
||||
// frame write anywhere this process can reach, so confine it here
|
||||
// rather than trusting the declaration.
|
||||
const cwd = this.options.getCwd?.() ?? this.options.cwd;
|
||||
const absolutePath = confineToWorkspace(downloadPath, cwd);
|
||||
if (!absolutePath) throw new Error(`Refusing to download outside the workspace: ${downloadPath}`);
|
||||
await writeWithoutFollowingLinks(absolutePath, payload);
|
||||
// The path echoed back is the one the frame asked for; the model
|
||||
// addresses it the same relative way.
|
||||
return { uri, mimeType: texts.length > 0 ? textMimeType : blobMimeType, downloadPath };
|
||||
}
|
||||
|
||||
if (texts.length > 0) return { uri, mimeType: textMimeType, text: texts.join("\n") };
|
||||
if (blob === undefined) return null;
|
||||
return { uri, mimeType: blobMimeType, blob: Buffer.from(blob, "base64") };
|
||||
}
|
||||
|
||||
/**
|
||||
* Settle a completed native Cursor todo call, mirroring its list when the
|
||||
@@ -510,4 +935,30 @@ export class CursorExecHandlers implements ICursorExecHandlers {
|
||||
const toolResultMessage = await executeTool(this.options, toolName, toolCallId, args);
|
||||
return toolResultMessage;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve an MCP call's approval without running it.
|
||||
*
|
||||
* Same resolution the wrapper applies at execution time, minus the
|
||||
* execution: an unknown tool is not approvable, and a `prompt` is not an
|
||||
* approval — the frame has no way to carry an interactive question, and the
|
||||
* user is asked for real when the call itself arrives.
|
||||
*/
|
||||
async mcpApprovalPreflight(call: CursorMcpCall) {
|
||||
const toolName = call.toolName || call.name;
|
||||
const tool = this.options.getExecutableTool?.(toolName) ?? this.options.tools.get(toolName);
|
||||
if (!tool) return false;
|
||||
const context = this.options.getToolContext?.();
|
||||
const settings = context?.settings;
|
||||
const approvalMode: ApprovalMode =
|
||||
context?.autoApprove === true ? "yolo" : (settings?.get("tools.approvalMode") ?? "yolo");
|
||||
const args = Object.keys(call.args ?? {}).length > 0 ? call.args : decodeMcpArgs(call.rawArgs ?? {});
|
||||
const approval = resolveApproval(
|
||||
tool,
|
||||
args,
|
||||
approvalMode,
|
||||
(settings?.get("tools.approval") ?? {}) as Record<string, unknown>,
|
||||
);
|
||||
return approval.policy === "allow";
|
||||
}
|
||||
}
|
||||
|
||||
@@ -399,14 +399,24 @@ export class EditTool implements AgentTool<TInput> {
|
||||
readonly #editMode?: EditMode;
|
||||
readonly #deferredDiagnostics: DeferredDiagnostics;
|
||||
|
||||
constructor(private readonly session: ToolSession) {
|
||||
/**
|
||||
* `mode` pins the edit variant for this instance, for callers whose protocol
|
||||
* fixes the shape of an edit. The Cursor `pi_edit` frame carries
|
||||
* `old_text`/`new_text` pairs, which only `replace` accepts — under the
|
||||
* default `hashline` mode those args do not match the schema at all. Left
|
||||
* unset, the env/settings resolution applies as before.
|
||||
*/
|
||||
constructor(
|
||||
private readonly session: ToolSession,
|
||||
mode?: EditMode,
|
||||
) {
|
||||
const {
|
||||
PI_EDIT_FUZZY: editFuzzy = "auto",
|
||||
PI_EDIT_FUZZY_THRESHOLD: editFuzzyThreshold = "auto",
|
||||
PI_EDIT_VARIANT: envEditVariant = "auto",
|
||||
} = Bun.env;
|
||||
|
||||
this.#editMode = resolveConfiguredEditMode(envEditVariant);
|
||||
this.#editMode = mode ?? resolveConfiguredEditMode(envEditVariant);
|
||||
this.#allowFuzzy = resolveAllowFuzzy(session, editFuzzy);
|
||||
this.#fuzzyThreshold = resolveFuzzyThreshold(session, editFuzzyThreshold);
|
||||
const deduplicateDiagnostics =
|
||||
|
||||
@@ -10,6 +10,7 @@ import type { Settings } from "../../config/settings";
|
||||
import type { LocalProtocolOptions } from "../../internal-urls/local-protocol";
|
||||
import type { MemoryRuntimeContext } from "../../memory-backend";
|
||||
import { type Theme, theme } from "../../modes/theme/theme";
|
||||
import type { AsyncJobSnapshot } from "../../session/agent-session";
|
||||
import type { SessionManager } from "../../session/session-manager";
|
||||
import type { BranchHandler, NavigateTreeHandler, NewSessionHandler } from "../session-handler-types";
|
||||
import { ManagedTimers } from "./managed-timers";
|
||||
@@ -40,6 +41,7 @@ import type {
|
||||
ExtensionUIDialogOptions,
|
||||
InputEvent,
|
||||
InputEventResult,
|
||||
McpNotificationEvent,
|
||||
MessageRenderer,
|
||||
RegisteredCommand,
|
||||
RegisteredTool,
|
||||
@@ -218,6 +220,14 @@ async function raceHandlerWithTimeout<T>(
|
||||
|
||||
const MAX_PENDING_CREDENTIAL_DISABLED = 32;
|
||||
|
||||
/**
|
||||
* Buffer cap for `mcp_notification` events received before {@link ExtensionRunner.initialize}
|
||||
* has run. Sized to match the manager-side buffer in `MCPManager.NOTIFICATION_BUFFER_CAP` so
|
||||
* the two layers can't drop different amounts of the same burst — the pipe drains, or it
|
||||
* spills, but it does so consistently at both ends. Drop-oldest under pressure.
|
||||
*/
|
||||
const MAX_PENDING_MCP_NOTIFICATIONS = 100;
|
||||
|
||||
/**
|
||||
* Events handled by the generic emit() method.
|
||||
* Events with dedicated emitXxx() methods are excluded for stronger type safety.
|
||||
@@ -327,6 +337,7 @@ export class ExtensionRunner {
|
||||
#getContextUsageFn: () => ContextUsage | undefined = () => undefined;
|
||||
#compactFn: (instructionsOrOptions?: string | CompactOptions) => Promise<void> = async () => {};
|
||||
#getSystemPromptFn: () => string[] = () => [];
|
||||
#getAsyncJobSnapshotFn: () => AsyncJobSnapshot | null = () => null;
|
||||
#newSessionHandler: NewSessionHandler = async () => ({ cancelled: false });
|
||||
#branchHandler: BranchHandler = async () => ({ cancelled: false });
|
||||
#navigateTreeHandler: NavigateTreeHandler = async () => ({ cancelled: false });
|
||||
@@ -345,6 +356,19 @@ export class ExtensionRunner {
|
||||
*/
|
||||
#pendingCredentialDisabled: CredentialDisabledEvent[] = [];
|
||||
|
||||
/**
|
||||
* Buffer for `mcp_notification` events received via {@link emitMcpNotification} before
|
||||
* {@link initialize} has run. Two-layer race: `MCPManager` also buffers frames until
|
||||
* its first `addNotificationListener` subscriber attaches, but the sdk.ts bridge is
|
||||
* registered inside `createAgentSession` — BEFORE the mode controller calls
|
||||
* `ExtensionRunner.initialize()`. Without this second buffer, the manager's drain
|
||||
* arrives at the bridge → the bridge calls `emitMcpNotification` → the runner drops
|
||||
* the frame because `#initialized === false`, and the frame evaporates a second time.
|
||||
* Bounded at {@link MAX_PENDING_MCP_NOTIFICATIONS}; oldest entries are dropped under
|
||||
* pressure. Drained in {@link initialize} once the runtime/UI context is wired.
|
||||
*/
|
||||
#pendingMcpNotifications: Array<Omit<McpNotificationEvent, "type">> = [];
|
||||
|
||||
/**
|
||||
* Timers scheduled by extensions through the sanctioned `ctx.setInterval` /
|
||||
* `ctx.setTimeout` helpers. Callbacks run with the same isolation as handler
|
||||
@@ -392,9 +416,11 @@ export class ExtensionRunner {
|
||||
getMemory?: () => MemoryRuntimeContext | undefined,
|
||||
private readonly settings?: Settings,
|
||||
private readonly localProtocolOptions?: LocalProtocolOptions,
|
||||
getAsyncJobSnapshot?: () => AsyncJobSnapshot | null,
|
||||
) {
|
||||
this.#uiContext = noOpUIContext;
|
||||
this.#getMemoryFn = getMemory;
|
||||
this.#getAsyncJobSnapshotFn = getAsyncJobSnapshot ?? (() => null);
|
||||
}
|
||||
|
||||
initialize(
|
||||
@@ -457,6 +483,23 @@ export class ExtensionRunner {
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// Drain events buffered by emitMcpNotification() before initialize ran, using the
|
||||
// same deferred-microtask ordering as the credential-disabled drain above so any
|
||||
// onError listener registered synchronously after initialize() still catches
|
||||
// handler errors during flush.
|
||||
const pendingMcp = this.#pendingMcpNotifications.splice(0);
|
||||
queueMicrotask(() => {
|
||||
for (const event of pendingMcp) {
|
||||
this.emit({ type: "mcp_notification", ...event }).catch((error: unknown) => {
|
||||
logger.warn("mcp_notification handler threw during initialize flush", {
|
||||
server: event.server,
|
||||
method: event.method,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
});
|
||||
});
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -484,6 +527,32 @@ export class ExtensionRunner {
|
||||
await this.emit({ type: "credential_disabled", ...event });
|
||||
}
|
||||
|
||||
/**
|
||||
* Forward an MCP server notification to extension handlers.
|
||||
*
|
||||
* If {@link initialize} has not yet run, the notification is buffered and replayed
|
||||
* once initialize wires the runtime/UI context. Matches the credential-disabled
|
||||
* deferral above: the sdk.ts bridge registers `MCPManager.addNotificationListener`
|
||||
* inside `createAgentSession` — BEFORE the mode controller calls `initialize()` on
|
||||
* this runner — so notification frames drained by the manager (either fresh
|
||||
* arrivals or replay from its own startup buffer) can reach us pre-init. Without
|
||||
* this buffer they would evaporate for a second time here.
|
||||
*
|
||||
* Bounded at {@link MAX_PENDING_MCP_NOTIFICATIONS}; oldest entries drop under
|
||||
* pressure. Never throws; per-handler errors are routed through {@link onError}
|
||||
* via {@link emit}'s normal isolation.
|
||||
*/
|
||||
async emitMcpNotification(event: Omit<McpNotificationEvent, "type">): Promise<void> {
|
||||
if (!this.#initialized) {
|
||||
if (this.#pendingMcpNotifications.length >= MAX_PENDING_MCP_NOTIFICATIONS) {
|
||||
this.#pendingMcpNotifications.shift();
|
||||
}
|
||||
this.#pendingMcpNotifications.push(event);
|
||||
return;
|
||||
}
|
||||
await this.emit({ type: "mcp_notification", ...event });
|
||||
}
|
||||
|
||||
/** Emits a session stop pass that can be cancelled with the active settle signal. */
|
||||
async emitSessionStop(event: Omit<SessionStopEvent, "type">): Promise<SessionStopEventResult | undefined> {
|
||||
if (event.signal.aborted) return undefined;
|
||||
@@ -666,6 +735,7 @@ export class ExtensionRunner {
|
||||
ui: this.#uiContext,
|
||||
getContextUsage: () => this.#getContextUsageFn(),
|
||||
compact: instructionsOrOptions => this.#compactFn(instructionsOrOptions),
|
||||
getAsyncJobSnapshot: () => this.#getAsyncJobSnapshotFn(),
|
||||
hasUI: this.hasUI(),
|
||||
cwd: this.cwd,
|
||||
sessionManager: this.sessionManager,
|
||||
|
||||
@@ -49,6 +49,7 @@ import type { LocalProtocolOptions } from "../../internal-urls/local-protocol";
|
||||
import type { MemoryRuntimeContext } from "../../memory-backend";
|
||||
import type { CustomEditor } from "../../modes/components/custom-editor";
|
||||
import type { Theme } from "../../modes/theme/theme";
|
||||
import type { AsyncJobSnapshot } from "../../session/agent-session";
|
||||
import type { CompactMode } from "../../session/compact-modes";
|
||||
import type { CustomMessage, CustomMessagePayload } from "../../session/messages";
|
||||
import type { ReadonlySessionManager, SessionManager } from "../../session/session-manager";
|
||||
@@ -415,6 +416,8 @@ export interface ExtensionContext {
|
||||
ui: ExtensionUIContext;
|
||||
/** Get current context usage for the active model. */
|
||||
getContextUsage(): ContextUsage | undefined;
|
||||
/** Get a read-only snapshot of async jobs owned by this session. */
|
||||
getAsyncJobSnapshot(): AsyncJobSnapshot | null;
|
||||
/** Compact the session context (interactive mode shows UI). */
|
||||
compact(instructionsOrOptions?: string | CompactOptions): Promise<void>;
|
||||
/** Whether UI is available (false in print/RPC mode) */
|
||||
@@ -713,6 +716,31 @@ export interface CredentialDisabledEvent {
|
||||
disabledCause: string;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// MCP Events
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
* Fired for every JSON-RPC notification received from a connected MCP server,
|
||||
* AFTER the runtime's own handling of known list/update methods. Unknown or
|
||||
* server-custom methods are delivered too — extensions can bridge them into
|
||||
* session behavior by inspecting `method`/`params` and injecting a follow-up
|
||||
* via `pi.sendMessage(..., { deliverAs })` or `pi.sendUserMessage(...)`.
|
||||
*/
|
||||
export interface McpNotificationEvent {
|
||||
type: "mcp_notification";
|
||||
/**
|
||||
* Server name as declared in the MCP config (raw, unsanitized). Note this
|
||||
* differs from the sanitized prefix used in `mcp__<sanitized_server>_<tool>`
|
||||
* tool names — filter by this raw name, not by tool-name prefix matching.
|
||||
*/
|
||||
server: string;
|
||||
/** JSON-RPC method (e.g. `notifications/tools/list_changed`, or server-custom). */
|
||||
method: string;
|
||||
/** JSON-RPC params, opaque to the runtime. */
|
||||
params: unknown;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// User Bash Events
|
||||
// ============================================================================
|
||||
@@ -941,6 +969,7 @@ export type ExtensionEvent =
|
||||
| TodoReminderEvent
|
||||
| GoalUpdatedEvent
|
||||
| CredentialDisabledEvent
|
||||
| McpNotificationEvent
|
||||
| UserBashEvent
|
||||
| UserPythonEvent
|
||||
| InputEvent
|
||||
@@ -1133,6 +1162,7 @@ export interface ExtensionAPI {
|
||||
on(event: "tool_result", handler: ExtensionHandler<ToolResultEvent, ToolResultEventResult>): void;
|
||||
on(event: "user_bash", handler: ExtensionHandler<UserBashEvent, UserBashEventResult>): void;
|
||||
on(event: "user_python", handler: ExtensionHandler<UserPythonEvent, UserPythonEventResult>): void;
|
||||
on(event: "mcp_notification", handler: ExtensionHandler<McpNotificationEvent>): void;
|
||||
|
||||
// =========================================================================
|
||||
// Tool Registration
|
||||
|
||||
@@ -17,6 +17,7 @@ import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import type { AgentToolResult, AgentToolUpdateCallback } from "@oh-my-pi/pi-agent-core";
|
||||
import { type AuthCredential, SqliteAuthCredentialStore, type TSchema } from "@oh-my-pi/pi-ai";
|
||||
import { piEscapeRegexLiteral, piJoinPath } from "@oh-my-pi/pi-ai/providers/cursor-pi-args";
|
||||
import { getKeybindings, type Keybinding, Text } from "@oh-my-pi/pi-tui";
|
||||
import {
|
||||
getAgentDbPath,
|
||||
@@ -298,16 +299,6 @@ function lineRangePath(readPath: string, offset: number | undefined, limit: numb
|
||||
return `${readPath}:${start}-${end}`;
|
||||
}
|
||||
|
||||
function escapeRegexLiteral(value: string): string {
|
||||
return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
||||
}
|
||||
|
||||
function joinLegacyGlob(searchPath: string, pattern: string): string {
|
||||
if (path.isAbsolute(pattern)) return pattern;
|
||||
if (!searchPath || searchPath === ".") return pattern;
|
||||
return path.join(searchPath, pattern);
|
||||
}
|
||||
|
||||
function normalizeLegacyLimit(limit: number | undefined, fallback: number): number {
|
||||
if (limit === undefined || !Number.isFinite(limit)) return fallback;
|
||||
return Math.max(1, Math.floor(limit));
|
||||
@@ -512,7 +503,7 @@ export function createGrepToolDefinition(cwd: string, options?: GrepToolOptions)
|
||||
renderResult: legacyRenderResult,
|
||||
execute: (toolCallId, params, signal, onUpdate) => {
|
||||
const rawPattern = stringField(params, "pattern") ?? "";
|
||||
const pattern = booleanField(params, "literal") ? escapeRegexLiteral(rawPattern) : rawPattern;
|
||||
const pattern = booleanField(params, "literal") ? piEscapeRegexLiteral(rawPattern) : rawPattern;
|
||||
const searchPath = stringField(params, "path") ?? ".";
|
||||
const glob = stringField(params, "glob");
|
||||
const context = numberField(params, "context");
|
||||
@@ -529,7 +520,7 @@ export function createGrepToolDefinition(cwd: string, options?: GrepToolOptions)
|
||||
toolCallId,
|
||||
{
|
||||
pattern,
|
||||
path: glob ? joinLegacyGlob(searchPath, glob) : searchPath,
|
||||
path: glob ? piJoinPath(searchPath, glob) : searchPath,
|
||||
case: booleanField(params, "ignoreCase") ? false : undefined,
|
||||
},
|
||||
signal,
|
||||
@@ -587,7 +578,7 @@ export function createFindToolDefinition(cwd: string, options?: FindToolOptions)
|
||||
}
|
||||
return tool.execute(
|
||||
toolCallId,
|
||||
{ path: joinLegacyGlob(searchPath, pattern), hidden: true, gitignore: true, limit },
|
||||
{ path: piJoinPath(searchPath, pattern), hidden: true, gitignore: true, limit },
|
||||
signal,
|
||||
onUpdate,
|
||||
);
|
||||
|
||||
@@ -41,7 +41,7 @@ import {
|
||||
type ScopedModel,
|
||||
} from "./config/model-resolver";
|
||||
import { ModelsConfigFile } from "./config/models-config";
|
||||
import { getDefault, type SettingPath, Settings, settings } from "./config/settings";
|
||||
import { getDefault, type SettingPath, Settings, type SettingValue, settings } from "./config/settings";
|
||||
import { initializeWithSettings } from "./discovery";
|
||||
import {
|
||||
clearPluginRootsAndCaches,
|
||||
@@ -90,14 +90,7 @@ import { createPersistedSubagentReviverFactory } from "./task/persisted-revive";
|
||||
import { createTelemetryExportConfig, initTelemetryExport, isTelemetryExportEnabled } from "./telemetry-export";
|
||||
import { concreteThinkingLevel, parseConfiguredThinkingLevel } from "./thinking";
|
||||
import type { LspStartupServerInfo } from "./tools";
|
||||
import {
|
||||
getChangelogPath,
|
||||
parseChangelog,
|
||||
parseChangelogVersion,
|
||||
readLastChangelogVersion,
|
||||
selectStartupChangelog,
|
||||
writeLastChangelogVersion,
|
||||
} from "./utils/changelog";
|
||||
import { getChangelogPath, resolveStartupChangelogForDisplay, type StartupChangelogSelection } from "./utils/changelog";
|
||||
import { EventBus } from "./utils/event-bus";
|
||||
import { withTimeoutSignal } from "./utils/fetch-timeout";
|
||||
|
||||
@@ -415,7 +408,7 @@ export function createAcpSessionFactory(args: AcpSessionFactoryOptions): AcpSess
|
||||
async function runInteractiveMode(
|
||||
session: AgentSession,
|
||||
version: string,
|
||||
changelogMarkdown: string | undefined,
|
||||
startupChangelog: StartupChangelogSelection | undefined,
|
||||
notifs: (InteractiveModeNotify | null)[],
|
||||
versionCheckPromise: Promise<string | undefined>,
|
||||
initialMessages: string[],
|
||||
@@ -433,7 +426,7 @@ async function runInteractiveMode(
|
||||
const mode = new InteractiveMode(
|
||||
session,
|
||||
version,
|
||||
changelogMarkdown,
|
||||
startupChangelog,
|
||||
setExtensionUIContext,
|
||||
lspServers,
|
||||
mcpManager,
|
||||
@@ -679,33 +672,19 @@ async function resolveScopedModels(
|
||||
);
|
||||
}
|
||||
|
||||
async function getChangelogForDisplay(parsed: Args): Promise<string | undefined> {
|
||||
async function getChangelogForDisplay(
|
||||
parsed: Args,
|
||||
mode: SettingValue<"startup.changelogMode">,
|
||||
): Promise<StartupChangelogSelection | undefined> {
|
||||
if (parsed.continue || parsed.resume || isForeignSessionImport(parsed)) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const lastVersion = await readLastChangelogVersion();
|
||||
const parsedLastVersion = parseChangelogVersion(lastVersion);
|
||||
if (!parsedLastVersion) {
|
||||
await writeLastChangelogVersion(VERSION);
|
||||
return undefined;
|
||||
}
|
||||
if (lastVersion === VERSION) {
|
||||
// Steady state: user already saw the current version's changelog. Skip the file read + parse.
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const changelogPath = getChangelogPath();
|
||||
const entries = await parseChangelog(changelogPath);
|
||||
const startupChangelog = selectStartupChangelog(entries, lastVersion, VERSION);
|
||||
if (startupChangelog.persistCurrentVersion) {
|
||||
await writeLastChangelogVersion(VERSION);
|
||||
}
|
||||
if (startupChangelog.markdown) {
|
||||
return startupChangelog.markdown;
|
||||
}
|
||||
|
||||
return undefined;
|
||||
return resolveStartupChangelogForDisplay({
|
||||
mode,
|
||||
currentVersion: VERSION,
|
||||
changelogPath: getChangelogPath(),
|
||||
});
|
||||
}
|
||||
|
||||
const SESSION_ID_ARG_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;
|
||||
@@ -1667,7 +1646,12 @@ export async function runRootCommand(
|
||||
await runRpcMode(session, mode === "rpc-ui" ? setToolUIContext : undefined, eventBus, rpcInput);
|
||||
} else if (isInteractive) {
|
||||
const versionCheckPromise = checkForNewVersion(VERSION).catch(() => undefined);
|
||||
const changelogMarkdown = await logger.time("main:getChangelogForDisplay", getChangelogForDisplay, parsedArgs);
|
||||
const startupChangelog = await logger.time(
|
||||
"main:getChangelogForDisplay",
|
||||
getChangelogForDisplay,
|
||||
parsedArgs,
|
||||
settingsInstance.get("startup.changelogMode"),
|
||||
);
|
||||
|
||||
const modelScopeNotification = buildModelScopeNotification(
|
||||
scopedModels,
|
||||
@@ -1692,7 +1676,7 @@ export async function runRootCommand(
|
||||
await runInteractiveMode(
|
||||
session,
|
||||
VERSION,
|
||||
changelogMarkdown,
|
||||
startupChangelog,
|
||||
notifs,
|
||||
versionCheckPromise,
|
||||
initialArgs.messages,
|
||||
|
||||
@@ -94,6 +94,14 @@ const STARTUP_TIMEOUT_MS = 250;
|
||||
const RECONNECT_BURST_WINDOW_MS = 30_000;
|
||||
const RECONNECT_BURST_LIMIT = 5;
|
||||
|
||||
/**
|
||||
* Bounded buffer for notifications received before any listener attaches.
|
||||
* Mirrors {@link IrcBus}'s `MAILBOX_CAP` — drop-oldest on overflow. Drained
|
||||
* into the first {@link MCPManager.addNotificationListener} subscriber, then
|
||||
* cleared; subsequent frames deliver directly to attached listeners.
|
||||
*/
|
||||
const NOTIFICATION_BUFFER_CAP = 100;
|
||||
|
||||
function trackPromise<T>(promise: Promise<T>): TrackedPromise<T> {
|
||||
const tracked: TrackedPromise<T> = { promise, status: "pending" };
|
||||
promise.then(
|
||||
@@ -194,8 +202,14 @@ export class MCPManager {
|
||||
#sources = new Map<string, SourceMeta>();
|
||||
#authStorage: AuthStorage | null = null;
|
||||
#authHandler?: MCPAuthHandler;
|
||||
#onNotification?: (serverName: string, method: string, params: unknown) => void;
|
||||
#onToolsChanged?: (tools: CustomTool<TSchema, MCPToolDetails>[]) => void;
|
||||
#notificationListeners = new Set<(serverName: string, method: string, params: unknown) => void>();
|
||||
/**
|
||||
* Notifications received before any listener attached, to be drained on
|
||||
* the first {@link addNotificationListener} call. Bounded by
|
||||
* {@link NOTIFICATION_BUFFER_CAP}, drop-oldest on overflow.
|
||||
*/
|
||||
#pendingNotifications: Array<{ server: string; method: string; params: unknown }> = [];
|
||||
#onToolsChanged?: (tools: CustomTool<TSchema, MCPToolDetails>[]) => void | Promise<void>;
|
||||
#onResourcesChanged?: (serverName: string, uri: string) => void;
|
||||
#onPromptsChanged?: (serverName: string) => void;
|
||||
#notificationsEnabled = false;
|
||||
@@ -219,16 +233,66 @@ export class MCPManager {
|
||||
) {}
|
||||
|
||||
/**
|
||||
* Set a callback to receive all server notifications.
|
||||
* Register a listener for server-initiated MCP notifications.
|
||||
*
|
||||
* The listener is called for every JSON-RPC notification received from any
|
||||
* connected server, AFTER the manager's own handling of known methods
|
||||
* (`notifications/tools/list_changed`, `notifications/resources/list_changed`,
|
||||
* `notifications/resources/updated`, `notifications/prompts/list_changed`).
|
||||
* For list-change methods the internal refresh promise is awaited before
|
||||
* fanout, so listeners observe up-to-date manager and tool state. Unknown
|
||||
* or server-custom methods are also delivered, letting consumers bridge
|
||||
* server-initiated events into session-level behavior (e.g. an extension
|
||||
* injecting a steer via `pi.sendMessage`).
|
||||
*
|
||||
* Notifications received before any listener attached are buffered
|
||||
* (bounded FIFO, cap {@link NOTIFICATION_BUFFER_CAP}, drop-oldest) and
|
||||
* drained into the first subscriber — matches {@link setOnPromptsChanged}'s
|
||||
* replay-on-attach and {@link IrcBus}'s mailbox semantics.
|
||||
*
|
||||
* Returns an unsubscribe function; call it to remove the listener.
|
||||
*
|
||||
* Multiple listeners are allowed; each is invoked with independent error
|
||||
* isolation — a listener that throws does not prevent other listeners from
|
||||
* firing.
|
||||
*/
|
||||
setOnNotification(handler: (serverName: string, method: string, params: unknown) => void): void {
|
||||
this.#onNotification = handler;
|
||||
addNotificationListener(listener: (serverName: string, method: string, params: unknown) => void): () => void {
|
||||
const wasEmpty = this.#notificationListeners.size === 0;
|
||||
this.#notificationListeners.add(listener);
|
||||
|
||||
// Drain startup-buffered notifications into the first attaching listener.
|
||||
if (wasEmpty && this.#pendingNotifications.length > 0) {
|
||||
const pending = this.#pendingNotifications.splice(0);
|
||||
for (const frame of pending) {
|
||||
try {
|
||||
listener(frame.server, frame.method, frame.params);
|
||||
} catch (error) {
|
||||
logger.debug("MCP notification listener threw during buffered drain", {
|
||||
path: `mcp:${frame.server}`,
|
||||
method: frame.method,
|
||||
error,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return () => {
|
||||
this.#notificationListeners.delete(listener);
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Set a callback to fire when any server's tools change.
|
||||
*
|
||||
* May return a Promise; if so, {@link refreshServerTools} awaits it so that
|
||||
* downstream consumers (e.g. `mcp_notification` listeners for
|
||||
* `notifications/tools/list_changed`) observe not just the manager's
|
||||
* refreshed tool set but also any session-level rebind driven by the
|
||||
* handler (`session.refreshMCPTools`). Other callsites (initial connect,
|
||||
* disconnect, reconnect) invoke the handler synchronously — their downstream
|
||||
* chains don't need to serialize on the rebind.
|
||||
*/
|
||||
setOnToolsChanged(handler: (tools: CustomTool<TSchema, MCPToolDetails>[]) => void): void {
|
||||
setOnToolsChanged(handler: (tools: CustomTool<TSchema, MCPToolDetails>[]) => void | Promise<void>): void {
|
||||
this.#onToolsChanged = handler;
|
||||
}
|
||||
|
||||
@@ -488,7 +552,7 @@ export class MCPManager {
|
||||
this.reconnectServer(name, options);
|
||||
const customTools = MCPTool.fromTools(connection, serverTools, reconnect);
|
||||
this.#replaceServerTools(name, customTools);
|
||||
this.#onToolsChanged?.(this.#tools);
|
||||
void this.#onToolsChanged?.(this.#tools);
|
||||
void this.toolCache?.set(name, config, serverTools);
|
||||
|
||||
onStatus?.({ type: "connected", serverName: name });
|
||||
@@ -597,7 +661,7 @@ export class MCPManager {
|
||||
sortMCPToolsByName(this.#tools);
|
||||
}
|
||||
|
||||
#triggerNotificationRefresh(serverName: string, kind: "tools" | "resources" | "prompts"): void {
|
||||
#triggerNotificationRefresh(serverName: string, kind: "tools" | "resources" | "prompts"): Promise<void> {
|
||||
const refresh = (() => {
|
||||
switch (kind) {
|
||||
case "tools":
|
||||
@@ -608,22 +672,33 @@ export class MCPManager {
|
||||
return this.refreshServerPrompts(serverName);
|
||||
}
|
||||
})();
|
||||
void refresh.catch(error => {
|
||||
return refresh.catch(error => {
|
||||
logger.debug("Failed MCP notification refresh", { path: `mcp:${serverName}`, kind, error });
|
||||
});
|
||||
}
|
||||
#handleServerNotification(serverName: string, method: string, params: unknown): void {
|
||||
async #handleServerNotification(serverName: string, method: string, params: unknown): Promise<void> {
|
||||
logger.debug("MCP notification received", { path: `mcp:${serverName}`, method });
|
||||
|
||||
// Only trigger refresh if the connection is already stored — during the
|
||||
// initial connect handshake, notifications may arrive before
|
||||
// `#connections.set()` completes, and `refreshServer*` would no-op
|
||||
// anyway. Skipping the await in that case preserves arrival order
|
||||
// across concurrently-dispatched notifications (an awaited refresh,
|
||||
// even a no-op, yields a microtask that lets later frames overtake).
|
||||
const connectionKnown = this.#connections.has(serverName);
|
||||
let refreshPromise: Promise<void> | undefined;
|
||||
switch (method) {
|
||||
case MCPNotificationMethods.TOOLS_LIST_CHANGED:
|
||||
this.#triggerNotificationRefresh(serverName, "tools");
|
||||
if (connectionKnown) refreshPromise = this.#triggerNotificationRefresh(serverName, "tools");
|
||||
break;
|
||||
case MCPNotificationMethods.RESOURCES_LIST_CHANGED:
|
||||
this.#triggerNotificationRefresh(serverName, "resources");
|
||||
if (connectionKnown) refreshPromise = this.#triggerNotificationRefresh(serverName, "resources");
|
||||
break;
|
||||
case MCPNotificationMethods.RESOURCES_UPDATED: {
|
||||
const uri = (params as { uri?: string })?.uri;
|
||||
const uri =
|
||||
params && typeof params === "object" && "uri" in params && typeof params.uri === "string"
|
||||
? params.uri
|
||||
: undefined;
|
||||
const subscribed = this.#subscribedResources.get(serverName);
|
||||
if (uri && subscribed?.has(uri)) {
|
||||
this.#onResourcesChanged?.(serverName, uri);
|
||||
@@ -631,13 +706,40 @@ export class MCPManager {
|
||||
break;
|
||||
}
|
||||
case MCPNotificationMethods.PROMPTS_LIST_CHANGED:
|
||||
this.#triggerNotificationRefresh(serverName, "prompts");
|
||||
if (connectionKnown) refreshPromise = this.#triggerNotificationRefresh(serverName, "prompts");
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
this.#onNotification?.(serverName, method, params);
|
||||
// Await internal refresh so listeners see the manager's post-refresh
|
||||
// state (satisfies the documented "AFTER the manager's own handling"
|
||||
// contract on `addNotificationListener` — otherwise an extension acting
|
||||
// on `tools/list_changed` could hit stale `getTools()`).
|
||||
if (refreshPromise) {
|
||||
await refreshPromise;
|
||||
}
|
||||
|
||||
// Buffer for late-attaching subscribers when no listener exists yet.
|
||||
if (this.#notificationListeners.size === 0) {
|
||||
this.#pendingNotifications.push({ server: serverName, method, params });
|
||||
if (this.#pendingNotifications.length > NOTIFICATION_BUFFER_CAP) {
|
||||
this.#pendingNotifications.shift();
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
for (const listener of this.#notificationListeners) {
|
||||
try {
|
||||
listener(serverName, method, params);
|
||||
} catch (error) {
|
||||
logger.debug("MCP notification listener threw", {
|
||||
path: `mcp:${serverName}`,
|
||||
method,
|
||||
error,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Handle server-to-client JSON-RPC requests (e.g. ping, roots/list). */
|
||||
@@ -781,7 +883,7 @@ export class MCPManager {
|
||||
// Remove tools from this server and notify consumers
|
||||
const hadTools = this.#tools.some(t => t.mcpServerName === name);
|
||||
this.#tools = this.#tools.filter(t => t.mcpServerName !== name);
|
||||
if (hadTools) this.#onToolsChanged?.(this.#tools);
|
||||
if (hadTools) void this.#onToolsChanged?.(this.#tools);
|
||||
|
||||
// Notify prompt consumers so stale commands are cleared
|
||||
if (connection?.prompts?.length) this.#onPromptsChanged?.(name);
|
||||
@@ -1022,7 +1124,7 @@ export class MCPManager {
|
||||
const customTools = MCPTool.fromTools(connection, serverTools, reconnect);
|
||||
void this.toolCache?.set(name, config, serverTools);
|
||||
this.#replaceServerTools(name, customTools);
|
||||
this.#onToolsChanged?.(this.#tools);
|
||||
void this.#onToolsChanged?.(this.#tools);
|
||||
void this.#loadServerResourcesAndPrompts(name, connection);
|
||||
return connection;
|
||||
} catch (error) {
|
||||
@@ -1075,7 +1177,7 @@ export class MCPManager {
|
||||
|
||||
// Replace tools from this server
|
||||
this.#replaceServerTools(name, customTools);
|
||||
this.#onToolsChanged?.(this.#tools);
|
||||
await this.#onToolsChanged?.(this.#tools);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -1581,7 +1581,7 @@ export class AcpAgent implements Agent {
|
||||
#buildThinkingOptions(session: AgentSession): Array<{ value: string; name: string; description?: string }> {
|
||||
return [
|
||||
{ value: THINKING_OFF, name: "Off" },
|
||||
{ value: AUTO_THINKING, name: "Auto", description: "Auto-detect per prompt (low–xhigh)" },
|
||||
{ value: AUTO_THINKING, name: "Auto", description: "Auto-detect per prompt" },
|
||||
...session.getAvailableThinkingLevels().map(level => ({
|
||||
value: level,
|
||||
name: level,
|
||||
|
||||
@@ -0,0 +1,369 @@
|
||||
import {
|
||||
type Component,
|
||||
getSegmenter,
|
||||
matchesKey,
|
||||
type OverlayHandle,
|
||||
type OverlayOptions,
|
||||
truncateToWidth,
|
||||
visibleWidth,
|
||||
} from "@oh-my-pi/pi-tui";
|
||||
import { type ThemeColor, theme } from "../theme/theme";
|
||||
|
||||
const FRAME_INTERVAL_MS = 85;
|
||||
const FRAME_COUNT = 34;
|
||||
|
||||
const FIREWORK_THEME_COLORS = {
|
||||
cyan: "mdLink",
|
||||
dim: "dim",
|
||||
gold: "warning",
|
||||
green: "success",
|
||||
pink: "accent",
|
||||
violet: "thinkingXhigh",
|
||||
white: "text",
|
||||
} as const satisfies Record<string, ThemeColor>;
|
||||
|
||||
type FireworkColor = keyof typeof FIREWORK_THEME_COLORS;
|
||||
|
||||
/** The active Codex account fields retained between status refreshes. */
|
||||
export interface CodexResetUsageSnapshot {
|
||||
/** When this usage report was observed, if supplied by the provider. */
|
||||
observedAt?: number;
|
||||
/** Weekly usage, its quota identity, and its previously scheduled reset deadline. */
|
||||
sevenDay?: { percent: number; resetsAt?: number; tier?: string; plan?: string };
|
||||
savedResets?: number;
|
||||
}
|
||||
|
||||
/** A detected Codex quota event that can trigger the fireworks presentation. */
|
||||
export type CodexResetFireworksEvent =
|
||||
| { kind: "unscheduled-weekly-reset" }
|
||||
| { kind: "saved-reset-banked"; added: number; available: number };
|
||||
|
||||
interface CanvasCell {
|
||||
glyph: string;
|
||||
color: FireworkColor;
|
||||
priority: number;
|
||||
}
|
||||
|
||||
interface FireworkBurst {
|
||||
x: number;
|
||||
y: number;
|
||||
start: number;
|
||||
color: FireworkColor;
|
||||
}
|
||||
|
||||
interface ActiveFireworks {
|
||||
component: CodexResetFireworksComponent;
|
||||
overlay: OverlayHandle;
|
||||
}
|
||||
|
||||
interface CodexResetFireworksHost {
|
||||
ui: {
|
||||
showOverlay(component: Component, options?: OverlayOptions): OverlayHandle;
|
||||
setFocus(component: Component): void;
|
||||
requestRender(): void;
|
||||
readonly terminal: { readonly rows: number };
|
||||
};
|
||||
}
|
||||
|
||||
const BURSTS: readonly FireworkBurst[] = [
|
||||
{ x: 0.17, y: 0.46, start: 5, color: "pink" },
|
||||
{ x: 0.48, y: 0.2, start: 9, color: "cyan" },
|
||||
{ x: 0.78, y: 0.42, start: 13, color: "gold" },
|
||||
{ x: 0.31, y: 0.24, start: 17, color: "violet" },
|
||||
{ x: 0.65, y: 0.28, start: 21, color: "green" },
|
||||
{ x: 0.88, y: 0.18, start: 25, color: "pink" },
|
||||
];
|
||||
|
||||
/**
|
||||
* Compare consecutive reports for one Codex account. A saved-reset grant takes
|
||||
* precedence when both changes arrive in the same report. A verified decrease,
|
||||
* or a prior positive balance becoming unavailable, suppresses the weekly event
|
||||
* because the user may have redeemed a credit. Other weekly usage drops are
|
||||
* celebrated only before the previously scheduled reset deadline.
|
||||
*/
|
||||
export function detectCodexResetFireworks(
|
||||
previous: CodexResetUsageSnapshot,
|
||||
current: CodexResetUsageSnapshot,
|
||||
): CodexResetFireworksEvent | undefined {
|
||||
const previousSavedResets = previous.savedResets;
|
||||
const currentSavedResets = current.savedResets;
|
||||
if (previousSavedResets !== undefined) {
|
||||
if (currentSavedResets === undefined) {
|
||||
if (previousSavedResets > 0) return undefined;
|
||||
} else {
|
||||
if (currentSavedResets > previousSavedResets) {
|
||||
return {
|
||||
kind: "saved-reset-banked",
|
||||
added: currentSavedResets - previousSavedResets,
|
||||
available: currentSavedResets,
|
||||
};
|
||||
}
|
||||
if (currentSavedResets < previousSavedResets) return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
if (!previous.sevenDay || !current.sevenDay) return undefined;
|
||||
if (previous.sevenDay.tier !== current.sevenDay.tier || previous.sevenDay.plan !== current.sevenDay.plan) {
|
||||
return undefined;
|
||||
}
|
||||
const previousWeeklyPercent = Math.round(Math.max(0, Math.min(100, previous.sevenDay.percent)));
|
||||
const currentWeeklyPercent = Math.round(Math.max(0, Math.min(100, current.sevenDay.percent)));
|
||||
if (previousWeeklyPercent === 0 || currentWeeklyPercent >= previousWeeklyPercent) return undefined;
|
||||
|
||||
const scheduledResetAt = previous.sevenDay.resetsAt;
|
||||
if (
|
||||
scheduledResetAt === undefined ||
|
||||
!Number.isFinite(scheduledResetAt) ||
|
||||
typeof current.observedAt !== "number" ||
|
||||
!Number.isFinite(current.observedAt) ||
|
||||
current.observedAt >= scheduledResetAt
|
||||
) {
|
||||
return undefined;
|
||||
}
|
||||
return { kind: "unscheduled-weekly-reset" };
|
||||
}
|
||||
|
||||
function setCell(
|
||||
canvas: Array<Array<CanvasCell | undefined>>,
|
||||
x: number,
|
||||
y: number,
|
||||
glyph: string,
|
||||
color: FireworkColor,
|
||||
priority: number,
|
||||
): void {
|
||||
const row = canvas[y];
|
||||
if (!row || x < 0 || x >= row.length) return;
|
||||
const current = row[x];
|
||||
if (!current || priority >= current.priority) row[x] = { glyph, color, priority };
|
||||
}
|
||||
|
||||
function drawText(
|
||||
canvas: Array<Array<CanvasCell | undefined>>,
|
||||
x: number,
|
||||
y: number,
|
||||
text: string,
|
||||
color: FireworkColor,
|
||||
priority: number,
|
||||
): void {
|
||||
let column = x;
|
||||
for (const { segment } of getSegmenter().segment(text)) {
|
||||
const width = visibleWidth(segment);
|
||||
if (width <= 0) continue;
|
||||
setCell(canvas, column, y, segment, color, priority);
|
||||
for (let continuation = 1; continuation < width; continuation++) {
|
||||
setCell(canvas, column + continuation, y, "", color, priority);
|
||||
}
|
||||
column += width;
|
||||
}
|
||||
}
|
||||
|
||||
function drawBanner(
|
||||
canvas: Array<Array<CanvasCell | undefined>>,
|
||||
left: number,
|
||||
artWidth: number,
|
||||
height: number,
|
||||
event: CodexResetFireworksEvent,
|
||||
): void {
|
||||
if (height < 3 || artWidth < 8) return;
|
||||
const panelWidth = Math.min(62, artWidth);
|
||||
const panelLeft = left + Math.floor((artWidth - panelWidth) / 2);
|
||||
const top = height - 3;
|
||||
const innerWidth = panelWidth - 2;
|
||||
const titleText =
|
||||
event.kind === "unscheduled-weekly-reset" ? " O P E N A I R E S E T " : " S A V E D R E S E T ";
|
||||
const subtitleText =
|
||||
event.kind === "unscheduled-weekly-reset"
|
||||
? "Weekly usage cleared early · ESC to return"
|
||||
: event.added === 1
|
||||
? `New reset banked · ${event.available} available · ESC to return`
|
||||
: `${event.added} resets banked · ${event.available} available · ESC to return`;
|
||||
const title = truncateToWidth(titleText, innerWidth, "");
|
||||
const subtitle = truncateToWidth(subtitleText, innerWidth, "");
|
||||
const titleOffset = Math.floor((innerWidth - visibleWidth(title)) / 2);
|
||||
const subtitleOffset = Math.floor((innerWidth - visibleWidth(subtitle)) / 2);
|
||||
|
||||
drawText(canvas, panelLeft, top, `╭${"─".repeat(innerWidth)}╮`, "violet", 20);
|
||||
drawText(canvas, panelLeft + 1 + titleOffset, top, title, "gold", 21);
|
||||
drawText(canvas, panelLeft, top + 1, `│${" ".repeat(innerWidth)}│`, "violet", 20);
|
||||
drawText(canvas, panelLeft + 1 + subtitleOffset, top + 1, subtitle, "cyan", 21);
|
||||
drawText(canvas, panelLeft, top + 2, `╰${"─".repeat(innerWidth)}╯`, "violet", 20);
|
||||
}
|
||||
|
||||
function drawStars(
|
||||
canvas: Array<Array<CanvasCell | undefined>>,
|
||||
left: number,
|
||||
artWidth: number,
|
||||
skyHeight: number,
|
||||
frame: number,
|
||||
): void {
|
||||
if (skyHeight <= 0) return;
|
||||
const count = Math.min(26, Math.max(5, Math.floor(artWidth / 3)));
|
||||
for (let index = 0; index < count; index++) {
|
||||
const x = left + ((index * 37 + 11) % artWidth);
|
||||
const y = (index * 7 + 2) % skyHeight;
|
||||
const bright = (index + Math.floor(frame / 3)) % 5 === 0;
|
||||
setCell(canvas, x, y, bright ? "+" : ".", bright ? "white" : "dim", bright ? 2 : 1);
|
||||
}
|
||||
}
|
||||
|
||||
function drawBurst(
|
||||
canvas: Array<Array<CanvasCell | undefined>>,
|
||||
burst: FireworkBurst,
|
||||
left: number,
|
||||
artWidth: number,
|
||||
skyHeight: number,
|
||||
frame: number,
|
||||
): void {
|
||||
if (skyHeight <= 1) return;
|
||||
const centerX = left + Math.round((artWidth - 1) * burst.x);
|
||||
const centerY = Math.max(0, Math.min(skyHeight - 2, Math.round((skyHeight - 1) * burst.y)));
|
||||
const age = frame - burst.start;
|
||||
|
||||
if (age >= -6 && age < 0) {
|
||||
const progress = (age + 6) / 6;
|
||||
const y = skyHeight - 1 - Math.round(progress * (skyHeight - 1 - centerY));
|
||||
setCell(canvas, centerX, y, "^", "white", 8);
|
||||
setCell(canvas, centerX, y + 1, "|", burst.color, 7);
|
||||
setCell(canvas, centerX, y + 2, ".", "gold", 6);
|
||||
return;
|
||||
}
|
||||
if (age < 0 || age > 8) return;
|
||||
|
||||
const radius = age === 0 ? 0 : 0.8 + age * 0.92;
|
||||
const gravity = Math.floor((age * age) / 22);
|
||||
const glyphs = ["@", "*", "*", "+", "o", "o", ".", ".", "."] as const;
|
||||
const particleColor: FireworkColor = age <= 5 ? burst.color : age <= 7 ? "gold" : "dim";
|
||||
|
||||
for (let particle = 0; particle < 20; particle++) {
|
||||
const angle = (particle / 20) * Math.PI * 2 + burst.start * 0.17;
|
||||
const x = centerX + Math.round(Math.cos(angle) * radius * 1.75);
|
||||
const y = centerY + Math.round(Math.sin(angle) * radius * 0.58 + gravity);
|
||||
setCell(canvas, x, y, glyphs[age], particleColor, 10);
|
||||
if (age >= 2 && age <= 6) {
|
||||
const trailRadius = Math.max(0, radius - 1.4);
|
||||
const trailX = centerX + Math.round(Math.cos(angle) * trailRadius * 1.75);
|
||||
const trailY = centerY + Math.round(Math.sin(angle) * trailRadius * 0.58 + gravity);
|
||||
setCell(canvas, trailX, trailY, ".", "dim", 5);
|
||||
}
|
||||
}
|
||||
if (age <= 2) setCell(canvas, centerX, centerY, age === 0 ? "@" : "+", "white", 12);
|
||||
}
|
||||
|
||||
function renderCanvas(canvas: Array<Array<CanvasCell | undefined>>): string[] {
|
||||
return canvas.map(row => {
|
||||
let output = "";
|
||||
let run = "";
|
||||
let runColor: FireworkColor | undefined;
|
||||
for (const cell of row) {
|
||||
if (cell && cell.color !== runColor) {
|
||||
if (run) output += runColor ? theme.fg(FIREWORK_THEME_COLORS[runColor], run) : run;
|
||||
run = "";
|
||||
runColor = cell.color;
|
||||
}
|
||||
run += cell?.glyph ?? " ";
|
||||
}
|
||||
if (run) output += runColor ? theme.fg(FIREWORK_THEME_COLORS[runColor], run) : run;
|
||||
return output;
|
||||
});
|
||||
}
|
||||
|
||||
/** Render one deterministic animation frame for the top-third overlay. */
|
||||
function renderCodexResetFireworks(
|
||||
width: number,
|
||||
height: number,
|
||||
frame: number,
|
||||
event: CodexResetFireworksEvent,
|
||||
): string[] {
|
||||
const safeWidth = Math.max(1, Math.floor(width));
|
||||
const safeHeight = Math.max(1, Math.floor(height));
|
||||
const artWidth = Math.min(96, safeWidth);
|
||||
const left = Math.floor((safeWidth - artWidth) / 2);
|
||||
const skyHeight = Math.max(0, safeHeight - 3);
|
||||
const canvas = Array.from({ length: safeHeight }, () => new Array<CanvasCell | undefined>(safeWidth));
|
||||
|
||||
drawStars(canvas, left, artWidth, skyHeight, frame);
|
||||
for (const burst of BURSTS) drawBurst(canvas, burst, left, artWidth, skyHeight, frame);
|
||||
drawBanner(canvas, left, artWidth, safeHeight, event);
|
||||
return renderCanvas(canvas);
|
||||
}
|
||||
|
||||
class CodexResetFireworksComponent implements Component {
|
||||
#timer: NodeJS.Timeout | undefined;
|
||||
#done = Promise.withResolvers<void>();
|
||||
#frame = 0;
|
||||
#disposed = false;
|
||||
|
||||
constructor(
|
||||
readonly host: CodexResetFireworksHost,
|
||||
readonly event: CodexResetFireworksEvent,
|
||||
) {}
|
||||
|
||||
run(): Promise<void> {
|
||||
this.#timer ??= setInterval(() => {
|
||||
if (this.#disposed) return;
|
||||
this.#frame = (this.#frame + 1) % FRAME_COUNT;
|
||||
this.host.ui.requestRender();
|
||||
}, FRAME_INTERVAL_MS);
|
||||
this.host.ui.requestRender();
|
||||
return this.#done.promise;
|
||||
}
|
||||
|
||||
dispose(): void {
|
||||
if (this.#disposed) return;
|
||||
this.#disposed = true;
|
||||
if (this.#timer) {
|
||||
clearInterval(this.#timer);
|
||||
this.#timer = undefined;
|
||||
}
|
||||
this.#done.resolve();
|
||||
}
|
||||
|
||||
handleInput(data: string): void {
|
||||
if (matchesKey(data, "escape") || matchesKey(data, "esc")) this.#done.resolve();
|
||||
}
|
||||
|
||||
render(width: number): readonly string[] {
|
||||
const height = Math.max(1, Math.floor(this.host.ui.terminal.rows * 0.33));
|
||||
return renderCodexResetFireworks(width, height, this.#frame, this.event);
|
||||
}
|
||||
}
|
||||
|
||||
/** Owns the at-most-one modal celebration lifecycle for an interactive session. */
|
||||
export class CodexResetFireworksController {
|
||||
#active: ActiveFireworks | undefined;
|
||||
|
||||
constructor(private readonly host: CodexResetFireworksHost) {}
|
||||
|
||||
/** Present a celebration unless another one already owns the modal overlay. */
|
||||
show(event: CodexResetFireworksEvent): boolean {
|
||||
if (this.#active) return false;
|
||||
const component = new CodexResetFireworksComponent(this.host, event);
|
||||
const overlay = this.host.ui.showOverlay(component, {
|
||||
anchor: "top-center",
|
||||
width: "100%",
|
||||
maxHeight: "33%",
|
||||
margin: 0,
|
||||
});
|
||||
this.#active = { component, overlay };
|
||||
this.host.ui.setFocus(component);
|
||||
void component.run().then(() => this.#finish(component));
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Stop the active celebration and release its overlay, if present. */
|
||||
dispose(): void {
|
||||
const active = this.#active;
|
||||
if (!active) return;
|
||||
this.#active = undefined;
|
||||
active.component.dispose();
|
||||
active.overlay.hide();
|
||||
}
|
||||
|
||||
#finish(component: CodexResetFireworksComponent): void {
|
||||
const active = this.#active;
|
||||
if (!active || active.component !== component) return;
|
||||
this.#active = undefined;
|
||||
component.dispose();
|
||||
active.overlay.hide();
|
||||
}
|
||||
}
|
||||
@@ -15,6 +15,11 @@ import { getSessionAccentAnsi, getSessionAccentHex } from "../../../utils/sessio
|
||||
import { calculateTokensPerSecond } from "../../../utils/token-rate";
|
||||
import { sanitizeStatusText } from "../../shared";
|
||||
import { theme } from "../../theme/theme";
|
||||
import {
|
||||
type CodexResetFireworksEvent,
|
||||
type CodexResetUsageSnapshot,
|
||||
detectCodexResetFireworks,
|
||||
} from "../codex-reset-fireworks";
|
||||
import { canReuseCachedPr, createPrCacheContext, isSamePrCacheContext, type PrCacheContext } from "./git-utils";
|
||||
import { getPreset } from "./presets";
|
||||
import { renderSegment, type SegmentContext } from "./segments";
|
||||
@@ -30,6 +35,36 @@ import type {
|
||||
const JJ_REFRESH_TTL_MS = 5000;
|
||||
const WATCHER_FAILURE_POLL_TTL_MS = 5000;
|
||||
|
||||
function normalizeCodexIdentityValue(value: unknown): string | undefined {
|
||||
return typeof value === "string" && value.trim() ? value.trim().toLowerCase() : undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fireworks are stateful, so their report match must be stricter than the
|
||||
* status display's fallback matching: every known credential identifier must
|
||||
* be present and equal or a workspace sibling can mutate this account's
|
||||
* baseline.
|
||||
*/
|
||||
function codexReportMatchesExactIdentity(report: UsageReport, identity: OAuthAccountIdentity | undefined): boolean {
|
||||
if (!identity) return false;
|
||||
const accountId = normalizeCodexIdentityValue(identity.accountId);
|
||||
const email = normalizeCodexIdentityValue(identity.email);
|
||||
const projectId = normalizeCodexIdentityValue(identity.projectId);
|
||||
const orgId = normalizeCodexIdentityValue(identity.orgId);
|
||||
if (!accountId && !email && !projectId && !orgId) return false;
|
||||
|
||||
const metadata = report.metadata ?? {};
|
||||
const reportAccountId =
|
||||
normalizeCodexIdentityValue(metadata.accountId) ?? normalizeCodexIdentityValue(metadata.account_id);
|
||||
const reportProjectId =
|
||||
normalizeCodexIdentityValue(metadata.projectId) ?? normalizeCodexIdentityValue(metadata.project_id);
|
||||
if (accountId && reportAccountId !== accountId) return false;
|
||||
if (email && normalizeCodexIdentityValue(metadata.email) !== email) return false;
|
||||
if (projectId && reportProjectId !== projectId) return false;
|
||||
if (orgId && normalizeCodexIdentityValue(metadata.orgId) !== orgId) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
// Context-usage memo
|
||||
// ═══════════════════════════════════════════════════════════════════════════
|
||||
@@ -367,6 +402,12 @@ export class StatusLineComponent implements Component {
|
||||
#usageFetchedAt = 0;
|
||||
#usageInFlight = false;
|
||||
#usageStartTimer: Timer | null = null;
|
||||
// A timed-out request may still resolve. Its result remains eligible only
|
||||
// until a newer request has applied.
|
||||
#usageRefreshSequence = 0;
|
||||
#latestAppliedUsageRefreshSequence = 0;
|
||||
#codexResetSnapshots = new Map<string, CodexResetUsageSnapshot>();
|
||||
#onCodexResetFireworks: ((event: CodexResetFireworksEvent) => void) | undefined;
|
||||
// Context-usage memo. The status line redraws on every agent event, so the
|
||||
// hot path must not recompute context tokens unless an input changed.
|
||||
// `getContextUsage()` anchors on the last assistant's real prompt-token
|
||||
@@ -585,6 +626,11 @@ export class StatusLineComponent implements Component {
|
||||
this.#collabStatus = status;
|
||||
}
|
||||
|
||||
/** Set the callback that presents detected Codex reset celebrations, or clear it with `undefined`. */
|
||||
setCodexResetFireworksHandler(handler: ((event: CodexResetFireworksEvent) => void) | undefined): void {
|
||||
this.#onCodexResetFireworks = handler;
|
||||
}
|
||||
|
||||
setHookStatus(key: string, text: string | undefined): void {
|
||||
if (text === undefined) {
|
||||
this.#hookStatuses.delete(key);
|
||||
@@ -660,6 +706,8 @@ export class StatusLineComponent implements Component {
|
||||
this.#resetJjRequests();
|
||||
this.#onBranchChange = null;
|
||||
this.#clearUsageStartTimer();
|
||||
this.#onCodexResetFireworks = undefined;
|
||||
this.#codexResetSnapshots.clear();
|
||||
this.#retireGitWatcher();
|
||||
}
|
||||
|
||||
@@ -1125,10 +1173,8 @@ export class StatusLineComponent implements Component {
|
||||
return this.#vibeWorkerTokenRate?.() ?? null;
|
||||
}
|
||||
|
||||
#getUsageContextKey(session: AgentSession): string {
|
||||
const activeProvider = session.state.model?.provider ?? session.model?.provider ?? "";
|
||||
#formatUsageContextKey(activeProvider: string | undefined, identity: OAuthAccountIdentity | undefined): string {
|
||||
if (!activeProvider) return "";
|
||||
const identity = session.modelRegistry?.authStorage?.getOAuthAccountIdentity(activeProvider, session.sessionId);
|
||||
// orgId is part of the key: rotating between two same-email Anthropic
|
||||
// subscriptions must invalidate the cached usage immediately instead of
|
||||
// showing the previous org's quota for the rest of the cache TTL.
|
||||
@@ -1141,6 +1187,14 @@ export class StatusLineComponent implements Component {
|
||||
].join("\0");
|
||||
}
|
||||
|
||||
#getUsageContextKey(session: AgentSession): string {
|
||||
const activeProvider = session.state.model?.provider ?? session.model?.provider;
|
||||
const identity = activeProvider
|
||||
? session.modelRegistry?.authStorage?.getOAuthAccountIdentity(activeProvider, session.sessionId)
|
||||
: undefined;
|
||||
return this.#formatUsageContextKey(activeProvider, identity);
|
||||
}
|
||||
|
||||
/**
|
||||
* Startup redraws only arm a short-delayed task; timeout releases the render
|
||||
* cadence while a late successful fetch can still refresh the cached segment.
|
||||
@@ -1170,40 +1224,60 @@ export class StatusLineComponent implements Component {
|
||||
this.#usageInFlight = false;
|
||||
return;
|
||||
}
|
||||
const sequence = ++this.#usageRefreshSequence;
|
||||
const signal = AbortSignal.timeout(STATUS_USAGE_REFRESH_TIMEOUT_MS);
|
||||
let reportsPromise: Promise<unknown> | undefined;
|
||||
try {
|
||||
reportsPromise = fetcher.call(session, signal);
|
||||
this.#applyUsageRefreshReports(session, await this.#raceUsageRefreshWithSignal(reportsPromise, signal));
|
||||
this.#applyUsageRefreshReports(
|
||||
session,
|
||||
await this.#raceUsageRefreshWithSignal(reportsPromise, signal),
|
||||
sequence,
|
||||
);
|
||||
} catch {
|
||||
if (this.session !== session) return;
|
||||
this.#usageFetchedAt = Date.now();
|
||||
if (signal.aborted && reportsPromise) {
|
||||
this.#observeLateUsageRefresh(session, reportsPromise);
|
||||
this.#observeLateUsageRefresh(session, reportsPromise, sequence);
|
||||
}
|
||||
} finally {
|
||||
if (this.session === session) this.#usageInFlight = false;
|
||||
}
|
||||
}
|
||||
|
||||
#applyUsageRefreshReports(session: AgentSession, reports: unknown): void {
|
||||
if (this.#disposed || this.session !== session) return;
|
||||
#applyUsageRefreshReports(session: AgentSession, reports: unknown, sequence: number): void {
|
||||
if (this.#disposed || this.session !== session || sequence < this.#latestAppliedUsageRefreshSequence) {
|
||||
return;
|
||||
}
|
||||
this.#latestAppliedUsageRefreshSequence = sequence;
|
||||
const activeProvider = session.state.model?.provider ?? session.model?.provider;
|
||||
const activeIdentity =
|
||||
activeProvider && session.modelRegistry?.authStorage
|
||||
? session.modelRegistry.authStorage.getOAuthAccountIdentity(activeProvider, session.sessionId)
|
||||
: undefined;
|
||||
this.#cachedUsage = this.#normalizeUsageReports(reports, activeProvider, activeIdentity);
|
||||
const normalized = this.#normalizeUsageReports(reports, activeProvider, activeIdentity);
|
||||
const resetSnapshot =
|
||||
activeProvider === "openai-codex" ? this.#normalizeCodexResetSnapshot(reports, activeIdentity) : null;
|
||||
this.#cachedUsage = normalized;
|
||||
this.#usageFetchedAt = Date.now();
|
||||
if (!resetSnapshot) return;
|
||||
const contextKey = this.#formatUsageContextKey(activeProvider, activeIdentity);
|
||||
const previous = this.#codexResetSnapshots.get(contextKey);
|
||||
this.#codexResetSnapshots.set(contextKey, resetSnapshot);
|
||||
if (!previous || !settings.get("tui.codexResetFireworks")) return;
|
||||
const event = detectCodexResetFireworks(previous, resetSnapshot);
|
||||
if (event) this.#onCodexResetFireworks?.(event);
|
||||
}
|
||||
|
||||
#observeLateUsageRefresh(session: AgentSession, reportsPromise: Promise<unknown>): void {
|
||||
#observeLateUsageRefresh(session: AgentSession, reportsPromise: Promise<unknown>, sequence: number): void {
|
||||
void reportsPromise
|
||||
.then(reports => {
|
||||
this.#applyUsageRefreshReports(session, reports);
|
||||
this.#applyUsageRefreshReports(session, reports, sequence);
|
||||
})
|
||||
.catch(() => {
|
||||
if (this.#disposed || this.session !== session) return;
|
||||
if (this.#disposed || this.session !== session || sequence < this.#latestAppliedUsageRefreshSequence) {
|
||||
return;
|
||||
}
|
||||
this.#usageFetchedAt = Date.now();
|
||||
});
|
||||
}
|
||||
@@ -1220,6 +1294,72 @@ export class StatusLineComponent implements Component {
|
||||
}
|
||||
}
|
||||
|
||||
#normalizeCodexResetSnapshot(
|
||||
reports: unknown,
|
||||
activeIdentity: OAuthAccountIdentity | undefined,
|
||||
): CodexResetUsageSnapshot | null {
|
||||
if (!Array.isArray(reports)) return null;
|
||||
let matchingReport: UsageReport | undefined;
|
||||
for (const report of reports) {
|
||||
if (!report || typeof report !== "object") continue;
|
||||
if (
|
||||
!("provider" in report) ||
|
||||
report.provider !== "openai-codex" ||
|
||||
!("limits" in report) ||
|
||||
!Array.isArray(report.limits)
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
// The report boundary above validates the fields this extractor iterates;
|
||||
// optional metadata and credit fields are narrowed again before use.
|
||||
const usageReport = report as UsageReport;
|
||||
if (!codexReportMatchesExactIdentity(usageReport, activeIdentity)) continue;
|
||||
matchingReport = usageReport;
|
||||
break;
|
||||
}
|
||||
if (!matchingReport) return null;
|
||||
|
||||
const plan =
|
||||
typeof matchingReport.metadata?.planType === "string" && matchingReport.metadata.planType
|
||||
? matchingReport.metadata.planType
|
||||
: undefined;
|
||||
let sevenDay: CodexResetUsageSnapshot["sevenDay"];
|
||||
let sevenDayTier: string | undefined;
|
||||
for (const limit of matchingReport.limits) {
|
||||
if (!limit || typeof limit !== "object") continue;
|
||||
const candidate = limit as {
|
||||
scope?: { windowId?: string; tier?: string };
|
||||
window?: { resetsAt?: number };
|
||||
amount?: { usedFraction?: number };
|
||||
};
|
||||
const fraction = candidate.amount?.usedFraction;
|
||||
if (candidate.scope?.windowId !== "7d" || typeof fraction !== "number" || !Number.isFinite(fraction)) {
|
||||
continue;
|
||||
}
|
||||
const tier =
|
||||
typeof candidate.scope?.tier === "string" && candidate.scope.tier ? candidate.scope.tier : undefined;
|
||||
if (sevenDay && (sevenDayTier === undefined || tier)) continue;
|
||||
const resetsAt = candidate.window?.resetsAt;
|
||||
sevenDay = {
|
||||
percent: fraction * 100,
|
||||
resetsAt: typeof resetsAt === "number" && Number.isFinite(resetsAt) ? resetsAt : undefined,
|
||||
tier,
|
||||
plan,
|
||||
};
|
||||
sevenDayTier = tier;
|
||||
}
|
||||
|
||||
const fetchedAt = matchingReport.fetchedAt;
|
||||
const availableCount = matchingReport.resetCredits?.availableCount;
|
||||
const observedAt = typeof fetchedAt === "number" && Number.isFinite(fetchedAt) ? fetchedAt : undefined;
|
||||
const savedResets =
|
||||
typeof availableCount === "number" && Number.isFinite(availableCount)
|
||||
? Math.max(0, Math.trunc(availableCount))
|
||||
: undefined;
|
||||
if (!sevenDay && savedResets === undefined) return null;
|
||||
return { observedAt, sevenDay, savedResets };
|
||||
}
|
||||
|
||||
#normalizeUsageReports(
|
||||
reports: unknown,
|
||||
activeProvider?: string,
|
||||
@@ -1241,12 +1381,10 @@ export class StatusLineComponent implements Component {
|
||||
if (activeProvider && provider !== activeProvider) continue;
|
||||
const limits = (report as { limits?: unknown }).limits;
|
||||
if (!Array.isArray(limits)) continue;
|
||||
const usageReport = report as UsageReport;
|
||||
for (const limit of limits) {
|
||||
if (!limit || typeof limit !== "object") continue;
|
||||
if (
|
||||
activeIdentity &&
|
||||
!limitMatchesActiveAccount(report as UsageReport, limit as UsageLimit, activeIdentity)
|
||||
) {
|
||||
if (activeIdentity && !limitMatchesActiveAccount(usageReport, limit as UsageLimit, activeIdentity)) {
|
||||
continue;
|
||||
}
|
||||
const l = limit as {
|
||||
|
||||
@@ -203,6 +203,13 @@ class SafeToolRendererComponent implements Component {
|
||||
*/
|
||||
export interface TranscriptLiveRegionProbe {
|
||||
isBlockInLiveRegion(component: Component): boolean;
|
||||
/**
|
||||
* Whether none of the block's rows have entered native scrollback (see
|
||||
* `TranscriptContainer.isBlockUncommitted`). Optional: standalone hosts
|
||||
* without commit tracking omit it, and blocks treat their rows as
|
||||
* uncommitted.
|
||||
*/
|
||||
isBlockUncommitted?(component: Component): boolean;
|
||||
}
|
||||
|
||||
/** Minimal TUI surface ToolExecutionComponent uses to schedule repaints and share image budget. */
|
||||
@@ -337,10 +344,21 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac
|
||||
// transcript, e.g. in tests): whether this block is still repaintable.
|
||||
#liveRegion?: TranscriptLiveRegionProbe;
|
||||
// One-way latch for a detached (`async.state === "running"`) task block
|
||||
// that left the transcript live region: its rows are commit-eligible
|
||||
// history, so progress renders static gray and further partial snapshots are
|
||||
// dropped (see #maybeFreezeBackgroundTask).
|
||||
// whose rows became native-scrollback history — it left the transcript
|
||||
// live region, or its head rows were committed while it was still the
|
||||
// live tail. Further partial snapshots are dropped so committed rows are
|
||||
// never mutated (see #maybeFreezeBackgroundTask).
|
||||
#backgroundTaskFrozen = false;
|
||||
// Whether the freeze may restyle the progress rows static gray. Set only
|
||||
// when the latch fired while no row was committed: a recolor of rows
|
||||
// already on the tape would itself diverge immutable history and force an
|
||||
// erase-replay (or, with scrollback rebuild off, a duplicate slab).
|
||||
#backgroundTaskFrozenStyled = false;
|
||||
// Wall clock captured at each repaintable rebuild of a task card and
|
||||
// reused verbatim once the card freezes or any of its rows commit, so
|
||||
// time-derived rows (current-tool elapsed, retry countdown) cannot drift
|
||||
// a committed byte on later rebuilds (theme epoch, image toggles).
|
||||
#taskRenderNowMs = Date.now();
|
||||
// Set on each `render()` when the last painted pending shape must be
|
||||
// replayed wholesale when the first result arrives. Reset gates key off
|
||||
// these so a topology-changing update that lands before the shape reaches
|
||||
@@ -559,13 +577,17 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac
|
||||
_toolCallId?: string,
|
||||
): void {
|
||||
// A detached task spawn keeps streaming progress snapshots after the
|
||||
// block froze (left the transcript live region). Drop them: the rows are
|
||||
// static gray history now, and repainting would rewrite rows the engine
|
||||
// may already have committed to native scrollback. The terminal snapshot
|
||||
// (async completed/failed → isPartial=false) still applies so a block
|
||||
// that is still on screen settles on real results.
|
||||
if (isPartial && this.#toolName === "task" && this.#maybeFreezeBackgroundTask()) {
|
||||
return;
|
||||
// block froze (left the transcript live region, or its rows entered
|
||||
// native scrollback). Drop them: repainting would rewrite rows the
|
||||
// engine may already have committed. The terminal snapshot (async
|
||||
// completed/failed → isPartial=false) still settles a card that is
|
||||
// wholly uncommitted (still on screen); once any row is on the tape
|
||||
// the card is immutable history — replacing it would re-commit the
|
||||
// whole slab below the stale copy — so the settlement is dropped and
|
||||
// the job's result surfaces through its own delivery message.
|
||||
if (this.#toolName === "task" && this.#maybeFreezeBackgroundTask()) {
|
||||
if (isPartial) return;
|
||||
if (!(this.#liveRegion?.isBlockUncommitted?.(this) ?? true)) return;
|
||||
}
|
||||
const hadNoResult = this.#result === undefined;
|
||||
const wasPartialResult = this.#result !== undefined && this.#isPartial;
|
||||
@@ -713,22 +735,30 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac
|
||||
}
|
||||
|
||||
/**
|
||||
* Freeze a detached (`async.state === "running"`) task block once it leaves
|
||||
* the transcript's live region. Past that seam its rows are commit-eligible
|
||||
* native-scrollback history: repaint the progress rows static gray and drop
|
||||
* further partial snapshots. One-way — blocks never re-enter the live
|
||||
* region. Returns whether the block is frozen.
|
||||
* Freeze a detached (`async.state === "running"`) task block once its rows
|
||||
* become native-scrollback history: the block left the transcript's live
|
||||
* region (a later block streams below it), or — while it is still the
|
||||
* live tail — its head rows were committed because the frame outgrew the
|
||||
* viewport. Committed rows are immutable, so from that point every further
|
||||
* partial snapshot is dropped. Rows restyle static gray only when nothing
|
||||
* is committed yet; otherwise the bytes stay exactly as painted. One-way —
|
||||
* blocks never re-enter the live region. Returns whether the block is
|
||||
* frozen.
|
||||
*/
|
||||
#maybeFreezeBackgroundTask(): boolean {
|
||||
if (this.#backgroundTaskFrozen) return true;
|
||||
if (this.#toolName !== "task" || this.#liveRegion === undefined) return false;
|
||||
const asyncState = (this.#result?.details as { async?: { state?: string } } | undefined)?.async?.state;
|
||||
if (asyncState !== "running") return false;
|
||||
if (this.#liveRegion.isBlockInLiveRegion(this)) return false;
|
||||
const uncommitted = this.#liveRegion.isBlockUncommitted?.(this) ?? true;
|
||||
if (uncommitted && this.#liveRegion.isBlockInLiveRegion(this)) return false;
|
||||
this.#backgroundTaskFrozen = true;
|
||||
this.#updateSpinnerAnimation();
|
||||
this.#updateDisplay();
|
||||
this.#ui.requestRender();
|
||||
if (uncommitted) {
|
||||
this.#backgroundTaskFrozenStyled = true;
|
||||
this.#updateDisplay();
|
||||
this.#ui.requestRender();
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -784,16 +814,16 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac
|
||||
|
||||
/**
|
||||
* Keeps in-flight TV-wall frames out of immutable native scrollback: the
|
||||
* `vibe_wait` wall and displaceable snapshots (`hub` waiting polls, `todo`
|
||||
* lists). Their frames replace each other rather than append, and their
|
||||
* rows mutate every spinner tick — an unpinned commit records a per-tick
|
||||
* frozen snapshot AND force-seals the block (see TranscriptContainer's
|
||||
* committed-snapshot seal), so the next poll stacks a new frame instead of
|
||||
* displacing this one.
|
||||
* `vibe_wait` wall, displaceable snapshots (`hub` waiting polls, `todo`
|
||||
* lists), and live `task` calls. Their frames replace each other rather
|
||||
* than append — task progress rows rewrite in place on every snapshot —
|
||||
* so an unpinned commit records a per-tick frozen snapshot (and for
|
||||
* displaceable blocks force-seals them, stacking the next poll below).
|
||||
* The finalized frame commits exactly once when the pin lifts.
|
||||
*/
|
||||
isNativeScrollbackLiveRegionPinned(): boolean {
|
||||
if (this.isTranscriptBlockFinalized()) return false;
|
||||
return this.#toolName === "vibe_wait" || this.#displaceableByToolName !== undefined;
|
||||
return this.#toolName === "vibe_wait" || this.#toolName === "task" || this.#displaceableByToolName !== undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -827,8 +857,12 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac
|
||||
this.#sealed = true;
|
||||
this.#displaceableByToolName = undefined;
|
||||
// A sealed detached task is abandoned history: settle its progress rows
|
||||
// on static gray.
|
||||
// on static gray — but only while none of them are committed; a recolor
|
||||
// on the tape would diverge immutable history.
|
||||
this.#backgroundTaskFrozen = true;
|
||||
if (this.#liveRegion?.isBlockUncommitted?.(this) ?? true) {
|
||||
this.#backgroundTaskFrozenStyled = true;
|
||||
}
|
||||
this.stopAnimation();
|
||||
this.#updateDisplay();
|
||||
this.#ui.requestRender();
|
||||
@@ -888,7 +922,7 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac
|
||||
// TUI startup, so a result rendered before it lands must re-shape once it
|
||||
// does (it gates Image children vs text fallback in #rebuildDisplay); keyed
|
||||
// here for the same reason markdown.ts keys its render cache on it.
|
||||
const key = `${this.#resultVersion}|${this.#expanded}|${this.#isPartial}|${this.#spinnerFrame ?? "-"}|${this.#showImages}|${getThemeEpoch()}|${this.#displayInputVersion}|${this.#backgroundTaskFrozen}|${TERMINAL.imageProtocol ?? "-"}|${this.#imageSizeKey()}`;
|
||||
const key = `${this.#resultVersion}|${this.#expanded}|${this.#isPartial}|${this.#spinnerFrame ?? "-"}|${this.#showImages}|${getThemeEpoch()}|${this.#displayInputVersion}|${this.#backgroundTaskFrozenStyled}|${TERMINAL.imageProtocol ?? "-"}|${this.#imageSizeKey()}`;
|
||||
if (key === this.#lastDisplayKey && this.#displayBuilt) return;
|
||||
this.#lastDisplayKey = key;
|
||||
|
||||
@@ -1292,9 +1326,17 @@ export class ToolExecutionComponent extends Container implements NativeScrollbac
|
||||
// draws every dispatched agent as a progress/result line, so tell
|
||||
// `renderCall` to drop its duplicate streaming preview list.
|
||||
context.hasResult = Boolean(this.#result);
|
||||
// Out of the transcript live region: progress rows render static gray
|
||||
// (see task/render.ts).
|
||||
context.frozen = this.#backgroundTaskFrozen;
|
||||
// Settled as history (out of the live region, before any row entered
|
||||
// the tape): progress rows render static gray (see task/render.ts).
|
||||
context.frozen = this.#backgroundTaskFrozenStyled;
|
||||
// Freeze the render clock alongside the latch — and independently the
|
||||
// moment any row commits, closing the window between a commit paint
|
||||
// and the next snapshot where a settings-triggered rebuild could
|
||||
// re-derive elapsed/countdown bytes under committed rows.
|
||||
if (!this.#backgroundTaskFrozen && (this.#liveRegion?.isBlockUncommitted?.(this) ?? true)) {
|
||||
this.#taskRenderNowMs = Date.now();
|
||||
}
|
||||
context.nowMs = this.#taskRenderNowMs;
|
||||
} else if (isEditLikeToolName(this.#toolName)) {
|
||||
context.editMode = this.#editMode;
|
||||
const previews = this.#editDiffPreview;
|
||||
|
||||
@@ -128,6 +128,7 @@ import {
|
||||
} from "../tools/todo";
|
||||
import { vocalizer } from "../tts/vocalizer";
|
||||
import { renderTreeList } from "../tui/tree-list";
|
||||
import { formatStartupChangelogSummary, type StartupChangelogSelection } from "../utils/changelog";
|
||||
import { copyToClipboard } from "../utils/clipboard";
|
||||
import type { EventBus } from "../utils/event-bus";
|
||||
import { getEditorCommand, openInEditor } from "../utils/external-editor";
|
||||
@@ -149,6 +150,7 @@ import {
|
||||
import type { AssistantMessageComponent } from "./components/assistant-message";
|
||||
import type { BashExecutionComponent } from "./components/bash-execution";
|
||||
import { ChatBlock, type ChatBlockHost } from "./components/chat-block";
|
||||
import { CodexResetFireworksController } from "./components/codex-reset-fireworks";
|
||||
import { CustomEditor } from "./components/custom-editor";
|
||||
import { DynamicBorder } from "./components/dynamic-border";
|
||||
import { ErrorBannerComponent } from "./components/error-banner";
|
||||
@@ -556,7 +558,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
#cleanupUnsubscribe?: () => void;
|
||||
#signalTeardown?: SessionTeardown;
|
||||
readonly #version: string;
|
||||
readonly #changelogMarkdown: string | undefined;
|
||||
readonly #startupChangelog: StartupChangelogSelection | undefined;
|
||||
#planModePreviousTools: string[] | undefined;
|
||||
#goalModePreviousTools: string[] | undefined;
|
||||
#vibeModePreviousTools: string[] | undefined;
|
||||
@@ -582,6 +584,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
mcpManager?: MCPManager;
|
||||
readonly #toolUiContextSetter: (uiContext: ExtensionUIContext, hasUI: boolean) => void;
|
||||
|
||||
readonly #codexResetFireworksController: CodexResetFireworksController;
|
||||
readonly #btwController: BtwController;
|
||||
readonly #tanCommandController: TanCommandController;
|
||||
readonly #omfgController: OmfgController;
|
||||
@@ -664,7 +667,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
constructor(
|
||||
session: AgentSession,
|
||||
version: string,
|
||||
changelogMarkdown: string | undefined = undefined,
|
||||
startupChangelog: StartupChangelogSelection | undefined = undefined,
|
||||
setToolUIContext: (uiContext: ExtensionUIContext, hasUI: boolean) => void = () => {},
|
||||
lspServers: LspStartupServerInfo[] | undefined = undefined,
|
||||
mcpManager?: MCPManager,
|
||||
@@ -676,7 +679,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
this.keybindings = KeybindingsManager.inMemory();
|
||||
this.agent = session.agent;
|
||||
this.#version = version;
|
||||
this.#changelogMarkdown = changelogMarkdown;
|
||||
this.#startupChangelog = startupChangelog;
|
||||
this.#toolUiContextSetter = setToolUIContext;
|
||||
this.lspServers = lspServers;
|
||||
this.mcpManager = mcpManager;
|
||||
@@ -752,6 +755,10 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
this.editorContainer.addChild(this.editor);
|
||||
this.statusLine = new StatusLineComponent(session);
|
||||
this.statusLine.setAutoCompactEnabled(session.autoCompactionEnabled);
|
||||
this.#codexResetFireworksController = new CodexResetFireworksController(this);
|
||||
this.statusLine.setCodexResetFireworksHandler(event => {
|
||||
this.#codexResetFireworksController.show(event);
|
||||
});
|
||||
// Vibe worker tok/s aggregator — keeps the status-line render layer off
|
||||
// the heavy vibe/task dependency graph. The director is often idle while
|
||||
// workers stream, so without this the tok/s badge would show a stale
|
||||
@@ -943,19 +950,20 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
}
|
||||
|
||||
// Add changelog if provided
|
||||
if (this.#changelogMarkdown) {
|
||||
if (this.#startupChangelog && settings.get("startup.changelogMode") !== "hidden") {
|
||||
this.ui.addChild(new DynamicBorder());
|
||||
if (settings.get("collapseChangelog")) {
|
||||
const versionMatch = this.#changelogMarkdown.match(/##\s+\[?(\d+\.\d+\.\d+)\]?/);
|
||||
const latestVersion = versionMatch ? versionMatch[1] : this.#version;
|
||||
const condensedText = `Updated to v${latestVersion}. Use ${theme.bold("/changelog")} to view full changelog.`;
|
||||
this.ui.addChild(new Text(condensedText, 1, 0));
|
||||
this.ui.addChild(new Text(theme.bold(theme.fg("accent", "What's New")), 1, 0));
|
||||
this.ui.addChild(new Spacer(1));
|
||||
if (settings.get("startup.changelogMode") === "summary") {
|
||||
const summary = formatStartupChangelogSummary(this.#startupChangelog).replace(
|
||||
/\/changelog(?: full)?/g,
|
||||
command => theme.bold(command),
|
||||
);
|
||||
this.ui.addChild(new Text(summary, 1, 0));
|
||||
} else {
|
||||
this.ui.addChild(new Text(theme.bold(theme.fg("accent", "What's New")), 1, 0));
|
||||
this.ui.addChild(new Spacer(1));
|
||||
this.ui.addChild(new Markdown(this.#changelogMarkdown.trim(), 1, 0, getMarkdownTheme()));
|
||||
this.ui.addChild(new Spacer(1));
|
||||
this.ui.addChild(new Markdown(this.#startupChangelog.markdown?.trim() ?? "", 1, 0, getMarkdownTheme()));
|
||||
}
|
||||
this.ui.addChild(new Spacer(1));
|
||||
this.ui.addChild(new DynamicBorder());
|
||||
}
|
||||
}
|
||||
@@ -3947,6 +3955,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
this.#agentRegistryUnsubscribe = undefined;
|
||||
this.#agentRegistrySubscriptionTarget = undefined;
|
||||
this.#eventController.dispose();
|
||||
this.#codexResetFireworksController.dispose();
|
||||
this.statusLine.dispose();
|
||||
if (this.#resizeHandler) {
|
||||
process.stdout.removeListener("resize", this.#resizeHandler);
|
||||
|
||||
@@ -598,6 +598,23 @@ export class RpcClient {
|
||||
*/
|
||||
async getState(): Promise<RpcSessionState> {
|
||||
const response = await this.#send({ type: "get_state" });
|
||||
const state = this.#getData<RpcSessionState>(response);
|
||||
return {
|
||||
...state,
|
||||
fastModeEnabled: state.fastModeEnabled === true,
|
||||
fastModeActive: state.fastModeActive === true,
|
||||
tokensPerSecond:
|
||||
typeof state.tokensPerSecond === "number" && Number.isFinite(state.tokensPerSecond)
|
||||
? state.tokensPerSecond
|
||||
: null,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Enable or disable fast mode for the active model family.
|
||||
*/
|
||||
async setFastMode(enabled: boolean): Promise<{ enabled: boolean; active: boolean }> {
|
||||
const response = await this.#send({ type: "set_fast_mode", enabled });
|
||||
return this.#getData(response);
|
||||
}
|
||||
|
||||
|
||||
@@ -32,6 +32,7 @@ import { executeAcpBuiltinSlashCommand } from "../../slash-commands/acp-builtins
|
||||
import { buildAvailableSlashCommands } from "../../slash-commands/available-commands";
|
||||
import { defaultLoadModeForToolName } from "../../tools/essential-tools";
|
||||
import type { EventBus } from "../../utils/event-bus";
|
||||
import { calculateTokensPerSecond } from "../../utils/token-rate";
|
||||
import { initializeExtensions } from "../runtime-init";
|
||||
import { isRpcHostToolResult, isRpcHostToolUpdate, RpcHostToolBridge } from "./host-tools";
|
||||
import { isRpcHostUriResult, RpcHostUriBridge } from "./host-uris";
|
||||
@@ -1084,9 +1085,12 @@ export async function runRpcMode(
|
||||
sessionId: session.sessionId,
|
||||
sessionName: session.sessionName,
|
||||
autoCompactionEnabled: session.autoCompactionEnabled,
|
||||
messageCount: session.messages.length,
|
||||
queuedMessageCount: session.queuedMessageCount,
|
||||
todoPhases: session.getTodoPhases(),
|
||||
fastModeEnabled: session.isFastModeEnabled(),
|
||||
tokensPerSecond: calculateTokensPerSecond(session.messages, session.isStreaming),
|
||||
fastModeActive: session.isFastModeActive(),
|
||||
messageCount: session.messages.length,
|
||||
systemPrompt: session.systemPrompt,
|
||||
dumpTools: session.agent.state.tools.map(tool => ({
|
||||
name: tool.name,
|
||||
@@ -1099,6 +1103,17 @@ export async function runRpcMode(
|
||||
return success(id, "get_state", state);
|
||||
}
|
||||
|
||||
case "set_fast_mode": {
|
||||
const supported = session.setFastMode(command.enabled);
|
||||
if (command.enabled && !supported) {
|
||||
return error(id, "set_fast_mode", "Fast mode is unavailable for the current model.");
|
||||
}
|
||||
return success(id, "set_fast_mode", {
|
||||
enabled: session.isFastModeEnabled(),
|
||||
active: session.isFastModeActive(),
|
||||
});
|
||||
}
|
||||
|
||||
case "get_available_commands": {
|
||||
return success(id, "get_available_commands", { commands: await getAvailableCommands() });
|
||||
}
|
||||
|
||||
@@ -39,6 +39,7 @@ export type RpcCommand =
|
||||
|
||||
// State
|
||||
| { id?: string; type: "get_state" }
|
||||
| { id?: string; type: "set_fast_mode"; enabled: boolean }
|
||||
| { id?: string; type: "get_available_commands" }
|
||||
| { id?: string; type: "set_todos"; phases: TodoPhase[] }
|
||||
| { id?: string; type: "set_host_tools"; tools: RpcHostToolDefinition[] }
|
||||
@@ -107,6 +108,9 @@ export interface RpcSessionState {
|
||||
sessionId: string;
|
||||
sessionName?: string;
|
||||
autoCompactionEnabled: boolean;
|
||||
fastModeEnabled: boolean;
|
||||
fastModeActive: boolean;
|
||||
tokensPerSecond: number | null;
|
||||
messageCount: number;
|
||||
queuedMessageCount: number;
|
||||
todoPhases: TodoPhase[];
|
||||
@@ -209,6 +213,13 @@ export type RpcResponse =
|
||||
|
||||
// State
|
||||
| { id?: string; type: "response"; command: "get_state"; success: true; data: RpcSessionState }
|
||||
| {
|
||||
id?: string;
|
||||
type: "response";
|
||||
command: "set_fast_mode";
|
||||
success: true;
|
||||
data: { enabled: boolean; active: boolean };
|
||||
}
|
||||
| {
|
||||
id?: string;
|
||||
type: "response";
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
You are a difficulty classifier for a coding agent. Read the user's request and decide how much reasoning effort the agent should spend on it this turn.
|
||||
|
||||
Reply with exactly one word — one of: `low`, `medium`, `high`, `xhigh`. No punctuation, no explanation, no other text.
|
||||
Reply with exactly one word — one of: `low`, `medium`, `high`, `xhigh`{{#if allowMax}}, `max`{{/if}}. No punctuation, no explanation, no other text.
|
||||
|
||||
Levels:
|
||||
|
||||
@@ -8,5 +8,7 @@ Levels:
|
||||
- `medium` — A localized change that needs some reasoning. A small self-contained feature, a straightforward bug fix in one place, or explaining a moderate piece of code.
|
||||
- `high` — A non-trivial change. Spans multiple files or callers, requires real debugging, a moderate design decision, or a refactor with several moving parts.
|
||||
- `xhigh` — Deep or open-ended. Subtle concurrency or algorithmic problems, cross-system reasoning, ambiguous requirements, large or risky refactors, or hard root-cause debugging.
|
||||
{{#if allowMax}}- `max` — Everything `xhigh` covers, and at least one of: there is no reproduction to work from, the operation is irreversible or can lose data, or a live cutover has to stay correct while it runs. Requires the `xhigh` bar first — difficulty alone is not enough.
|
||||
{{/if}}
|
||||
|
||||
Judge the inherent difficulty of the task, not how politely or verbosely it is phrased. When torn between two levels, choose the lower one.
|
||||
Judge the inherent difficulty of the task, not how politely or verbosely it is phrased. When torn between two levels, choose the lower one{{#if allowMax}} — except between `xhigh` and `max`, where a request that meets the `max` conditions takes `max`{{/if}}.
|
||||
|
||||
@@ -5,7 +5,7 @@ Use this when you need to investigate with many intermediate tool calls (read/gr
|
||||
Rules:
|
||||
- You MUST call `rewind` before yielding after starting a checkpoint.
|
||||
- You NEVER call `checkpoint` while another checkpoint is active.
|
||||
- Not available in subagents.
|
||||
- Disabled by default in subagents. To enable, list `checkpoint` or `rewind` in the agent definition's `tools:` frontmatter (the sister tool is auto-included; requires `checkpoint.enabled` setting).
|
||||
|
||||
Typical flow:
|
||||
1. `checkpoint(goal: …)`
|
||||
|
||||
@@ -62,7 +62,8 @@ import { loadPromptTemplates as loadPromptTemplatesInternal, type PromptTemplate
|
||||
import { applyProviderGlobalsFromSettings } from "./config/provider-globals";
|
||||
import { buildServiceTierByFamily } from "./config/service-tier";
|
||||
import { Settings, type SkillsSettings } from "./config/settings";
|
||||
import { CursorExecHandlers } from "./cursor";
|
||||
import { CursorExecHandlers, type CursorMcpResourceAdapter } from "./cursor";
|
||||
import { createBridgeEditTool, createBridgeGrepFactory } from "./cursor-bridge-tools";
|
||||
import "./discovery";
|
||||
import { initializeWithSettings } from "./discovery";
|
||||
import { disposeAllJuliaKernelSessions, disposeJuliaKernelSessionsByOwner } from "./eval/jl/executor";
|
||||
@@ -2536,6 +2537,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
() => (hasSession ? createSessionMemoryRuntimeContext(session, agentDir, cwd) : undefined),
|
||||
settings,
|
||||
localProtocolOptions,
|
||||
() => (hasSession ? session.getAsyncJobSnapshot() : null),
|
||||
);
|
||||
|
||||
credentialDisabledTarget = extensionRunner;
|
||||
@@ -2606,6 +2608,43 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
for (const tool of toolRegistry.values()) {
|
||||
toolRegistry.set(tool.name, new ExtensionToolWrapper(tool, extensionRunner));
|
||||
}
|
||||
// Cursor's own client owns file edits, so `edit` is not advertised to the
|
||||
// model (commit 8ba0498eb: full-file `write` is used instead). The exec
|
||||
// bridge is a different consumer: the server sends native `pi_edit`
|
||||
// frames regardless of the advertised catalog, and answering them needs
|
||||
// a real tool.
|
||||
//
|
||||
// It must be a `replace`-mode instance. `PiEditExecArgs` carries
|
||||
// `old_text`/`new_text` pairs, which is exactly `replace`'s schema and
|
||||
// nothing else's — under the default `hashline` mode the frame's args do
|
||||
// not match the tool's parameters at all. The registry instance follows
|
||||
// the session's configured mode, so the bridge builds its own.
|
||||
//
|
||||
// The grant is captured HERE, before the Cursor branch below deletes
|
||||
// `edit` from the registry, and independently of the session's provider:
|
||||
// a session that starts on another provider can switch to Cursor later,
|
||||
// and the roster is built once, at session creation. Reading the registry
|
||||
// at frame time would see the switched-to state, not the grant.
|
||||
const editWasGranted = toolRegistry.has("edit");
|
||||
// Built on first use rather than eagerly: a session that never reaches
|
||||
// Cursor never constructs it.
|
||||
let cursorBridgeEditTool: AgentTool | undefined;
|
||||
const getCursorBridgeEditTool = (): AgentTool | undefined => {
|
||||
// Only when the session actually granted `edit`. `createTools` omits
|
||||
// it entirely for a restricted tool set, and the bridge answers native
|
||||
// frames that arrive regardless of the advertised catalog — so
|
||||
// building one unconditionally would hand a read-only agent a
|
||||
// mutating tool it was denied (the issue #5680 escalation).
|
||||
if (!editWasGranted) return undefined;
|
||||
cursorBridgeEditTool ??= createBridgeEditTool(toolSession, extensionRunner);
|
||||
return cursorBridgeEditTool;
|
||||
};
|
||||
// Whether this session granted a file-writing tool. Same capture-early
|
||||
// reasoning, plus `write` may be auto-registered further down as an xdev
|
||||
// transport. The exec bridge answers native `delete` and
|
||||
// resource-download frames that mutate files without running a registry
|
||||
// tool, so it needs the grant as the session actually made it.
|
||||
const cursorCanMutateFiles = editWasGranted || toolRegistry.has("write");
|
||||
if (model?.provider === "cursor") {
|
||||
toolRegistry.delete("edit");
|
||||
builtInRegistryToolNames.delete("edit");
|
||||
@@ -2648,15 +2687,52 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
if (!state) return undefined;
|
||||
return resolveMountedXdevExecutable(state, name);
|
||||
};
|
||||
// Cursor's resource frames ask what THIS client's servers advertise; only
|
||||
// live connections have any. Built once: the advisor bridges answer from
|
||||
// the same connections the primary does.
|
||||
const cursorMcpResources: CursorMcpResourceAdapter | undefined = mcpManager && {
|
||||
serverNames: () => mcpManager.getConnectedServers(),
|
||||
getServerResources: async name => {
|
||||
// The manager registers a server's tools before its background
|
||||
// resource load finishes, so a frame arriving in that window
|
||||
// would read an empty cache and report "advertises nothing".
|
||||
await mcpManager.ensureServerResources(name);
|
||||
return mcpManager.getServerResources(name);
|
||||
},
|
||||
readServerResource: (name, uri) => mcpManager.readServerResource(name, uri),
|
||||
};
|
||||
const cursorExecHandlers = new CursorExecHandlers({
|
||||
cwd,
|
||||
// The session's cwd moves (`/cd`, resume, branch restore) while this
|
||||
// bridge is built once at startup. Path-confining frames — the native
|
||||
// `delete` and a `download_path` resource read — resolve against
|
||||
// whichever of the two they are given, so without the live resolver the
|
||||
// primary would write into the workspace the session has left while
|
||||
// reporting success for the path the server asked about. The advisor
|
||||
// bridge already passes one.
|
||||
getCwd: () => sessionManager.getCwd(),
|
||||
tools: toolRegistry,
|
||||
getExecutableTool: resolveDeviceTool,
|
||||
// `pi_edit` needs the `replace`-mode instance specifically, and the
|
||||
// registry may still hold the session's own `edit` (any mode) when
|
||||
// this session did not start on Cursor.
|
||||
getEditReplaceTool: getCursorBridgeEditTool,
|
||||
getToolContext: () => toolContextStore.getContext(),
|
||||
mcpResources: cursorMcpResources,
|
||||
emitEvent: event => cursorEventEmitter?.(event),
|
||||
getTodoPhases: () => session.getTodoPhases(),
|
||||
setTodoPhases: phases => session.setTodoPhases(phases),
|
||||
persistTodoPhases: phases => sessionManager.appendCustomEntry(USER_TODO_EDIT_CUSTOM_TYPE, { phases }),
|
||||
// `pi_grep` carries its own context width and match cap, which the
|
||||
// shared grep instance fixed at construction cannot express. Gated on
|
||||
// the grant: the factory builds a fresh tool and `executeTool` prefers
|
||||
// it over the registry, so installing it unconditionally would let a
|
||||
// session without `grep` search anyway.
|
||||
createGrepTool: toolRegistry.has("grep") ? createBridgeGrepFactory(toolSession, extensionRunner) : undefined,
|
||||
// The native `delete` and resource-download frames mutate files
|
||||
// without running a registry tool, so this grant is the only thing
|
||||
// standing between a restricted session and a workspace write.
|
||||
allowDirectFileMutation: cursorCanMutateFiles,
|
||||
});
|
||||
|
||||
// Resolve the inline-descriptors setting against the session-start model.
|
||||
@@ -2818,6 +2894,19 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Checkpoint and rewind are a pair: `createTools` auto-includes the sister
|
||||
// tool in the registry, but an explicit `toolNames` list would otherwise
|
||||
// drop it from the ACTIVE set — leaving the agent able to checkpoint but
|
||||
// unable to rewind (or vice versa). Mirror the pairing here. Unlike the
|
||||
// manage_skill/learn mirror above, this is a safety pairing — it applies
|
||||
// to restricted sessions too.
|
||||
if (explicitlyRequestedToolNames) {
|
||||
if (builtInToolNames.includes("checkpoint") && !explicitlyRequestedToolNames.includes("rewind")) {
|
||||
explicitlyRequestedToolNames.push("rewind");
|
||||
} else if (builtInToolNames.includes("rewind") && !explicitlyRequestedToolNames.includes("checkpoint")) {
|
||||
explicitlyRequestedToolNames.push("checkpoint");
|
||||
}
|
||||
}
|
||||
const requestedToolNames = explicitlyRequestedToolNames ?? toolNamesFromRegistry;
|
||||
const normalizedRequested = requestedToolNames.filter(name => toolRegistry.has(name));
|
||||
const defaultInactiveToolNames = new Set(
|
||||
@@ -3137,7 +3226,15 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
advisorToolBuilds.push(BUILTIN_TOOLS[name as keyof typeof BUILTIN_TOOLS](advisorToolSession));
|
||||
}
|
||||
const built = await Promise.all(advisorToolBuilds);
|
||||
const advisorTools: Tool[] = built.filter((tool): tool is Tool => tool != null).map(wrapToolWithMetaNotice);
|
||||
// Wrapped like every registry tool: `ExtensionToolWrapper` is where the
|
||||
// approval mode, per-tool `tools.approval.<tool>` policies and
|
||||
// `autoApprove` are enforced. The advisor's loop and its Cursor exec
|
||||
// bridge both run these instances directly, so a raw one would execute a
|
||||
// `bash`/`write` the user configured as `ask` or `deny`. Meta-notice
|
||||
// first, matching the registry's wrap order.
|
||||
const advisorTools: Tool[] = built
|
||||
.filter((tool): tool is Tool => tool != null)
|
||||
.map(tool => new ExtensionToolWrapper(wrapToolWithMetaNotice(tool), extensionRunner) as Tool);
|
||||
|
||||
const advisorWatchdogPrompts = [...watchdogFiles];
|
||||
if (activeRepoContext) {
|
||||
@@ -3247,6 +3344,20 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
providerPromptCacheKeySource,
|
||||
parentEvalSessionId: options.parentEvalSessionId,
|
||||
advisorTools,
|
||||
// Same per-call `grep` seam the primary bridge gets, built against the
|
||||
// advisor's own tool session so a `pi_grep` frame's context width and
|
||||
// match cap are honored there too.
|
||||
advisorCreateGrepTool: createBridgeGrepFactory(advisorToolSession, extensionRunner),
|
||||
// Same `replace`-mode requirement as the primary bridge; the advisor
|
||||
// path gates it on the advisor's own `edit` grant.
|
||||
advisorCreateEditTool: () => createBridgeEditTool(advisorToolSession, extensionRunner),
|
||||
// The advisor's bridge tools are wrapped for approval, but the wrapper
|
||||
// reads the mode and per-tool policies only from the execute-time
|
||||
// context — the primary bridge passes the same store.
|
||||
advisorGetToolContext: () => toolContextStore.getContext(),
|
||||
// Same live connections the primary bridge reads; an advisor's
|
||||
// resource frame would otherwise report every server as empty.
|
||||
advisorMcpResources: cursorMcpResources,
|
||||
titleSystemPrompt: options.titleSystemPrompt,
|
||||
});
|
||||
hasSession = true;
|
||||
@@ -3274,6 +3385,15 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
throw new Error(`Agent "${resolvedAgentId}" was replaced during session initialization.`);
|
||||
}
|
||||
hasRegistered = true;
|
||||
// MCP notification bridge cleanup — assigned when the bridge is wired below,
|
||||
// invoked from the dispose wrapper AND registered as a postmortem so both
|
||||
// explicit-dispose (SDK embedders that reuse the process across sessions) and
|
||||
// process-exit paths tear the listener down. Nulled after use so the closure
|
||||
// graph (`extensionRunner`, `session`) can be GC'd instead of retained by the
|
||||
// process-global postmortem list.
|
||||
let unsubscribeMcpNotifications: (() => void) | undefined;
|
||||
let unregisterMcpPostmortem: (() => void) | undefined;
|
||||
|
||||
{
|
||||
const originalDispose = session.dispose.bind(session);
|
||||
session.dispose = async () => {
|
||||
@@ -3304,6 +3424,12 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
} finally {
|
||||
unregisterUnlessParked();
|
||||
unsubscribeCredentialDisabled?.();
|
||||
unsubscribeMcpNotifications?.();
|
||||
unregisterMcpPostmortem?.();
|
||||
// Drop refs so the process-global postmortem list doesn't retain
|
||||
// the bridge closure past explicit dispose.
|
||||
unsubscribeMcpNotifications = undefined;
|
||||
unregisterMcpPostmortem = undefined;
|
||||
}
|
||||
};
|
||||
}
|
||||
@@ -3480,19 +3606,24 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
}
|
||||
}
|
||||
|
||||
// Wire MCP manager callbacks to session for reactive tool updates.
|
||||
// Skip when reusing a parent's manager — the parent owns the callbacks.
|
||||
// MCP manager wiring has two ownership models:
|
||||
// * Single-slot callbacks (tools/prompts/resources changed) — exactly one
|
||||
// owner per manager. When reusing a parent's manager (subagent path,
|
||||
// see task/executor.ts), the parent already owns these slots so we
|
||||
// MUST NOT overwrite them. Guarded by `!options.mcpManager`.
|
||||
// * Notification listener — multi-listener by design. Every session with
|
||||
// an MCP manager (fresh OR reused) needs its own bridge to its own
|
||||
// `extensionRunner` so extensions loaded in that session receive frames.
|
||||
// Guarded only by `mcpManager` (see the second `if` below).
|
||||
if (mcpManager && !options.mcpManager) {
|
||||
mcpManager.setOnToolsChanged(tools => {
|
||||
void (async () => {
|
||||
try {
|
||||
await session.refreshMCPTools(tools);
|
||||
} catch (error) {
|
||||
logger.warn("MCP tool refresh failed", {
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
});
|
||||
}
|
||||
})();
|
||||
mcpManager.setOnToolsChanged(async tools => {
|
||||
try {
|
||||
await session.refreshMCPTools(tools);
|
||||
} catch (error) {
|
||||
logger.warn("MCP tool refresh failed", {
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
});
|
||||
}
|
||||
});
|
||||
// Wire prompt refresh → rebuild MCP prompt slash commands
|
||||
mcpManager.setOnPromptsChanged(serverName => {
|
||||
@@ -3525,6 +3656,29 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
});
|
||||
}
|
||||
|
||||
if (mcpManager) {
|
||||
// Bridge server-initiated notifications to this session's extension
|
||||
// handlers. Multi-listener registration: fresh-manager and reused-manager
|
||||
// sessions both install their own listener here, so a subagent's
|
||||
// extensions get frames even though the parent owns the single-slot
|
||||
// tool/prompt/resource callbacks above. MCPManager fires known
|
||||
// list/update refreshes internally, then invokes all registered
|
||||
// listeners with (server, method, params) for every frame (including
|
||||
// server-custom methods). Two-layer buffering protects the startup
|
||||
// race: MCPManager buffers frames received before the first
|
||||
// `addNotificationListener` subscriber (drains here); ExtensionRunner
|
||||
// buffers frames received before `initialize()` and drains them on
|
||||
// init. Both drop-oldest under pressure at cap 100.
|
||||
unsubscribeMcpNotifications = mcpManager.addNotificationListener((server, method, params) => {
|
||||
void extensionRunner.emitMcpNotification({ server, method, params });
|
||||
});
|
||||
// postmortem.register returns a cancel function; capture it so explicit
|
||||
// session.dispose can remove this from the global list (see finally above).
|
||||
unregisterMcpPostmortem = postmortem.register("mcp-notification-listener-cleanup", () =>
|
||||
unsubscribeMcpNotifications?.(),
|
||||
);
|
||||
}
|
||||
|
||||
startDeferredMCPDiscovery?.(session);
|
||||
|
||||
return {
|
||||
|
||||
@@ -1,4 +1,11 @@
|
||||
import type { Agent, AgentMessage, AgentTool, StreamFn, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import type {
|
||||
Agent,
|
||||
AgentMessage,
|
||||
AgentTool,
|
||||
AgentToolContext,
|
||||
StreamFn,
|
||||
ThinkingLevel,
|
||||
} from "@oh-my-pi/pi-agent-core";
|
||||
import type {
|
||||
Context,
|
||||
Effort,
|
||||
@@ -17,6 +24,7 @@ import type { AsyncJob, AsyncJobDeliveryState, AsyncJobManager } from "../async"
|
||||
import type { ModelRegistry } from "../config/model-registry";
|
||||
import type { PromptTemplate } from "../config/prompt-templates";
|
||||
import type { Settings, SkillsSettings } from "../config/settings";
|
||||
import type { CursorMcpResourceAdapter } from "../cursor";
|
||||
import type { RawSseDebugBuffer } from "../debug/raw-sse-buffer";
|
||||
import type { TtsrManager } from "../export/ttsr";
|
||||
import type { LoadedCustomCommand } from "../extensibility/custom-commands";
|
||||
@@ -209,6 +217,35 @@ export interface AgentSessionConfig {
|
||||
providerPromptCacheKeySource?: "explicit" | "fork";
|
||||
/** Full advisor toolset built against an advisor-scoped tool session. */
|
||||
advisorTools?: AgentTool[];
|
||||
/**
|
||||
* Build a `grep` honoring a Cursor `pi_grep` frame's own context width and
|
||||
* match cap, against the advisor-scoped tool session. Without it an advisor
|
||||
* running on Cursor silently drops both fields.
|
||||
*/
|
||||
advisorCreateGrepTool?(options: { context?: number; totalMatchLimit?: number }): AgentTool | undefined;
|
||||
/**
|
||||
* Build the `replace`-mode `edit` a Cursor `pi_edit` frame needs, against the
|
||||
* advisor-scoped tool session. The advisor's ordinary instance follows the
|
||||
* configured `edit.mode` and rejects the frame's `old_text`/`new_text` pairs.
|
||||
*/
|
||||
advisorCreateEditTool?(): AgentTool | undefined;
|
||||
/**
|
||||
* The execute-time context the advisor's bridge tools resolve approval from.
|
||||
*
|
||||
* `ExtensionToolWrapper` reads `tools.approvalMode`, per-tool
|
||||
* `tools.approval.<tool>` policies and `autoApprove` only from this context;
|
||||
* with none it defaults to `yolo` with empty policies, so a bridge tool would
|
||||
* run a native frame the user configured `ask` or `deny` for.
|
||||
*/
|
||||
advisorGetToolContext?: () => AgentToolContext | undefined;
|
||||
/**
|
||||
* The live MCP connections the advisor's Cursor resource frames answer from.
|
||||
*
|
||||
* Advisors share the session's connections and may be granted tools from
|
||||
* those same servers; without this their `list_mcp_resources` reports an
|
||||
* empty catalog and every `read_mcp_resource` a `not_found`.
|
||||
*/
|
||||
advisorMcpResources?: CursorMcpResourceAdapter;
|
||||
/** Preloaded watchdog prompt content for the advisor. */
|
||||
advisorWatchdogPrompt?: string;
|
||||
/** Shared advisor instructions loaded from WATCHDOG.yml. */
|
||||
|
||||
@@ -398,6 +398,9 @@ type SetSessionNameWithTrigger = (
|
||||
trigger?: SessionNameTrigger,
|
||||
) => Promise<boolean>;
|
||||
|
||||
const kPersistedSessionEntryId = Symbol("persistedSessionEntryId");
|
||||
type PersistedAssistantMessage = AssistantMessage & { [kPersistedSessionEntryId]?: string };
|
||||
|
||||
export class AgentSession {
|
||||
readonly agent: Agent;
|
||||
readonly sessionManager: SessionManager;
|
||||
@@ -956,6 +959,7 @@ export class AgentSession {
|
||||
scheduleAgentContinue: options => this.#scheduleAgentContinue(options),
|
||||
waitForSessionMessagePersistence: message => this.#waitForSessionMessagePersistence(message),
|
||||
appendSessionMessage: message => this.#appendSessionMessage(message),
|
||||
persistedAssistantEntryId: message => (message as PersistedAssistantMessage)[kPersistedSessionEntryId],
|
||||
sessionMessageAlreadyPersisted: message => this.#sessionMessageAlreadyPersisted(message),
|
||||
setModelWithProviderSessionReset: model => this.#setModelWithProviderSessionReset(model),
|
||||
resetCurrentResponsesProviderSession: reason => this.#resetCurrentResponsesProviderSession(reason),
|
||||
@@ -1290,6 +1294,10 @@ export class AgentSession {
|
||||
this.#advisors = new SessionAdvisors(advisorsHost, {
|
||||
enabled: this.settings.get("advisor.enabled"),
|
||||
tools: config.advisorTools,
|
||||
createGrepTool: config.advisorCreateGrepTool,
|
||||
createEditTool: config.advisorCreateEditTool,
|
||||
getToolContext: config.advisorGetToolContext,
|
||||
mcpResources: config.advisorMcpResources,
|
||||
watchdogPrompt: config.advisorWatchdogPrompt,
|
||||
sharedInstructions: config.advisorSharedInstructions,
|
||||
contextPrompt: config.advisorContextPrompt,
|
||||
@@ -2065,6 +2073,9 @@ export class AgentSession {
|
||||
const cache = this.#persistedMessageKeys;
|
||||
const wasFresh = cache !== undefined && cache.anchor === this.#persistedMessageKeysAnchor();
|
||||
const entryId = this.sessionManager.appendMessage(message);
|
||||
if (message.role === "assistant") {
|
||||
(message as PersistedAssistantMessage)[kPersistedSessionEntryId] = entryId;
|
||||
}
|
||||
const key = sessionMessagePersistenceKey(message);
|
||||
if (wasFresh && cache && key) {
|
||||
cache.keys.add(key);
|
||||
@@ -5180,6 +5191,7 @@ export class AgentSession {
|
||||
void this.dispose().finally(() => process.exit(0));
|
||||
},
|
||||
getContextUsage: () => this.getContextUsage(),
|
||||
getAsyncJobSnapshot: () => this.getAsyncJobSnapshot(),
|
||||
waitForIdle: () => this.waitForIdle(),
|
||||
newSession: async options => {
|
||||
const success = await this.newSession({ parentSession: options?.parentSession });
|
||||
|
||||
@@ -14,7 +14,7 @@ import type {
|
||||
import { isRecord } from "@oh-my-pi/pi-utils";
|
||||
import { readForeignJsonRecords } from "./foreign-session-jsonl";
|
||||
import type { ForeignSessionInfo, ForeignSessionStore } from "./foreign-session-store";
|
||||
import type { ModelChangeEntry, SessionEntry, SessionMessageEntry } from "./session-entries";
|
||||
import type { CompactionEntry, ModelChangeEntry, SessionEntry, SessionMessageEntry } from "./session-entries";
|
||||
import { SessionManager } from "./session-manager";
|
||||
|
||||
interface CodexThreadRow {
|
||||
@@ -33,6 +33,12 @@ interface CodexIndexRow {
|
||||
updated_at: string;
|
||||
}
|
||||
|
||||
interface CodexCompaction {
|
||||
summary: string;
|
||||
replacementHistory?: Array<Record<string, unknown>>;
|
||||
compactionItem?: Record<string, unknown>;
|
||||
}
|
||||
|
||||
interface ConvertedRecord {
|
||||
message?: UserMessage | AssistantMessage | ToolResultMessage;
|
||||
followingMessage?: ToolResultMessage;
|
||||
@@ -40,6 +46,7 @@ interface ConvertedRecord {
|
||||
rollbackTurns?: number;
|
||||
title?: string;
|
||||
timestamp?: number;
|
||||
compaction?: CodexCompaction;
|
||||
}
|
||||
|
||||
const EMPTY_USAGE: AssistantMessage["usage"] = {
|
||||
@@ -577,6 +584,28 @@ export class CodexSessionStore implements ForeignSessionStore {
|
||||
canonical.toolCalls,
|
||||
toolNames,
|
||||
);
|
||||
} else if (record.type === "compacted") {
|
||||
const sourceSummary = stringField(record.payload, "message")?.trim();
|
||||
const rawReplacementHistory = record.payload.replacement_history;
|
||||
const replacementHistory =
|
||||
Array.isArray(rawReplacementHistory) && rawReplacementHistory.every(isRecord)
|
||||
? rawReplacementHistory
|
||||
: undefined;
|
||||
if (sourceSummary || replacementHistory) {
|
||||
const compactionItem = replacementHistory?.findLast(
|
||||
candidate =>
|
||||
(candidate.type === "compaction" && typeof candidate.encrypted_content === "string") ||
|
||||
candidate.type === "compaction_summary",
|
||||
);
|
||||
item = {
|
||||
compaction: {
|
||||
summary: sourceSummary || "Context compacted by Codex.",
|
||||
replacementHistory,
|
||||
compactionItem,
|
||||
},
|
||||
timestamp,
|
||||
};
|
||||
}
|
||||
}
|
||||
if (!item) continue;
|
||||
if (item.rollbackTurns) rollback(converted, item.rollbackTurns);
|
||||
@@ -609,6 +638,30 @@ export class CodexSessionStore implements ForeignSessionStore {
|
||||
model: `openai-codex/${item.model}`,
|
||||
};
|
||||
entry = modelEntry;
|
||||
} else if (item.compaction) {
|
||||
let preserveData: Record<string, unknown> | undefined;
|
||||
if (item.compaction.replacementHistory) {
|
||||
const remoteCompaction: Record<string, unknown> = {
|
||||
provider: "openai-codex",
|
||||
replacementHistory: item.compaction.replacementHistory,
|
||||
};
|
||||
if (item.compaction.compactionItem) {
|
||||
remoteCompaction.compactionItem = item.compaction.compactionItem;
|
||||
}
|
||||
preserveData = { openaiRemoteCompaction: remoteCompaction };
|
||||
}
|
||||
const compactionEntry: CompactionEntry = {
|
||||
type: "compaction",
|
||||
id,
|
||||
parentId,
|
||||
timestamp,
|
||||
summary: item.compaction.summary,
|
||||
shortSummary: "Imported Codex compaction",
|
||||
firstKeptEntryId: id,
|
||||
tokensBefore: 0,
|
||||
preserveData,
|
||||
};
|
||||
entry = compactionEntry;
|
||||
}
|
||||
if (!entry) continue;
|
||||
manager.ingestReplicatedEntry(entry);
|
||||
|
||||
@@ -3,6 +3,7 @@ import type { Model, ProviderSessionState, ServiceTier, ServiceTierByFamily, Ser
|
||||
import {
|
||||
clearAnthropicFastModeFallback,
|
||||
Effort,
|
||||
isAnthropicFastModeFallbackDisabled,
|
||||
realizesPriorityServiceTier,
|
||||
resolveModelServiceTier,
|
||||
serviceTierFamily,
|
||||
@@ -596,8 +597,8 @@ export class ModelControls {
|
||||
let resolved: Effort | undefined;
|
||||
if (this.#host.magicKeywordEnabled("ultrathink") && containsUltrathink(promptText)) {
|
||||
// The user explicitly asked for maximum thinking; bypass the classifier
|
||||
// (and its xhigh auto ceiling) and jump straight to the highest
|
||||
// supported level for this model.
|
||||
// (and the `providers.autoThinkingMaxEffort` ceiling) and jump straight
|
||||
// to the highest supported level for this model.
|
||||
resolved = clampAutoThinkingEffort(model, Effort.Max);
|
||||
} else {
|
||||
const controller = new AbortController();
|
||||
@@ -666,7 +667,11 @@ export class ModelControls {
|
||||
*/
|
||||
isFastModeActive(): boolean {
|
||||
const model = this.#model;
|
||||
return !!model && realizesPriorityServiceTier(this.effectiveServiceTier(model), model);
|
||||
if (!model || !realizesPriorityServiceTier(this.effectiveServiceTier(model), model)) return false;
|
||||
if (model.provider === "anthropic") {
|
||||
return !isAnthropicFastModeFallbackDisabled(this.#host.providerSessionState, model);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -731,6 +736,9 @@ export class ModelControls {
|
||||
if (this.#serviceTierByFamily[family] === "priority") this.setServiceTierFamily(family, undefined);
|
||||
return true;
|
||||
}
|
||||
if (family === "anthropic" && this.#serviceTierByFamily.anthropic === "priority") {
|
||||
clearAnthropicFastModeFallback(this.#host.providerSessionState);
|
||||
}
|
||||
this.setServiceTierFamily(family, "priority");
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ import {
|
||||
Agent,
|
||||
type AgentMessage,
|
||||
type AgentTool,
|
||||
type AgentToolContext,
|
||||
AppendOnlyContextManager,
|
||||
type CompactionSummaryMessage,
|
||||
countTokens,
|
||||
@@ -16,9 +17,11 @@ import {
|
||||
compactionContextTokens,
|
||||
createCompactionSummaryMessage,
|
||||
estimateTokens,
|
||||
NativeCompactionError,
|
||||
prepareCompaction,
|
||||
type SessionMessageEntry,
|
||||
shouldCompact,
|
||||
shouldUseProviderNativeCompaction,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction";
|
||||
import type {
|
||||
AssistantMessage,
|
||||
@@ -67,7 +70,8 @@ import {
|
||||
import { MODEL_ROLES } from "../config/model-roles";
|
||||
import { serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier";
|
||||
import type { Settings } from "../config/settings";
|
||||
import { CursorExecHandlers } from "../cursor";
|
||||
import { CursorExecHandlers, type CursorMcpResourceAdapter } from "../cursor";
|
||||
import { bridgeToolMap } from "../cursor-bridge-tools";
|
||||
import { estimateToolSchemaTokens } from "../modes/utils/context-usage";
|
||||
import type { PlanModeState } from "../plan-mode/state";
|
||||
import advisorSystemPrompt from "../prompts/advisor/system.md" with { type: "text" };
|
||||
@@ -177,6 +181,36 @@ interface AdvisorRuntimeDescriptor {
|
||||
export interface SessionAdvisorsOptions {
|
||||
enabled: boolean;
|
||||
tools?: AgentTool[];
|
||||
/**
|
||||
* Build a `grep` honoring a Cursor `pi_grep` frame's own context width and
|
||||
* match cap. The advisor's tools are fixed instances carrying session
|
||||
* defaults, so without this an advisor running against Cursor silently
|
||||
* drops both fields — the same gap the primary bridge closes.
|
||||
*/
|
||||
createGrepTool?(options: { context?: number; totalMatchLimit?: number }): AgentTool | undefined;
|
||||
/**
|
||||
* Build the `replace`-mode `edit` a Cursor `pi_edit` frame needs. The
|
||||
* advisor's own instance follows the configured `edit.mode` (`hashline` by
|
||||
* default), whose schema the frame's `old_text`/`new_text` pairs do not
|
||||
* match, so without this every native advisor edit fails validation.
|
||||
*/
|
||||
createEditTool?(): AgentTool | undefined;
|
||||
/**
|
||||
* The execute-time context the bridge's tools resolve approval from.
|
||||
*
|
||||
* `ExtensionToolWrapper` reads the approval mode, per-tool policies and
|
||||
* `autoApprove` only from here; with none it falls back to `yolo` and empty
|
||||
* policies, so a native frame would run past a configured `ask` or `deny`.
|
||||
*/
|
||||
getToolContext?: () => AgentToolContext | undefined;
|
||||
/**
|
||||
* The live MCP connections Cursor's resource frames answer from.
|
||||
*
|
||||
* Advisors share the session's connections and may hold tools from those
|
||||
* same servers, so without this their frames report that every server
|
||||
* advertises nothing.
|
||||
*/
|
||||
mcpResources?: CursorMcpResourceAdapter;
|
||||
watchdogPrompt?: string;
|
||||
sharedInstructions?: string;
|
||||
contextPrompt?: string;
|
||||
@@ -250,6 +284,10 @@ export class SessionAdvisors {
|
||||
readonly #host: SessionAdvisorsHost;
|
||||
#advisorEnabled: boolean;
|
||||
#advisorTools: AgentTool[] | undefined;
|
||||
#advisorCreateGrepTool: SessionAdvisorsOptions["createGrepTool"];
|
||||
#advisorCreateEditTool: SessionAdvisorsOptions["createEditTool"];
|
||||
#advisorGetToolContext: SessionAdvisorsOptions["getToolContext"];
|
||||
#advisorMcpResources: SessionAdvisorsOptions["mcpResources"];
|
||||
#advisorWatchdogPrompt: string | undefined;
|
||||
#advisorSharedInstructions: string | undefined;
|
||||
#advisorContextPrompt: string | undefined;
|
||||
@@ -272,6 +310,10 @@ export class SessionAdvisors {
|
||||
this.#host = host;
|
||||
this.#advisorEnabled = options.enabled;
|
||||
this.#advisorTools = options.tools;
|
||||
this.#advisorCreateGrepTool = options.createGrepTool;
|
||||
this.#advisorCreateEditTool = options.createEditTool;
|
||||
this.#advisorGetToolContext = options.getToolContext;
|
||||
this.#advisorMcpResources = options.mcpResources;
|
||||
this.#advisorWatchdogPrompt = options.watchdogPrompt;
|
||||
this.#advisorSharedInstructions = options.sharedInstructions;
|
||||
this.#advisorContextPrompt = options.contextPrompt;
|
||||
@@ -718,11 +760,29 @@ export class SessionAdvisors {
|
||||
// to delete workspace files it was never granted (issue #5680 review).
|
||||
const advisorCanMutateFiles = advisorToolMap.has("write") || advisorToolMap.has("edit");
|
||||
if (advisorCanMutateFiles) availableAdvisorToolNames.add("delete");
|
||||
// `pi_edit` speaks `replace`'s `old_text`/`new_text` schema, which the
|
||||
// advisor's ordinary `EditTool` (built at the session's configured
|
||||
// `edit.mode`, `hashline` by default) does not accept. The bridge map
|
||||
// swaps in a `replace` instance for the exec channel only — the
|
||||
// advisor's own loop keeps the tool it was given — and only when
|
||||
// `edit` was actually granted.
|
||||
const advisorCursorExecHandlers = new CursorExecHandlers({
|
||||
cwd: this.#host.sessionManager.getCwd(),
|
||||
getCwd: () => this.#host.sessionManager.getCwd(),
|
||||
tools: advisorToolMap,
|
||||
allowNativeDelete: advisorCanMutateFiles,
|
||||
tools: bridgeToolMap(advisorToolMap, this.#advisorCreateEditTool),
|
||||
// Approval mode, per-tool policies and `autoApprove` live only on
|
||||
// this context; without it every bridge tool resolves as `yolo`.
|
||||
getToolContext: this.#advisorGetToolContext,
|
||||
allowDirectFileMutation: advisorCanMutateFiles,
|
||||
// Gated on the advisor's own grant: the factory builds a fresh
|
||||
// tool, so handing it over unconditionally would give a roster
|
||||
// without `grep` a search tool it was denied.
|
||||
createGrepTool: advisorToolMap.has("grep") ? this.#advisorCreateGrepTool : undefined,
|
||||
// Advisors share the session's live MCP connections, so their
|
||||
// resource frames answer from the same catalog the primary sees.
|
||||
// Not gated on a tool grant: reading what a server advertises is
|
||||
// not the same permission as calling one of its tools.
|
||||
mcpResources: this.#advisorMcpResources,
|
||||
});
|
||||
const baseAdvisorStreamFn = this.#advisorStreamFn ?? streamSimple;
|
||||
const advisorStreamFn: StreamFn = (requestModel, context, options) =>
|
||||
@@ -816,6 +876,7 @@ export class SessionAdvisors {
|
||||
maintainContext: (incomingTokens, signal) =>
|
||||
this.#maintainAdvisorContext(advisorRef, incomingTokens, signal),
|
||||
obfuscator: this.#host.obfuscator,
|
||||
getModelIdentity: () => formatModelString(advisorRef.agent.state.model),
|
||||
beginAdvisorUpdate: () => advisorRef.emissionGuard.beginUpdate(),
|
||||
onTurnError: (error, failedMessages, signal) =>
|
||||
this.#recoverAdvisorTurn(advisorRef, error, failedMessages, signal),
|
||||
@@ -1341,6 +1402,7 @@ export class SessionAdvisors {
|
||||
|
||||
let compactResult: CompactionResult | undefined;
|
||||
let lastError: unknown;
|
||||
let nativeCompactionFailure: { error: NativeCompactionError; provider: string } | undefined;
|
||||
// Instrument the advisor's overflow-compaction one-shot like the primary
|
||||
// compaction path so the advisor model's maintenance call also emits spans.
|
||||
const telemetry = resolveTelemetry(agent.telemetry, advisorProviderSessionId);
|
||||
@@ -1354,6 +1416,14 @@ export class SessionAdvisors {
|
||||
for (const candidate of candidates) {
|
||||
const apiKey = await this.#host.modelRegistry.getApiKey(candidate, advisorProviderSessionId, { signal });
|
||||
if (!apiKey) continue;
|
||||
if (
|
||||
nativeCompactionFailure &&
|
||||
(candidate.provider !== nativeCompactionFailure.provider ||
|
||||
!shouldUseProviderNativeCompaction(candidate, compactionSettings))
|
||||
) {
|
||||
throw nativeCompactionFailure.error;
|
||||
}
|
||||
|
||||
// The advisor overflow-compaction one-shot bypasses the advisor `Agent`,
|
||||
// so its installed metadata resolver never runs. Emit the same
|
||||
// `metadata.user_id` identity here (resolved per candidate provider,
|
||||
@@ -1385,10 +1455,18 @@ export class SessionAdvisors {
|
||||
break;
|
||||
} catch (error) {
|
||||
if (signal.aborted) throw error;
|
||||
const id = AIError.classify(error, candidate.api);
|
||||
if (error instanceof NativeCompactionError && !AIError.is(id, AIError.Flag.AuthFailed)) {
|
||||
nativeCompactionFailure ??= { error, provider: candidate.provider };
|
||||
lastError = nativeCompactionFailure.error;
|
||||
continue;
|
||||
}
|
||||
lastError = error;
|
||||
}
|
||||
}
|
||||
|
||||
if (!compactResult && nativeCompactionFailure) throw nativeCompactionFailure.error;
|
||||
|
||||
if (!compactResult) {
|
||||
logger.warn("Advisor compaction failed, falling back to re-prime", { error: String(lastError) });
|
||||
return true;
|
||||
|
||||
@@ -26,6 +26,7 @@ import {
|
||||
DEFAULT_SHAKE_CONFIG,
|
||||
effectiveReserveTokens,
|
||||
estimateTokens,
|
||||
NativeCompactionError,
|
||||
prepareCompaction,
|
||||
resolveBudgetReserveTokens,
|
||||
resolveThresholdTokens,
|
||||
@@ -34,6 +35,7 @@ import {
|
||||
type SummaryOptions,
|
||||
shouldCompact,
|
||||
shouldUseOpenAiRemoteCompaction,
|
||||
shouldUseProviderNativeCompaction,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction";
|
||||
import {
|
||||
DEFAULT_PRUNE_CONFIG,
|
||||
@@ -1458,10 +1460,18 @@ export class SessionMaintenance {
|
||||
const candidates =
|
||||
precomputedCandidates ?? this.#getCompactionModelCandidates(this.#host.modelRegistry.getAvailable());
|
||||
const telemetry = resolveTelemetry(this.#host.agent.telemetry, this.#host.sessionId());
|
||||
let nativeCompactionFailure: { error: NativeCompactionError; provider: string } | undefined;
|
||||
|
||||
for (const candidate of candidates) {
|
||||
const apiKey = await this.#host.modelRegistry.getApiKey(candidate, this.#host.sessionId());
|
||||
if (!apiKey) continue;
|
||||
if (
|
||||
nativeCompactionFailure &&
|
||||
(candidate.provider !== nativeCompactionFailure.provider ||
|
||||
!shouldUseProviderNativeCompaction(candidate, preparation.settings))
|
||||
) {
|
||||
throw nativeCompactionFailure.error;
|
||||
}
|
||||
|
||||
try {
|
||||
return await compact(
|
||||
@@ -1499,12 +1509,17 @@ export class SessionMaintenance {
|
||||
},
|
||||
);
|
||||
} catch (error) {
|
||||
if (!AIError.is(AIError.classify(error, candidate.api), AIError.Flag.AuthFailed)) {
|
||||
throw error;
|
||||
const id = AIError.classify(error instanceof NativeCompactionError ? error.cause : error, candidate.api);
|
||||
if (AIError.is(id, AIError.Flag.AuthFailed)) continue;
|
||||
if (error instanceof NativeCompactionError) {
|
||||
nativeCompactionFailure ??= { error, provider: candidate.provider };
|
||||
continue;
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
if (nativeCompactionFailure) throw nativeCompactionFailure.error;
|
||||
throw this.#buildCompactionAuthError();
|
||||
}
|
||||
|
||||
@@ -2477,6 +2492,7 @@ export class SessionMaintenance {
|
||||
const telemetry = resolveTelemetry(this.#host.agent.telemetry, this.#host.sessionId());
|
||||
let compactResult: CompactionResult | undefined;
|
||||
let lastError: unknown;
|
||||
let nativeCompactionFailure: { error: NativeCompactionError; provider: string } | undefined;
|
||||
codexCompaction = createCodexCompactionContext({
|
||||
trigger: "auto",
|
||||
reason: "context_limit",
|
||||
@@ -2490,6 +2506,13 @@ export class SessionMaintenance {
|
||||
const hasMoreCandidates = candidateIndex < candidates.length - 1;
|
||||
const apiKey = await this.#host.modelRegistry.getApiKey(candidate, this.#host.sessionId());
|
||||
if (!apiKey) continue;
|
||||
if (
|
||||
nativeCompactionFailure &&
|
||||
(candidate.provider !== nativeCompactionFailure.provider ||
|
||||
!shouldUseProviderNativeCompaction(candidate, preparation.settings))
|
||||
) {
|
||||
throw nativeCompactionFailure.error;
|
||||
}
|
||||
|
||||
let attempt = 0;
|
||||
while (true) {
|
||||
@@ -2527,22 +2550,33 @@ export class SessionMaintenance {
|
||||
}
|
||||
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
const id = AIError.classify(error, candidate.api);
|
||||
const id = AIError.classify(
|
||||
error instanceof NativeCompactionError ? error.cause : error,
|
||||
candidate.api,
|
||||
);
|
||||
if (AIError.is(id, AIError.Flag.AuthFailed)) {
|
||||
lastError = this.#buildCompactionAuthError();
|
||||
if (!nativeCompactionFailure) lastError = this.#buildCompactionAuthError();
|
||||
break;
|
||||
}
|
||||
if (AIError.is(id, AIError.Flag.Timeout)) {
|
||||
const nativeFailure = error instanceof NativeCompactionError;
|
||||
logger.warn(
|
||||
hasMoreCandidates
|
||||
? "Auto-compaction summarization timed out, trying next model"
|
||||
: "Auto-compaction summarization timed out, not retrying same model",
|
||||
nativeFailure
|
||||
? "Provider-native auto-compaction timed out, preserving native failure"
|
||||
: hasMoreCandidates
|
||||
? "Auto-compaction summarization timed out, trying next model"
|
||||
: "Auto-compaction summarization timed out, not retrying same model",
|
||||
{
|
||||
error: message,
|
||||
model: `${candidate.provider}/${candidate.id}`,
|
||||
},
|
||||
);
|
||||
lastError = error;
|
||||
if (nativeFailure) {
|
||||
nativeCompactionFailure ??= { error, provider: candidate.provider };
|
||||
lastError = nativeCompactionFailure.error;
|
||||
} else {
|
||||
lastError = error;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -2554,7 +2588,12 @@ export class SessionMaintenance {
|
||||
AIError.is(id, AIError.Flag.Transient) ||
|
||||
AIError.is(id, AIError.Flag.UsageLimit));
|
||||
if (!shouldRetry) {
|
||||
lastError = error;
|
||||
if (error instanceof NativeCompactionError) {
|
||||
nativeCompactionFailure ??= { error, provider: candidate.provider };
|
||||
lastError = nativeCompactionFailure.error;
|
||||
} else {
|
||||
lastError = error;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -2564,6 +2603,11 @@ export class SessionMaintenance {
|
||||
// If retry delay is too long (>30s), try next candidate instead of waiting
|
||||
const maxAcceptableDelayMs = 30_000;
|
||||
if (delayMs > maxAcceptableDelayMs && hasMoreCandidates) {
|
||||
if (error instanceof NativeCompactionError) {
|
||||
nativeCompactionFailure ??= { error, provider: candidate.provider };
|
||||
lastError = nativeCompactionFailure.error;
|
||||
break;
|
||||
}
|
||||
logger.warn("Auto-compaction retry delay too long, trying next model", {
|
||||
delayMs,
|
||||
retryAfterMs,
|
||||
|
||||
@@ -58,6 +58,7 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn =
|
||||
textVerbosity: streamOptions?.textVerbosity ?? textVerbosity,
|
||||
streamFirstEventTimeoutMs: streamOptions?.streamFirstEventTimeoutMs ?? streamFirstEventTimeoutMs,
|
||||
streamIdleTimeoutMs: streamOptions?.streamIdleTimeoutMs ?? streamIdleTimeoutMs,
|
||||
maxRetryDelayMs: streamOptions?.maxRetryDelayMs ?? settings.get("retry.maxDelayMs"),
|
||||
maxInFlightRequests: validateProviderMaxInFlightRequests(
|
||||
streamOptions?.maxInFlightRequests ?? settings.get("providers.maxInFlightRequests"),
|
||||
),
|
||||
|
||||
@@ -117,6 +117,7 @@ export interface TurnRecoveryHost {
|
||||
scheduleAgentContinue(options: { delayMs?: number; generation?: number; onError?: (error: unknown) => void }): void;
|
||||
waitForSessionMessagePersistence(message: AssistantMessage): Promise<void>;
|
||||
appendSessionMessage(message: AssistantMessage): void;
|
||||
persistedAssistantEntryId(message: AssistantMessage): string | undefined;
|
||||
sessionMessageAlreadyPersisted(message: AssistantMessage): boolean;
|
||||
setModelWithProviderSessionReset(model: Model): Promise<void>;
|
||||
resetCurrentResponsesProviderSession(reason: string): void;
|
||||
@@ -770,16 +771,24 @@ export class TurnRecovery {
|
||||
discardAssistantTurn(assistantMessage: AssistantMessage): void {
|
||||
this.removeAssistantMessageFromActiveContext(assistantMessage);
|
||||
|
||||
const branchEntry = this.#host.sessionManager
|
||||
.getBranch()
|
||||
.slice()
|
||||
.reverse()
|
||||
.find(
|
||||
entry =>
|
||||
entry.type === "message" &&
|
||||
entry.message.role === "assistant" &&
|
||||
this.#isSameAssistantMessage(entry.message as AssistantMessage, assistantMessage),
|
||||
);
|
||||
const branch = this.#host.sessionManager.getBranch();
|
||||
const persistedEntryId = this.#host.persistedAssistantEntryId(assistantMessage);
|
||||
const branchEntry =
|
||||
(persistedEntryId === undefined
|
||||
? undefined
|
||||
: branch.find(
|
||||
entry =>
|
||||
entry.id === persistedEntryId && entry.type === "message" && entry.message.role === "assistant",
|
||||
)) ??
|
||||
branch
|
||||
.slice()
|
||||
.reverse()
|
||||
.find(
|
||||
entry =>
|
||||
entry.type === "message" &&
|
||||
entry.message.role === "assistant" &&
|
||||
this.#isSameAssistantMessage(entry.message as AssistantMessage, assistantMessage),
|
||||
);
|
||||
if (!branchEntry) {
|
||||
return;
|
||||
}
|
||||
@@ -798,7 +807,9 @@ export class TurnRecovery {
|
||||
(left.timestamp === right.timestamp &&
|
||||
left.provider === right.provider &&
|
||||
left.model === right.model &&
|
||||
left.stopReason === right.stopReason)
|
||||
left.stopReason === right.stopReason &&
|
||||
left.errorMessage === right.errorMessage &&
|
||||
Bun.hash(JSON.stringify(left.content)) === Bun.hash(JSON.stringify(right.content)))
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -49,6 +49,13 @@ interface TaskRenderContext {
|
||||
* commit-eligible rows do not repaint after entering native scrollback.
|
||||
*/
|
||||
frozen?: boolean;
|
||||
/**
|
||||
* Wall clock for time-derived rows (current-tool elapsed, retry countdown).
|
||||
* The component freezes it once the block settles or any of its rows enter
|
||||
* native scrollback, so identical-input rebuilds stay byte-identical with
|
||||
* committed history. Absent: render with the live clock.
|
||||
*/
|
||||
nowMs?: number;
|
||||
}
|
||||
type TaskRenderOptions = RenderResultOptions & { renderContext?: TaskRenderContext };
|
||||
|
||||
@@ -888,6 +895,7 @@ function renderAgentProgress(
|
||||
frozen = false,
|
||||
seenNestedTasks?: WeakSet<object>,
|
||||
nestedDepth = 0,
|
||||
nowMs = Date.now(),
|
||||
): string[] {
|
||||
const lines: string[] = [];
|
||||
|
||||
@@ -964,7 +972,7 @@ function renderAgentProgress(
|
||||
toolLine += `: ${theme.fg("dim", previewLine(sanitizeText(toolDetail), 40))}`;
|
||||
}
|
||||
if (progress.currentToolStartMs) {
|
||||
const elapsed = Date.now() - progress.currentToolStartMs;
|
||||
const elapsed = nowMs - progress.currentToolStartMs;
|
||||
if (elapsed > 5000) {
|
||||
toolLine += `${theme.sep.dot}${theme.fg("warning", formatDuration(elapsed))}`;
|
||||
}
|
||||
@@ -986,7 +994,7 @@ function renderAgentProgress(
|
||||
// long until the next attempt. Without this, the parent UI would just
|
||||
// keep spinning while a child sleeps on a 3-hour provider rate-limit.
|
||||
if (progress.retryState && progress.status === "running") {
|
||||
const remainingMs = Math.max(0, progress.retryState.startedAtMs + progress.retryState.delayMs - Date.now());
|
||||
const remainingMs = Math.max(0, progress.retryState.startedAtMs + progress.retryState.delayMs - nowMs);
|
||||
const waitLabel = remainingMs > 0 ? `in ${formatDuration(remainingMs)}` : "now";
|
||||
const summary =
|
||||
`retrying ${progress.retryState.attempt}/${progress.retryState.maxAttempts} ${waitLabel}: ` +
|
||||
@@ -1079,6 +1087,7 @@ function renderAgentProgress(
|
||||
frozen,
|
||||
seenNestedTasks,
|
||||
nestedDepth,
|
||||
nowMs,
|
||||
);
|
||||
for (const line of nestedLines) {
|
||||
lines.push(`${continuePrefix}${line}`);
|
||||
@@ -1560,6 +1569,7 @@ export function renderResult(
|
||||
return framedBlock(theme, width => {
|
||||
const { expanded, isPartial, spinnerFrame } = options;
|
||||
const frozen = options.renderContext?.frozen === true;
|
||||
const nowMs = options.renderContext?.nowMs ?? Date.now();
|
||||
const lines: string[] = [];
|
||||
|
||||
// Result rows win once any exist; progress rows for spawns without a
|
||||
@@ -1577,7 +1587,9 @@ export function renderResult(
|
||||
lines.push(formatHiddenProgressLine(ordered.slice(0, ordered.length - visible.length), theme));
|
||||
}
|
||||
for (const progress of visible) {
|
||||
lines.push(...renderAgentProgress(progress, "", " ", expanded, theme, spinnerFrame, frozen));
|
||||
lines.push(
|
||||
...renderAgentProgress(progress, "", " ", expanded, theme, spinnerFrame, frozen, undefined, 0, nowMs),
|
||||
);
|
||||
}
|
||||
} else if (details.results && details.results.length > 0) {
|
||||
const ordered = orderResultsForDisplay(details.results);
|
||||
@@ -1602,7 +1614,9 @@ export function renderResult(
|
||||
)
|
||||
: [];
|
||||
for (const progress of supplementalProgress) {
|
||||
lines.push(...renderAgentProgress(progress, "", " ", expanded, theme, spinnerFrame, frozen));
|
||||
lines.push(
|
||||
...renderAgentProgress(progress, "", " ", expanded, theme, spinnerFrame, frozen, undefined, 0, nowMs),
|
||||
);
|
||||
}
|
||||
|
||||
const summaryParts: string[] = [];
|
||||
@@ -1742,6 +1756,7 @@ function renderNestedTaskTree(
|
||||
frozen = false,
|
||||
seen: WeakSet<object> = new WeakSet<object>(),
|
||||
depth = 0,
|
||||
nowMs = Date.now(),
|
||||
): string[] {
|
||||
const lines: string[] = [];
|
||||
for (const details of detailsList) {
|
||||
@@ -1788,6 +1803,7 @@ function renderNestedTaskTree(
|
||||
frozen,
|
||||
seen,
|
||||
depth + 1,
|
||||
nowMs,
|
||||
),
|
||||
);
|
||||
});
|
||||
|
||||
@@ -191,7 +191,7 @@ export interface ConfiguredThinkingLevelMetadata {
|
||||
const AUTO_THINKING_METADATA: ConfiguredThinkingLevelMetadata = {
|
||||
value: AUTO_THINKING,
|
||||
label: "auto",
|
||||
description: "Auto-detect per prompt (low–xhigh)",
|
||||
description: "Auto-detect per prompt",
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -236,6 +236,12 @@ export function parseCliThinkingLevel(value: string | null | undefined): Configu
|
||||
* above Low (falling back to the full supported set only when the model maxes
|
||||
* out below Low). Within that pool the request snaps to the highest level not
|
||||
* exceeding it, or the pool minimum when the request is below the pool.
|
||||
* `ceiling` bounds the pool from above, so a policy ceiling survives the model
|
||||
* clamp: a sparse ladder such as `["max"]` must not snap an `xhigh` request up
|
||||
* to `max`. The Low floor is resolved against the model's own ladder *before*
|
||||
* the ceiling applies — a ceiling that hides every tier at or above Low means
|
||||
* there is nothing legal to pick (`undefined`), not a licence to fall through
|
||||
* to a sub-Low tier the model happens to expose.
|
||||
*
|
||||
* Returns `undefined` for reasoning-capable models without a controllable
|
||||
* effort surface (`thinking.efforts` empty — e.g. devin-agent models, where
|
||||
@@ -244,12 +250,19 @@ export function parseCliThinkingLevel(value: string | null | undefined): Configu
|
||||
* forward a concrete effort that would then trip {@link requireSupportedEffort}
|
||||
* downstream.
|
||||
*/
|
||||
export function clampAutoThinkingEffort(model: Model | undefined, effort: Effort): Effort | undefined {
|
||||
export function clampAutoThinkingEffort(
|
||||
model: Model | undefined,
|
||||
effort: Effort,
|
||||
ceiling: Effort = Effort.Max,
|
||||
): Effort | undefined {
|
||||
const supported = model ? getSupportedEfforts(model) : THINKING_EFFORTS;
|
||||
if (supported.length === 0) return undefined;
|
||||
const lowIndex = THINKING_EFFORTS.indexOf(Effort.Low);
|
||||
const eligible = supported.filter(level => THINKING_EFFORTS.indexOf(level) >= lowIndex);
|
||||
const pool = eligible.length > 0 ? eligible : supported;
|
||||
const ceilingIndex = THINKING_EFFORTS.indexOf(ceiling);
|
||||
const atOrAboveLow = supported.filter(level => THINKING_EFFORTS.indexOf(level) >= lowIndex);
|
||||
const floored = atOrAboveLow.length > 0 ? atOrAboveLow : supported;
|
||||
const pool = floored.filter(level => THINKING_EFFORTS.indexOf(level) <= ceilingIndex);
|
||||
if (pool.length === 0) return undefined;
|
||||
const requestedIndex = THINKING_EFFORTS.indexOf(effort);
|
||||
let chosen = pool[0];
|
||||
for (const candidate of pool) {
|
||||
@@ -351,14 +364,20 @@ export function modelSupportsEffortCeiling(model: Model, ceiling: Effort): boole
|
||||
|
||||
/**
|
||||
* The provisional concrete level shown while `auto` is configured but before a
|
||||
* turn has been classified. Prefers the model's `defaultLevel`, otherwise High,
|
||||
* clamped into the auto range. Auto never provisions {@link Effort.Max} (the
|
||||
* classifier ceiling is XHigh; only an explicit user request reaches Max), so a
|
||||
* `defaultLevel` of `max` is capped at XHigh before clamping. Returns
|
||||
* `undefined` for non-reasoning models.
|
||||
* turn has been classified, and the fallback when classification fails. Prefers
|
||||
* the model's `defaultLevel`, otherwise High, clamped into the auto range.
|
||||
*
|
||||
* Deliberately stays below {@link Effort.Max}: the placeholder must not bill the
|
||||
* top tier for a turn nobody classified, so XHigh is passed as a hard ceiling
|
||||
* rather than only capping the preferred level — otherwise a sparse `["max"]`
|
||||
* ladder would snap straight back up. A model whose ladder offers nothing at or
|
||||
* below XHigh therefore has no provisional level, and `auto` leaves the current
|
||||
* one in place. Classification itself may still resolve Max on models that
|
||||
* expose the tier when the user opts in. Returns `undefined` for non-reasoning
|
||||
* models.
|
||||
*/
|
||||
export function resolveProvisionalAutoLevel(model: Model | undefined): Effort | undefined {
|
||||
if (!model?.reasoning) return undefined;
|
||||
const preferred = model.thinking?.defaultLevel ?? Effort.High;
|
||||
return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred);
|
||||
return clampAutoThinkingEffort(model, preferred === Effort.Max ? Effort.XHigh : preferred, Effort.XHigh);
|
||||
}
|
||||
|
||||
@@ -95,6 +95,11 @@ function resolveBrowserKind(params: BrowserParams, session: ToolSession): Browse
|
||||
const exe = resolveToCwd(app.path, session.cwd);
|
||||
return { kind: "spawned", path: exe };
|
||||
}
|
||||
// A configured endpoint is a default, not an override: explicit app options win.
|
||||
const configuredCdpUrl = (session.settings.get("browser.cdpUrl") as string | undefined)?.trim();
|
||||
if (configuredCdpUrl) {
|
||||
return { kind: "connected", cdpUrl: configuredCdpUrl.replace(/\/+$/, "") };
|
||||
}
|
||||
const cmuxKind = resolveCmuxKind({
|
||||
settingEnabled: session.settings.get("browser.cmux") as boolean | undefined,
|
||||
});
|
||||
|
||||
@@ -50,11 +50,6 @@ export interface RewindToolDetails {
|
||||
meta?: OutputMeta;
|
||||
}
|
||||
|
||||
function isTopLevelSession(session: ToolSession): boolean {
|
||||
const depth = session.taskDepth;
|
||||
return depth === undefined || depth === 0;
|
||||
}
|
||||
|
||||
export class CheckpointTool implements AgentTool<typeof checkpointSchema, CheckpointToolDetails> {
|
||||
readonly name = "checkpoint";
|
||||
readonly approval = "read" as const;
|
||||
@@ -71,7 +66,6 @@ export class CheckpointTool implements AgentTool<typeof checkpointSchema, Checkp
|
||||
}
|
||||
|
||||
static createIf(session: ToolSession): CheckpointTool | null {
|
||||
if (!isTopLevelSession(session)) return null;
|
||||
return new CheckpointTool(session);
|
||||
}
|
||||
|
||||
@@ -82,9 +76,6 @@ export class CheckpointTool implements AgentTool<typeof checkpointSchema, Checkp
|
||||
_onUpdate?: AgentToolUpdateCallback<CheckpointToolDetails>,
|
||||
_context?: AgentToolContext,
|
||||
): Promise<AgentToolResult<CheckpointToolDetails>> {
|
||||
if (!isTopLevelSession(this.session)) {
|
||||
throw new ToolError("Checkpoint not available in subagents.");
|
||||
}
|
||||
if (this.session.getCheckpointState?.()) {
|
||||
throw new ToolError("Checkpoint already active.");
|
||||
}
|
||||
@@ -117,7 +108,6 @@ export class RewindTool implements AgentTool<typeof rewindSchema, RewindToolDeta
|
||||
}
|
||||
|
||||
static createIf(session: ToolSession): RewindTool | null {
|
||||
if (!isTopLevelSession(session)) return null;
|
||||
return new RewindTool(session);
|
||||
}
|
||||
|
||||
@@ -128,9 +118,6 @@ export class RewindTool implements AgentTool<typeof rewindSchema, RewindToolDeta
|
||||
_onUpdate?: AgentToolUpdateCallback<RewindToolDetails>,
|
||||
_context?: AgentToolContext,
|
||||
): Promise<AgentToolResult<RewindToolDetails>> {
|
||||
if (!isTopLevelSession(this.session)) {
|
||||
throw new ToolError("Checkpoint not available in subagents.");
|
||||
}
|
||||
if (!this.session.getCheckpointState?.()) {
|
||||
if (this.session.getLastCompletedRewind?.()) {
|
||||
throw new ToolError(
|
||||
|
||||
@@ -884,6 +884,22 @@ export interface GrepToolDetails {
|
||||
|
||||
type SearchParams = typeof searchSchema.infer;
|
||||
|
||||
/**
|
||||
* Construction-time overrides for callers that are not the model.
|
||||
*
|
||||
* The model-facing schema deliberately does not grow these: they exist for
|
||||
* wire bridges (the Cursor `pi_grep` frame) whose protocol carries an explicit
|
||||
* context width and total match cap, and which would otherwise have to drop
|
||||
* them. Unset means "use the session settings / built-in caps" — the behavior
|
||||
* every model-issued call keeps.
|
||||
*/
|
||||
export interface GrepToolOptions {
|
||||
/** Overrides `grep.contextBefore`/`grep.contextAfter` for every call on this instance. */
|
||||
context?: number;
|
||||
/** Caps total surfaced matches. Applied on top of the built-in per-file and file-window caps, never above them. */
|
||||
totalMatchLimit?: number;
|
||||
}
|
||||
|
||||
export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails> {
|
||||
readonly name = "grep";
|
||||
readonly approval = (args: unknown): ToolTier => {
|
||||
@@ -897,7 +913,17 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
readonly parameters = searchSchema;
|
||||
readonly strict = true;
|
||||
|
||||
constructor(private readonly session: ToolSession) {
|
||||
readonly #contextOverride?: number;
|
||||
readonly #totalMatchLimit?: number;
|
||||
|
||||
constructor(
|
||||
private readonly session: ToolSession,
|
||||
options?: GrepToolOptions,
|
||||
) {
|
||||
const context = options?.context;
|
||||
this.#contextOverride = context !== undefined ? Math.max(0, Math.floor(context)) : undefined;
|
||||
const total = options?.totalMatchLimit;
|
||||
this.#totalMatchLimit = total !== undefined ? Math.max(1, Math.floor(total)) : undefined;
|
||||
const displayMode = resolveFileDisplayMode(session);
|
||||
this.description = prompt.render(grepDescription, {
|
||||
IS_HL_MODE: displayMode.hashLines,
|
||||
@@ -978,8 +1004,8 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
`or pass a UTF-8 text member.`,
|
||||
);
|
||||
}
|
||||
const normalizedContextBefore = this.session.settings.get("grep.contextBefore");
|
||||
const normalizedContextAfter = this.session.settings.get("grep.contextAfter");
|
||||
const normalizedContextBefore = this.#contextOverride ?? this.session.settings.get("grep.contextBefore");
|
||||
const normalizedContextAfter = this.#contextOverride ?? this.session.settings.get("grep.contextAfter");
|
||||
const ignoreCase = !(caseSensitive ?? true);
|
||||
const useGitignore = gitignore ?? true;
|
||||
const patternHasNewline = normalizedPattern.includes("\n") || normalizedPattern.includes("\\n");
|
||||
@@ -1297,9 +1323,22 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
// Single-file scopes can't paginate — there is one file by definition.
|
||||
const canPaginate = isMultiScope;
|
||||
const skipFiles = canPaginate ? Math.min(normalizedSkip, totalFiles) : 0;
|
||||
const windowFiles = canPaginate ? fileOrder.slice(skipFiles, skipFiles + DEFAULT_FILE_LIMIT) : fileOrder;
|
||||
const fileLimitReached = canPaginate && totalFiles > skipFiles + DEFAULT_FILE_LIMIT;
|
||||
// A caller with a total match cap is not paginating: the cap bounds the
|
||||
// output, and the only consumer that sets one (`pi_grep`) has no `skip`
|
||||
// field to follow a "use skip=N" suggestion with. Windowing it to the
|
||||
// first 20 files would silently return fewer matches than it asked for
|
||||
// while reporting the cap as unreached.
|
||||
//
|
||||
// The window is cap+1 files, not cap: with one match per file, a cap
|
||||
// of N over exactly N files is complete, while over N+1 files it is
|
||||
// clipped — and only reading that extra file distinguishes the two.
|
||||
// The cap below then does the trimming and records that it bit, so
|
||||
// `match_limit_reached` reaches the frame set.
|
||||
const fileWindow = this.#totalMatchLimit !== undefined ? this.#totalMatchLimit + 1 : DEFAULT_FILE_LIMIT;
|
||||
const windowFiles = canPaginate ? fileOrder.slice(skipFiles, skipFiles + fileWindow) : fileOrder;
|
||||
const fileLimitReached = canPaginate && totalFiles > skipFiles + fileWindow;
|
||||
const selectedMatches: GrepMatch[] = [];
|
||||
let totalMatchLimitReached = false;
|
||||
if (windowFiles.length > 0) {
|
||||
const lists = windowFiles.map(file => matchesByPath.get(file) ?? []);
|
||||
const cursors = new Array<number>(lists.length).fill(0);
|
||||
@@ -1313,6 +1352,14 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
}
|
||||
}
|
||||
}
|
||||
// Round-robin above interleaves files for diversity, so the cap is
|
||||
// applied after selection rather than as a per-list bound: trimming
|
||||
// mid-rotation would silently favour whichever files sort first.
|
||||
const cap = this.#totalMatchLimit;
|
||||
if (cap !== undefined && selectedMatches.length > cap) {
|
||||
selectedMatches.length = cap;
|
||||
totalMatchLimitReached = true;
|
||||
}
|
||||
}
|
||||
const nextSkip = skipFiles + windowFiles.length;
|
||||
const limitMessage = fileLimitReached
|
||||
@@ -1503,7 +1550,12 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
const output = truncation.content;
|
||||
const displayText = displayLines.join("\n");
|
||||
const truncated = Boolean(
|
||||
fileLimitReached || perFileLimitReached || result.limitReached || truncation.truncated || linesTruncated,
|
||||
fileLimitReached ||
|
||||
perFileLimitReached ||
|
||||
totalMatchLimitReached ||
|
||||
result.limitReached ||
|
||||
truncation.truncated ||
|
||||
linesTruncated,
|
||||
);
|
||||
const details: GrepToolDetails = {
|
||||
scopePath,
|
||||
@@ -1517,8 +1569,12 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
count: fileMatchCounts.get(path) ?? 0,
|
||||
})),
|
||||
truncated,
|
||||
fileLimitReached: fileLimitReached ? DEFAULT_FILE_LIMIT : undefined,
|
||||
perFileLimitReached: perFileLimitReached ? perFileMatchCap : undefined,
|
||||
fileLimitReached: fileLimitReached ? fileWindow : undefined,
|
||||
perFileLimitReached: totalMatchLimitReached
|
||||
? this.#totalMatchLimit
|
||||
: perFileLimitReached
|
||||
? perFileMatchCap
|
||||
: undefined,
|
||||
displayContent: displayText,
|
||||
missingPaths: missingPaths.length > 0 ? missingPaths : undefined,
|
||||
};
|
||||
|
||||
@@ -509,6 +509,18 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
// unreachable, in which case eval dispatches exclusively to the others.
|
||||
const allowEval = effectivePythonAllowed || allowJs || effectiveRubyAllowed || effectiveJuliaAllowed;
|
||||
|
||||
// Checkpoint and rewind are a pair: listing one without the other strands
|
||||
// the agent (it can checkpoint but not rewind, or vice versa). Auto-include
|
||||
// the sister tool so a one-sided frontmatter `tools:` entry still works.
|
||||
// Unlike the AST/auto-learn convenience auto-includes below, this is a
|
||||
// safety pairing — it applies to restricted sessions too.
|
||||
if (requestedTools && session.settings.get("checkpoint.enabled")) {
|
||||
if (requestedTools.includes("checkpoint") && !requestedTools.includes("rewind")) {
|
||||
requestedTools.push("rewind");
|
||||
} else if (requestedTools.includes("rewind") && !requestedTools.includes("checkpoint")) {
|
||||
requestedTools.push("checkpoint");
|
||||
}
|
||||
}
|
||||
// Auto-include AST counterparts when their text-based sibling is present.
|
||||
// Restricted callers own the active list and must not have it widened.
|
||||
if (requestedTools && !restrictToolNames) {
|
||||
@@ -582,7 +594,11 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
if (name === "ask") return session.settings.get("ask.enabled");
|
||||
if (name === "browser") return session.settings.get("browser.enabled");
|
||||
if (name === "computer") return session.settings.get("computer.enabled");
|
||||
if (name === "checkpoint" || name === "rewind") return session.settings.get("checkpoint.enabled");
|
||||
if (name === "checkpoint" || name === "rewind")
|
||||
return (
|
||||
session.settings.get("checkpoint.enabled") &&
|
||||
((session.taskDepth ?? 0) === 0 || requestedTools !== undefined)
|
||||
);
|
||||
if (name === "hub") {
|
||||
return (
|
||||
!restrictToolNames && session.enableIrc !== false && isIrcEnabled(session.settings, session.taskDepth ?? 0)
|
||||
@@ -592,11 +608,15 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
return ["hindsight", "mnemopi"].includes(session.settings.get("memory.backend") ?? "");
|
||||
}
|
||||
if (name === "memory_edit") return session.settings.get("memory.backend") === "mnemopi";
|
||||
if (name === "manage_skill") return session.settings.get("autolearn.enabled") && (session.taskDepth ?? 0) === 0;
|
||||
if (name === "manage_skill")
|
||||
return (
|
||||
session.settings.get("autolearn.enabled") &&
|
||||
((session.taskDepth ?? 0) === 0 || requestedTools !== undefined)
|
||||
);
|
||||
if (name === "learn") {
|
||||
return (
|
||||
session.settings.get("autolearn.enabled") &&
|
||||
(session.taskDepth ?? 0) === 0 &&
|
||||
((session.taskDepth ?? 0) === 0 || requestedTools !== undefined) &&
|
||||
["hindsight", "mnemopi", "local"].includes(session.settings.get("memory.backend") ?? "")
|
||||
);
|
||||
}
|
||||
|
||||
@@ -523,6 +523,94 @@ export function resolveToCwd(filePath: string, cwd: string): string {
|
||||
return path.resolve(cwd, expanded);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a path that MUST stay inside `cwd`, or `null` when it would escape.
|
||||
*
|
||||
* {@link resolveToCwd} deliberately honors absolute paths, `~`, and `..` —
|
||||
* correct for a path a user typed, wrong for one a remote peer supplied.
|
||||
* Callers handling untrusted input (Cursor's `download_path`) use this instead:
|
||||
* only a non-empty relative path landing under the live cwd is accepted, so
|
||||
* neither `/etc/passwd` nor `../../escape` can be written through.
|
||||
*
|
||||
* The lexical check alone is not containment: a symlink inside the workspace
|
||||
* can point anywhere, so `out/config` under a `ws/out -> /elsewhere` link is
|
||||
* relative, `..`-free, and still writes outside. Both the target and its
|
||||
* deepest existing ancestor are therefore realpath-resolved — the ancestor
|
||||
* because a download names a file that does not exist yet, so the link in its
|
||||
* path is the only thing that can be resolved before the write.
|
||||
*
|
||||
* The cwd itself is rejected: a download names a file, never the directory.
|
||||
*/
|
||||
export function confineToWorkspace(filePath: string, cwd: string): string | null {
|
||||
if (!filePath || path.isAbsolute(filePath)) return null;
|
||||
// `~` expands to an absolute path, and an internal URL is not a filesystem
|
||||
// target at all; neither is a relative workspace path.
|
||||
if (filePath.startsWith("~") || isInternalUrlPath(filePath)) return null;
|
||||
const root = path.resolve(cwd);
|
||||
const resolved = path.resolve(root, filePath);
|
||||
if (!isUnderRootLexical(resolved, root)) return null;
|
||||
|
||||
// A workspace reached through a link of its own is legitimate (/tmp on
|
||||
// macOS), so the real root is the comparison basis. An unresolvable root is
|
||||
// not a workspace to contain anything in.
|
||||
const realRoot = tryRealpath(root);
|
||||
if (!realRoot) return null;
|
||||
|
||||
// An existing target is authoritative: resolve it outright.
|
||||
const realTarget = tryRealpath(resolved);
|
||||
if (realTarget) return isUnderRootLexical(realTarget, realRoot) ? resolved : null;
|
||||
|
||||
// `realpath` also fails on a *dangling* link, and a write follows that link
|
||||
// to wherever it points. Chasing the chain to decide would mean
|
||||
// reimplementing symlink resolution (multi-hop, relative hops, loops, and
|
||||
// a TOCTOU window against a link that can be re-pointed between the check
|
||||
// and the write). A download names a file to create, so a path that is
|
||||
// already an unresolvable link is refused outright — the one shape where
|
||||
// "cannot tell where this lands" is the whole answer.
|
||||
if (isSymlink(resolved)) return null;
|
||||
|
||||
// Otherwise walk up to the deepest ancestor that does exist and check that,
|
||||
// then re-apply the segments below it. Those segments are `..`-free by the
|
||||
// lexical check above, so they cannot climb back out.
|
||||
let ancestor = path.dirname(resolved);
|
||||
const tail: string[] = [path.basename(resolved)];
|
||||
for (;;) {
|
||||
const real = tryRealpath(ancestor);
|
||||
if (real) {
|
||||
return isUnderRootLexical(path.join(real, ...tail.reverse()), realRoot) ? resolved : null;
|
||||
}
|
||||
const parent = path.dirname(ancestor);
|
||||
// Ran past the root without finding anything real: the workspace itself
|
||||
// resolved above, so this cannot happen unless it vanished mid-check.
|
||||
if (parent === ancestor || !isUnderRootLexical(ancestor, root)) return null;
|
||||
tail.push(path.basename(ancestor));
|
||||
ancestor = parent;
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether `target` is a strict descendant of `root`, ignoring symlinks. */
|
||||
function isUnderRootLexical(target: string, root: string): boolean {
|
||||
const relative = path.relative(root, target);
|
||||
return !!relative && !relative.startsWith("..") && !path.isAbsolute(relative);
|
||||
}
|
||||
|
||||
function tryRealpath(target: string): string | null {
|
||||
try {
|
||||
return fs.realpathSync.native(target);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether the path itself is a symlink, without following it. */
|
||||
function isSymlink(target: string): boolean {
|
||||
try {
|
||||
return fs.lstatSync(target).isSymbolicLink();
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
export function formatPathRelativeToCwd(
|
||||
filePath: string,
|
||||
cwd: string,
|
||||
|
||||
@@ -735,6 +735,8 @@ export interface ReadToolDetails {
|
||||
method?: string;
|
||||
notes?: string[];
|
||||
meta?: OutputMeta;
|
||||
/** Full on-disk byte size recorded before applying a file range. */
|
||||
fileSize?: number;
|
||||
/** Raw text + start line for user-visible TUI rendering, set when content is text-like.
|
||||
* Mirrors the same lines the model receives but without hashline/line-number prefixes,
|
||||
* so the TUI can render the file content with its own gutter without re-parsing the formatted text. */
|
||||
@@ -2959,6 +2961,7 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
}
|
||||
}
|
||||
|
||||
details.fileSize = fileSize;
|
||||
this.#markMarkdownContentType(details, absolutePath);
|
||||
if (suffixResolution) {
|
||||
details.suffixResolution = suffixResolution;
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import * as path from "node:path";
|
||||
import { getLastChangelogVersionPath, isEnoent, logger } from "@oh-my-pi/pi-utils";
|
||||
import bundledChangelogPath from "../../CHANGELOG.md" with { type: "file" };
|
||||
import type { SettingValue } from "../config/settings";
|
||||
|
||||
export interface ChangelogEntry {
|
||||
major: number;
|
||||
@@ -28,6 +29,101 @@ export interface StartupChangelogSelection {
|
||||
persistCurrentVersion: boolean;
|
||||
truncated: boolean;
|
||||
selectedEntries: number;
|
||||
totalUnseenEntries: number;
|
||||
latestVersion: string | undefined;
|
||||
changeCount: number;
|
||||
categoryCounts: Record<string, number>;
|
||||
}
|
||||
|
||||
const CHANGELOG_CATEGORY_ORDER = [
|
||||
"Breaking Changes",
|
||||
"Added",
|
||||
"Changed",
|
||||
"Deprecated",
|
||||
"Removed",
|
||||
"Fixed",
|
||||
"Security",
|
||||
] as const;
|
||||
|
||||
function emptyStartupSelection(persistCurrentVersion: boolean): StartupChangelogSelection {
|
||||
return {
|
||||
markdown: undefined,
|
||||
persistCurrentVersion,
|
||||
truncated: false,
|
||||
selectedEntries: 0,
|
||||
totalUnseenEntries: 0,
|
||||
latestVersion: undefined,
|
||||
changeCount: 0,
|
||||
categoryCounts: {},
|
||||
};
|
||||
}
|
||||
|
||||
function summarizeChangelogEntries(entries: readonly ChangelogEntry[]): {
|
||||
changeCount: number;
|
||||
categoryCounts: Record<string, number>;
|
||||
} {
|
||||
const categoryCounts: Record<string, number> = {};
|
||||
let changeCount = 0;
|
||||
|
||||
for (const entry of entries) {
|
||||
let category: string | undefined;
|
||||
for (const line of entry.content.split("\n")) {
|
||||
const heading = line.match(/^###\s+(.+?)\s*$/);
|
||||
if (heading) {
|
||||
category = heading[1];
|
||||
continue;
|
||||
}
|
||||
if (!category || !/^-\s+\S/.test(line)) continue;
|
||||
categoryCounts[category] = (categoryCounts[category] ?? 0) + 1;
|
||||
changeCount++;
|
||||
}
|
||||
}
|
||||
|
||||
return { changeCount, categoryCounts };
|
||||
}
|
||||
|
||||
function categoryLabel(category: string, count: number): string {
|
||||
if (category === "Breaking Changes") {
|
||||
return count === 1 ? "breaking change" : "breaking changes";
|
||||
}
|
||||
return category.toLowerCase();
|
||||
}
|
||||
|
||||
/** Format the compact, deterministic startup update notice. */
|
||||
export function formatStartupChangelogSummary(selection: StartupChangelogSelection): string {
|
||||
const latestVersion = selection.latestVersion;
|
||||
if (!latestVersion || selection.selectedEntries === 0) {
|
||||
return "Updated omp. Use /changelog for recent changes.";
|
||||
}
|
||||
|
||||
const releaseCount = selection.selectedEntries;
|
||||
const changeCount = selection.changeCount;
|
||||
const releaseWord = releaseCount === 1 ? "release" : "releases";
|
||||
const changeWord = changeCount === 1 ? "change" : "changes";
|
||||
const firstLine =
|
||||
releaseCount === 1
|
||||
? `Updated to v${latestVersion} · ${changeCount} ${changeWord} in 1 release`
|
||||
: `Updated to v${latestVersion} · ${changeCount} ${changeWord} across ${releaseCount} ${releaseWord}`;
|
||||
|
||||
const orderedCategories = [
|
||||
...CHANGELOG_CATEGORY_ORDER.filter(category => selection.categoryCounts[category]),
|
||||
...Object.keys(selection.categoryCounts)
|
||||
.filter(category => !(CHANGELOG_CATEGORY_ORDER as readonly string[]).includes(category))
|
||||
.sort(),
|
||||
];
|
||||
const breakdown = orderedCategories
|
||||
.map(
|
||||
category =>
|
||||
`${selection.categoryCounts[category]} ${categoryLabel(category, selection.categoryCounts[category])}`,
|
||||
)
|
||||
.join(" · ");
|
||||
const omittedReleases = selection.totalUnseenEntries - selection.selectedEntries;
|
||||
const detailHint =
|
||||
omittedReleases > 0
|
||||
? `+${omittedReleases} earlier ${omittedReleases === 1 ? "release" : "releases"} · Use /changelog full for history.`
|
||||
: "Use /changelog for details.";
|
||||
|
||||
return breakdown ? `${firstLine}\n${breakdown} · ${detailHint}` : `${firstLine}\n${detailHint}`;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -177,16 +273,17 @@ export function selectStartupChangelog(
|
||||
): StartupChangelogSelection {
|
||||
const parsedLastVersion = parseChangelogVersion(lastVersion);
|
||||
if (!parsedLastVersion) {
|
||||
return { markdown: undefined, persistCurrentVersion: true, truncated: false, selectedEntries: 0 };
|
||||
return emptyStartupSelection(true);
|
||||
}
|
||||
const markerVersion = lastVersion ?? "";
|
||||
if (markerVersion === currentVersion) {
|
||||
return { markdown: undefined, persistCurrentVersion: false, truncated: false, selectedEntries: 0 };
|
||||
return emptyStartupSelection(false);
|
||||
}
|
||||
|
||||
const newEntries = getNewEntries(entries, markerVersion).slice(0, RECENT_CHANGELOG_ENTRY_LIMIT);
|
||||
const allNewEntries = getNewEntries(entries, markerVersion);
|
||||
const newEntries = allNewEntries.slice(0, RECENT_CHANGELOG_ENTRY_LIMIT);
|
||||
if (newEntries.length === 0) {
|
||||
return { markdown: undefined, persistCurrentVersion: false, truncated: false, selectedEntries: 0 };
|
||||
return emptyStartupSelection(false);
|
||||
}
|
||||
|
||||
const rendered = renderChangelogEntries(newEntries, {
|
||||
@@ -194,14 +291,57 @@ export function selectStartupChangelog(
|
||||
truncationHint: STARTUP_CHANGELOG_FULL_HINT,
|
||||
oldestFirst: false,
|
||||
});
|
||||
const summary = summarizeChangelogEntries(newEntries);
|
||||
const latestEntry = newEntries[0];
|
||||
return {
|
||||
markdown: rendered.markdown,
|
||||
persistCurrentVersion: true,
|
||||
truncated: rendered.truncated,
|
||||
selectedEntries: newEntries.length,
|
||||
totalUnseenEntries: allNewEntries.length,
|
||||
latestVersion: latestEntry ? `${latestEntry.major}.${latestEntry.minor}.${latestEntry.patch}` : undefined,
|
||||
...summary,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve and persist the automatic startup changelog decision.
|
||||
*
|
||||
* Hidden mode advances the marker only for an upgrade, so downgrades do not
|
||||
* erase knowledge of a newer version the user has already seen.
|
||||
*/
|
||||
export async function resolveStartupChangelogForDisplay(options: {
|
||||
mode: SettingValue<"startup.changelogMode">;
|
||||
currentVersion: string;
|
||||
changelogPath?: string;
|
||||
agentDir?: string;
|
||||
}): Promise<StartupChangelogSelection | undefined> {
|
||||
const lastVersion = await readLastChangelogVersion(options.agentDir);
|
||||
const parsedLastVersion = parseChangelogVersion(lastVersion);
|
||||
if (!parsedLastVersion) {
|
||||
await writeLastChangelogVersion(options.currentVersion, options.agentDir);
|
||||
return undefined;
|
||||
}
|
||||
if (lastVersion === options.currentVersion) {
|
||||
// Steady state: skip the changelog file read and parse.
|
||||
return undefined;
|
||||
}
|
||||
if (options.mode === "hidden") {
|
||||
const currentVersion = parseChangelogVersion(options.currentVersion);
|
||||
if (currentVersion && compareVersions(currentVersion, parsedLastVersion) > 0) {
|
||||
await writeLastChangelogVersion(options.currentVersion, options.agentDir);
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const entries = await parseChangelog(options.changelogPath);
|
||||
const startupChangelog = selectStartupChangelog(entries, lastVersion, options.currentVersion);
|
||||
if (startupChangelog.persistCurrentVersion) {
|
||||
await writeLastChangelogVersion(options.currentVersion, options.agentDir);
|
||||
}
|
||||
return startupChangelog.markdown ? startupChangelog : undefined;
|
||||
}
|
||||
|
||||
// Re-export getChangelogPath from paths.ts for convenience
|
||||
export { getChangelogPath } from "../config";
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ export const SEARCH_PROVIDER_OPTIONS = [
|
||||
},
|
||||
{ value: "xai", label: "xAI", description: "Grok web search via xAI Responses API (requires XAI_API_KEY)" },
|
||||
{ value: "zai", label: "Z.AI", description: "Calls Z.AI webSearchPrime MCP" },
|
||||
{ value: "exa", label: "Exa", description: "Uses Exa API when EXA_API_KEY is set; falls back to Exa MCP" },
|
||||
{ value: "exa", label: "Exa", description: "API via /login exa or EXA_API_KEY; explicit keyless fallback via MCP" },
|
||||
{ value: "tinyfish", label: "TinyFish", description: "Requires TINYFISH_API_KEY" },
|
||||
{ value: "jina", label: "Jina", description: "Requires JINA_API_KEY" },
|
||||
{ value: "kagi", label: "Kagi", description: "Requires KAGI_API_KEY and Kagi Search API beta access" },
|
||||
|
||||
@@ -1,8 +1,10 @@
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import { Agent, type AgentMessage, type CompactionSummaryMessage, countTokens } from "@oh-my-pi/pi-agent-core";
|
||||
import * as compactionModule from "@oh-my-pi/pi-agent-core/compaction";
|
||||
import { calculateContextTokens, estimateTokens, resolveThresholdTokens } from "@oh-my-pi/pi-agent-core/compaction";
|
||||
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
|
||||
import { createMockModel, type MockModel, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { estimateToolSchemaTokens } from "@oh-my-pi/pi-coding-agent/modes/utils/context-usage";
|
||||
@@ -122,6 +124,63 @@ describe("AgentSession advisor context maintenance", () => {
|
||||
};
|
||||
}
|
||||
|
||||
function createAdvisorFallbackHarness(options?: { sameProviderNativeEnabled?: boolean }) {
|
||||
const primaryMock = createMockModel({
|
||||
provider: "anthropic",
|
||||
responses: [{ content: ["primary complete"] }],
|
||||
});
|
||||
const advisorMock = createMockModel({
|
||||
provider: "openai",
|
||||
responses: [{ content: ["advisor reviewed current update"] }],
|
||||
});
|
||||
const nativeModel = getBundledModel("openai", "gpt-5");
|
||||
const sameProviderBase = getBundledModel("openai", "gpt-5-mini");
|
||||
const sameProviderModel =
|
||||
sameProviderBase && options?.sameProviderNativeEnabled === false
|
||||
? { ...sameProviderBase, remoteCompaction: { ...sameProviderBase.remoteCompaction, enabled: false } }
|
||||
: sameProviderBase;
|
||||
const crossProviderModel = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
if (!nativeModel || !sameProviderModel || !crossProviderModel) {
|
||||
throw new Error("Expected bundled compaction models");
|
||||
}
|
||||
|
||||
authStorage.setRuntimeApiKey(nativeModel.provider, "openai-key");
|
||||
const modelRegistry = new ModelRegistry(authStorage, tempDir.join("models.yml"));
|
||||
const settings = Settings.isolated({
|
||||
"advisor.syncBacklog": "1",
|
||||
"compaction.enabled": true,
|
||||
"compaction.strategy": "context-full",
|
||||
"contextPromotion.enabled": false,
|
||||
});
|
||||
settings.setModelRole("advisor", `${nativeModel.provider}/${nativeModel.id}`);
|
||||
settings.setModelRole("smol", `${sameProviderModel.provider}/${sameProviderModel.id}`);
|
||||
settings.setModelRole("slow", `${crossProviderModel.provider}/${crossProviderModel.id}`);
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: { model: primaryMock, systemPrompt: [], tools: [] },
|
||||
streamFn: primaryMock.stream,
|
||||
});
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
modelRegistry,
|
||||
advisorTools: [],
|
||||
advisorStreamFn: advisorMock.stream,
|
||||
});
|
||||
expect(session.setAdvisorEnabled(true)).toBe(true);
|
||||
const advisor = session.getAdvisorAgent();
|
||||
if (!advisor) throw new Error("Expected advisor agent to be active");
|
||||
advisor.setModel(nativeModel);
|
||||
const apiKeySpy = vi.spyOn(modelRegistry, "getApiKey").mockResolvedValue("test-key");
|
||||
vi.spyOn(modelRegistry, "getAvailable").mockReturnValue([nativeModel, sameProviderModel, crossProviderModel]);
|
||||
advisor.state.messages.push(
|
||||
usageAnchor(advisorMock, Date.now() - 2_000),
|
||||
usageAnchor(advisorMock, Date.now() - 1_000),
|
||||
);
|
||||
return { advisor, apiKeySpy, crossProviderModel, nativeModel, sameProviderModel, settings };
|
||||
}
|
||||
|
||||
it("maintains a 371,200-token cached advisor context before the 372,000-token window", async () => {
|
||||
const { advisor, advisorMock, settings } = createHarness();
|
||||
const anchor = usageAnchor(advisorMock, Date.now() - 1_000, 0.5);
|
||||
@@ -353,4 +412,144 @@ describe("AgentSession advisor context maintenance", () => {
|
||||
expect((JSON.parse(userId) as { session_id?: string }).session_id).toBe(advisor.sessionId);
|
||||
}
|
||||
});
|
||||
|
||||
it("continues same-provider advisor candidates but stops before crossing providers on non-auth failure", async () => {
|
||||
const { advisor, crossProviderModel, nativeModel, sameProviderModel } = createAdvisorFallbackHarness();
|
||||
const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async (preparation, model) => {
|
||||
if (model.provider === nativeModel.provider || model.provider === sameProviderModel.provider) {
|
||||
throw new compactionModule.NativeCompactionError(new Error("V2 native compaction transport failed"));
|
||||
}
|
||||
if (model.provider !== crossProviderModel.provider || model.id !== crossProviderModel.id) {
|
||||
throw new Error(`Unexpected compaction model ${model.provider}/${model.id}`);
|
||||
}
|
||||
return {
|
||||
summary: "cross-provider summary",
|
||||
shortSummary: "cross-provider",
|
||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||
tokensBefore: 42,
|
||||
};
|
||||
});
|
||||
|
||||
await session.prompt("small current update");
|
||||
|
||||
expect(compactSpy.mock.calls.map(([, model]) => `${model.provider}/${model.id}`)).toEqual([
|
||||
`${nativeModel.provider}/${nativeModel.id}`,
|
||||
`${sameProviderModel.provider}/${sameProviderModel.id}`,
|
||||
]);
|
||||
expect(JSON.stringify(advisor.state.messages)).toContain("prior advisor output");
|
||||
});
|
||||
|
||||
it("applies a successful same-provider native advisor fallback", async () => {
|
||||
const { advisor, crossProviderModel, nativeModel, sameProviderModel } = createAdvisorFallbackHarness();
|
||||
const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async (preparation, model) => {
|
||||
if (model.provider === nativeModel.provider && model.id === nativeModel.id) {
|
||||
throw new compactionModule.NativeCompactionError(new Error("V2 native compaction transport failed"));
|
||||
}
|
||||
if (model.provider === sameProviderModel.provider && model.id === sameProviderModel.id) {
|
||||
return {
|
||||
summary: "same-provider native summary",
|
||||
shortSummary: "same-provider native",
|
||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||
tokensBefore: 42,
|
||||
};
|
||||
}
|
||||
throw new Error(
|
||||
`Unexpected cross-provider compaction ${crossProviderModel.provider}/${crossProviderModel.id}`,
|
||||
);
|
||||
});
|
||||
|
||||
await session.prompt("small current update");
|
||||
|
||||
expect(compactSpy.mock.calls.map(([, model]) => `${model.provider}/${model.id}`)).toEqual([
|
||||
`${nativeModel.provider}/${nativeModel.id}`,
|
||||
`${sameProviderModel.provider}/${sameProviderModel.id}`,
|
||||
]);
|
||||
expect(JSON.stringify(advisor.state.messages)).toContain("same-provider native summary");
|
||||
});
|
||||
|
||||
it("skips unauthenticated advisor candidates before enforcing the native boundary", async () => {
|
||||
const { advisor, apiKeySpy, crossProviderModel, nativeModel, sameProviderModel, settings } =
|
||||
createAdvisorFallbackHarness();
|
||||
settings.setModelRole("smol", `${crossProviderModel.provider}/${crossProviderModel.id}`);
|
||||
settings.setModelRole("slow", `${sameProviderModel.provider}/${sameProviderModel.id}`);
|
||||
apiKeySpy.mockImplementation(async model =>
|
||||
model.provider === crossProviderModel.provider && model.id === crossProviderModel.id ? undefined : "test-key",
|
||||
);
|
||||
const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async (preparation, model) => {
|
||||
if (model.provider === nativeModel.provider && model.id === nativeModel.id) {
|
||||
throw new compactionModule.NativeCompactionError(new Error("V2 native compaction transport failed"));
|
||||
}
|
||||
if (model.provider === sameProviderModel.provider && model.id === sameProviderModel.id) {
|
||||
return {
|
||||
summary: "authenticated same-provider advisor summary",
|
||||
shortSummary: "authenticated same-provider advisor",
|
||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||
tokensBefore: 42,
|
||||
};
|
||||
}
|
||||
throw new Error(`Unexpected advisor compaction model ${model.provider}/${model.id}`);
|
||||
});
|
||||
|
||||
await session.prompt("small current update");
|
||||
|
||||
expect(compactSpy.mock.calls.map(([, model]) => `${model.provider}/${model.id}`)).toEqual([
|
||||
`${nativeModel.provider}/${nativeModel.id}`,
|
||||
`${sameProviderModel.provider}/${sameProviderModel.id}`,
|
||||
]);
|
||||
expect(JSON.stringify(advisor.state.messages)).toContain("authenticated same-provider advisor summary");
|
||||
});
|
||||
|
||||
it("stops before a same-provider advisor candidate with native compaction disabled", async () => {
|
||||
const { advisor, nativeModel, sameProviderModel } = createAdvisorFallbackHarness({
|
||||
sameProviderNativeEnabled: false,
|
||||
});
|
||||
const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async (preparation, model) => {
|
||||
if (model.provider === nativeModel.provider && model.id === nativeModel.id) {
|
||||
throw new compactionModule.NativeCompactionError(new Error("V2 native compaction transport failed"));
|
||||
}
|
||||
return {
|
||||
summary: "generic same-provider summary",
|
||||
shortSummary: "generic same-provider",
|
||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||
tokensBefore: 42,
|
||||
};
|
||||
});
|
||||
|
||||
await session.prompt("small current update");
|
||||
|
||||
expect(compactSpy.mock.calls.map(([, model]) => `${model.provider}/${model.id}`)).toEqual([
|
||||
`${nativeModel.provider}/${nativeModel.id}`,
|
||||
]);
|
||||
expect(JSON.stringify(advisor.state.messages)).not.toContain("generic same-provider summary");
|
||||
expect(sameProviderModel.remoteCompaction?.enabled).toBe(false);
|
||||
});
|
||||
|
||||
it("allows advisor compaction to cross providers after auth-classified native failures", async () => {
|
||||
const { advisor, crossProviderModel, nativeModel, sameProviderModel } = createAdvisorFallbackHarness();
|
||||
const compactSpy = vi.spyOn(compactionModule, "compact").mockImplementation(async (preparation, model) => {
|
||||
if (model.provider === nativeModel.provider || model.provider === sameProviderModel.provider) {
|
||||
throw new compactionModule.NativeCompactionError(
|
||||
Object.assign(new Error("native compaction authentication failed"), { status: 401 }),
|
||||
);
|
||||
}
|
||||
if (model.provider !== crossProviderModel.provider || model.id !== crossProviderModel.id) {
|
||||
throw new Error(`Unexpected compaction model ${model.provider}/${model.id}`);
|
||||
}
|
||||
return {
|
||||
summary: "authenticated fallback summary",
|
||||
shortSummary: "authenticated fallback",
|
||||
firstKeptEntryId: preparation.firstKeptEntryId,
|
||||
tokensBefore: 42,
|
||||
};
|
||||
});
|
||||
|
||||
await session.prompt("small current update");
|
||||
|
||||
expect(compactSpy.mock.calls.map(([, model]) => `${model.provider}/${model.id}`)).toEqual([
|
||||
`${nativeModel.provider}/${nativeModel.id}`,
|
||||
`${sameProviderModel.provider}/${sameProviderModel.id}`,
|
||||
`${crossProviderModel.provider}/${crossProviderModel.id}`,
|
||||
]);
|
||||
expect(JSON.stringify(advisor.state.messages)).toContain("authenticated fallback summary");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -3475,6 +3475,193 @@ describe("advisor", () => {
|
||||
expect(runtime.backlog).toBe(0);
|
||||
});
|
||||
|
||||
it("strips echoed thinking after a classifier refusal and succeeds without a notice", async () => {
|
||||
const promptInputs: string[] = [];
|
||||
const failures: unknown[] = [];
|
||||
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
|
||||
let promptCalls = 0;
|
||||
const agent: AdvisorAgent = {
|
||||
prompt: async input => {
|
||||
promptInputs.push(input);
|
||||
promptCalls++;
|
||||
if (promptCalls === 1) {
|
||||
state.error = "Refusal (reasoning_extraction): reasoning may not be echoed";
|
||||
state.messages.push({
|
||||
role: "assistant",
|
||||
content: [],
|
||||
stopReason: "error",
|
||||
stopDetails: { type: "refusal", category: "reasoning_extraction" },
|
||||
errorMessage: state.error,
|
||||
timestamp: 2,
|
||||
} as unknown as AgentMessage);
|
||||
return;
|
||||
}
|
||||
state.error = undefined;
|
||||
state.messages.push({
|
||||
role: "assistant",
|
||||
content: [],
|
||||
stopReason: "stop",
|
||||
timestamp: 3,
|
||||
} as unknown as AgentMessage);
|
||||
},
|
||||
abort: () => {},
|
||||
reset: () => {},
|
||||
rollbackTo: count => {
|
||||
state.messages.length = count;
|
||||
state.error = undefined;
|
||||
},
|
||||
state,
|
||||
};
|
||||
const messages = [
|
||||
{
|
||||
role: "assistant",
|
||||
content: [
|
||||
{ type: "thinking", thinking: "private reasoning" },
|
||||
{ type: "text", text: "answer" },
|
||||
],
|
||||
timestamp: 1,
|
||||
} as AgentMessage,
|
||||
];
|
||||
const runtime = new AdvisorRuntime(
|
||||
agent,
|
||||
{
|
||||
snapshotMessages: () => messages,
|
||||
enqueueAdvice: () => {},
|
||||
notifyFailure: error => failures.push(error),
|
||||
},
|
||||
0,
|
||||
);
|
||||
|
||||
runtime.onTurnEnd(messages);
|
||||
await settleUntil(() => runtime.backlog === 0);
|
||||
|
||||
expect(promptInputs).toHaveLength(2);
|
||||
expect(promptInputs[0]).toContain("private reasoning");
|
||||
expect(promptInputs[1]).not.toContain("private reasoning");
|
||||
expect(promptInputs[1]).toContain("answer");
|
||||
expect(failures).toEqual([]);
|
||||
});
|
||||
|
||||
it("surfaces a persistent classifier refusal after one stripped resend", async () => {
|
||||
const promptInputs: string[] = [];
|
||||
const failures: unknown[] = [];
|
||||
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
|
||||
const agent: AdvisorAgent = {
|
||||
prompt: async input => {
|
||||
promptInputs.push(input);
|
||||
state.error = "Refusal (reasoning_extraction): reasoning may not be echoed";
|
||||
state.messages.push({
|
||||
role: "assistant",
|
||||
content: [],
|
||||
stopReason: "error",
|
||||
stopDetails: { type: "refusal", category: "reasoning_extraction" },
|
||||
errorMessage: state.error,
|
||||
timestamp: promptInputs.length + 1,
|
||||
} as unknown as AgentMessage);
|
||||
},
|
||||
abort: () => {},
|
||||
reset: () => {},
|
||||
rollbackTo: count => {
|
||||
state.messages.length = count;
|
||||
state.error = undefined;
|
||||
},
|
||||
state,
|
||||
};
|
||||
const messages = [
|
||||
{
|
||||
role: "assistant",
|
||||
content: [
|
||||
{ type: "thinking", thinking: "private reasoning" },
|
||||
{ type: "text", text: "answer" },
|
||||
],
|
||||
timestamp: 1,
|
||||
} as AgentMessage,
|
||||
];
|
||||
const runtime = new AdvisorRuntime(
|
||||
agent,
|
||||
{
|
||||
snapshotMessages: () => messages,
|
||||
enqueueAdvice: () => {},
|
||||
notifyFailure: error => failures.push(error),
|
||||
},
|
||||
0,
|
||||
);
|
||||
|
||||
runtime.onTurnEnd(messages);
|
||||
await settleUntil(() => failures.length === 1 && runtime.backlog === 0);
|
||||
|
||||
expect(promptInputs).toHaveLength(2);
|
||||
expect(promptInputs[0]).toContain("private reasoning");
|
||||
expect(promptInputs[1]).not.toContain("private reasoning");
|
||||
expect(failures).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("degrades on a category-less refusal", async () => {
|
||||
const promptInputs: string[] = [];
|
||||
const failures: unknown[] = [];
|
||||
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
|
||||
let promptCalls = 0;
|
||||
const agent: AdvisorAgent = {
|
||||
prompt: async input => {
|
||||
promptInputs.push(input);
|
||||
promptCalls++;
|
||||
if (promptCalls === 1) {
|
||||
state.error = "Refusal: reasoning may not be echoed";
|
||||
state.messages.push({
|
||||
role: "assistant",
|
||||
content: [],
|
||||
stopReason: "error",
|
||||
stopDetails: { type: "refusal" },
|
||||
errorMessage: state.error,
|
||||
timestamp: 2,
|
||||
} as unknown as AgentMessage);
|
||||
return;
|
||||
}
|
||||
state.error = undefined;
|
||||
state.messages.push({
|
||||
role: "assistant",
|
||||
content: [],
|
||||
stopReason: "stop",
|
||||
timestamp: 3,
|
||||
} as unknown as AgentMessage);
|
||||
},
|
||||
abort: () => {},
|
||||
reset: () => {},
|
||||
rollbackTo: count => {
|
||||
state.messages.length = count;
|
||||
state.error = undefined;
|
||||
},
|
||||
state,
|
||||
};
|
||||
const messages = [
|
||||
{
|
||||
role: "assistant",
|
||||
content: [
|
||||
{ type: "thinking", thinking: "private reasoning" },
|
||||
{ type: "text", text: "answer" },
|
||||
],
|
||||
timestamp: 1,
|
||||
} as AgentMessage,
|
||||
];
|
||||
const runtime = new AdvisorRuntime(
|
||||
agent,
|
||||
{
|
||||
snapshotMessages: () => messages,
|
||||
enqueueAdvice: () => {},
|
||||
notifyFailure: error => failures.push(error),
|
||||
},
|
||||
0,
|
||||
);
|
||||
|
||||
runtime.onTurnEnd(messages);
|
||||
await settleUntil(() => runtime.backlog === 0);
|
||||
|
||||
expect(promptInputs).toHaveLength(2);
|
||||
expect(promptInputs[0]).toContain("private reasoning");
|
||||
expect(promptInputs[1]).not.toContain("private reasoning");
|
||||
expect(failures).toEqual([]);
|
||||
});
|
||||
|
||||
it("calls onTurnError with state.error before retrying the batch", async () => {
|
||||
const promptInputs: string[] = [];
|
||||
const turnErrors: unknown[] = [];
|
||||
|
||||
@@ -80,13 +80,19 @@ function signedThinkingOnlyStop(): MockResponse {
|
||||
async function createHarness(
|
||||
responses: MockResponse[],
|
||||
settingsOverrides: SettingsOverrides = {},
|
||||
options: { persistSession?: boolean; extensionRunner?: ExtensionRunner } = {},
|
||||
options: {
|
||||
persistSession?: boolean;
|
||||
extensionRunner?: ExtensionRunner;
|
||||
provider?: string;
|
||||
id?: string;
|
||||
} = {},
|
||||
): Promise<Harness & { mock: MockModel }> {
|
||||
const tempDir = TempDir.createSync("@pi-empty-stop-guard-");
|
||||
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db"));
|
||||
authStorage.setRuntimeApiKey("mock", "test-key");
|
||||
|
||||
const mock = createMockModel({ responses });
|
||||
const mock = createMockModel({ provider: options.provider, id: options.id, responses });
|
||||
authStorage.setRuntimeApiKey(mock.provider, "test-key");
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
@@ -493,6 +499,46 @@ describe("AgentSession empty stop guard", () => {
|
||||
expect(reminderMessages(session.agent.state.messages)).toHaveLength(1);
|
||||
});
|
||||
|
||||
it("preserves Codex commentary when discarding a colliding empty final stop", async () => {
|
||||
const timestamp = 1_725_287_000_000;
|
||||
vi.spyOn(Date, "now").mockReturnValue(timestamp);
|
||||
const commentary = "Codex commentary before the empty final answer.";
|
||||
const recovered = "Recovered after the empty final-answer retry.";
|
||||
const { session, mock } = await createHarness(
|
||||
[{ content: [commentary], stopReason: "stop" }, emptyStop(), { content: [recovered], stopReason: "stop" }],
|
||||
{},
|
||||
{ provider: "openai-codex", id: "gpt-5.5-codex" },
|
||||
);
|
||||
|
||||
await session.prompt("produce commentary");
|
||||
await session.waitForIdle();
|
||||
// Persisted identity must win regardless of branch enumeration order. The
|
||||
// coarse matcher otherwise selects the commentary when it is encountered first.
|
||||
const getBranch = session.sessionManager.getBranch.bind(session.sessionManager);
|
||||
const branchSpy = vi
|
||||
.spyOn(session.sessionManager, "getBranch")
|
||||
.mockImplementation(() => getBranch().slice().reverse());
|
||||
await session.followUp("continue after commentary");
|
||||
await session.waitForIdle();
|
||||
branchSpy.mockRestore();
|
||||
|
||||
const assistantTexts = (messages: AgentMessage[]): string[] =>
|
||||
messages
|
||||
.filter((message): message is Extract<AgentMessage, { role: "assistant" }> => message.role === "assistant")
|
||||
.flatMap(message => message.content.flatMap(block => (block.type === "text" ? [block.text] : [])));
|
||||
|
||||
expect(mock.calls).toHaveLength(3);
|
||||
expect(assistantTexts(session.agent.state.messages)).toEqual([commentary, recovered]);
|
||||
expect(emptyAssistantStops(session.agent.state.messages)).toHaveLength(0);
|
||||
|
||||
const persistedMessages = session.sessionManager
|
||||
.getBranch()
|
||||
.filter(entry => entry.type === "message")
|
||||
.map(entry => entry.message as AgentMessage);
|
||||
expect(assistantTexts(persistedMessages)).toEqual([commentary, recovered]);
|
||||
expect(emptyAssistantStops(persistedMessages)).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("does not retry normal stop or tool-use turns", async () => {
|
||||
const normal = await createHarness([{ content: ["already done"], stopReason: "stop" }]);
|
||||
|
||||
|
||||
@@ -846,6 +846,31 @@ describe("AgentSession handoff", () => {
|
||||
expect(fallbackCandidateKey).toBeDefined();
|
||||
expect(promptSpy).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("does not switch providers after provider-native auto-compaction fails", async () => {
|
||||
session.settings.set("compaction.strategy", "context-full");
|
||||
session.settings.set("compaction.thresholdTokens", 50);
|
||||
session.settings.set("compaction.keepRecentTokens", 1);
|
||||
session.settings.set("contextPromotion.enabled", false);
|
||||
|
||||
const attemptedCandidates: string[] = [];
|
||||
vi.spyOn(compactionModule, "compact").mockImplementation(async (_preparation, candidate) => {
|
||||
attemptedCandidates.push(`${candidate.provider}/${candidate.id}`);
|
||||
throw new compactionModule.NativeCompactionError(new Error("native compaction transport failed"));
|
||||
});
|
||||
|
||||
await session.prompt("pending prompt ".repeat(120));
|
||||
await waitFor(() =>
|
||||
events.some(
|
||||
event =>
|
||||
event.type === "auto_compaction_end" &&
|
||||
event.errorMessage?.includes("native compaction transport failed") === true,
|
||||
),
|
||||
);
|
||||
|
||||
expect(attemptedCandidates.length).toBeGreaterThan(0);
|
||||
expect(new Set(attemptedCandidates.map(candidate => candidate.split("/", 1)[0]))).toHaveLength(1);
|
||||
});
|
||||
it("keeps pre-prompt context-full checks aligned with provider-anchored usage", async () => {
|
||||
await session.dispose();
|
||||
authStorage.setRuntimeApiKey("openai", "test-key");
|
||||
|
||||
@@ -78,11 +78,12 @@ describe("auto thinking classifier helpers", () => {
|
||||
expect(parseCliThinkingLevel("bogus")).toBeUndefined();
|
||||
});
|
||||
|
||||
it("maps online 4-way classifier labels to effort levels", () => {
|
||||
it("maps online level labels to effort levels", () => {
|
||||
expect(parseDifficultyLevel("x-high")).toBe(Effort.XHigh);
|
||||
expect(parseDifficultyLevel("The answer is HIGH.")).toBe(Effort.High);
|
||||
expect(parseDifficultyLevel("med")).toBe(Effort.Medium);
|
||||
expect(parseDifficultyLevel("low")).toBe(Effort.Low);
|
||||
expect(parseDifficultyLevel("max")).toBe(Effort.Max);
|
||||
expect(parseDifficultyLevel("unknown")).toBeUndefined();
|
||||
});
|
||||
|
||||
@@ -115,6 +116,34 @@ describe("auto thinking classifier helpers", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("keeps the local classifier capped at xhigh even when opted in to max", async () => {
|
||||
// The local backend only ever emits trivial/moderate/hard, so a sparse
|
||||
// ladder must not let the opt-in ceiling snap `hard` up to a tier the
|
||||
// on-device model never selected. `max` is the model's only tier at or
|
||||
// above Low, and the local ceiling hides it, so nothing is eligible —
|
||||
// falling through to `minimal` would breach the Low floor.
|
||||
const fixture = await createLocalClassifierFixture("qwen3-1.7b");
|
||||
const sparse = buildLadderModel("mock-minimal-max", [Effort.Minimal, Effort.Max]);
|
||||
vi.spyOn(tinyModelClient, "complete").mockResolvedValue("hard");
|
||||
|
||||
try {
|
||||
const settings = Settings.isolated({
|
||||
"providers.autoThinkingModel": "qwen3-1.7b",
|
||||
"providers.autoThinkingMaxEffort": "max",
|
||||
});
|
||||
|
||||
expect(
|
||||
await classifyDifficulty("cut over the storage layer", {
|
||||
settings,
|
||||
registry: fixture.registry,
|
||||
model: sparse,
|
||||
}),
|
||||
).toBeUndefined();
|
||||
} finally {
|
||||
fixture.cleanup();
|
||||
}
|
||||
});
|
||||
|
||||
it("uses a larger local non-reasoning classifier floor", async () => {
|
||||
let maxTokens: number | undefined;
|
||||
const fixture = await createLocalClassifierFixture("qwen2.5-1.5b");
|
||||
@@ -202,6 +231,155 @@ describe("auto thinking classifier helpers", () => {
|
||||
expect(options).toMatchObject({ disableReasoning: true, maxTokens: 1024 });
|
||||
});
|
||||
|
||||
function createOnlineFixture(targetModel: Model, answer: string, maxEffort: "xhigh" | "max" = "xhigh") {
|
||||
const classifierModel = getBundledModel("anthropic", "claude-sonnet-4-6");
|
||||
if (!classifierModel) throw new Error("Expected bundled Claude Sonnet 4.6 model");
|
||||
const settings = {
|
||||
get(path: string) {
|
||||
if (path === "providers.autoThinkingModel") return "online";
|
||||
return path === "providers.autoThinkingMaxEffort" ? maxEffort : undefined;
|
||||
},
|
||||
getModelRole(role: string) {
|
||||
return role === "smol" ? `${classifierModel.provider}/${classifierModel.id}` : undefined;
|
||||
},
|
||||
getStorage() {
|
||||
return undefined;
|
||||
},
|
||||
} as never;
|
||||
const registry = {
|
||||
getAvailable: () => [classifierModel],
|
||||
getApiKey: async () => "test-key",
|
||||
resolver: () => async () => "test-key",
|
||||
} as never;
|
||||
const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({
|
||||
stopReason: "stop",
|
||||
content: [{ type: "text", text: answer }],
|
||||
} as never);
|
||||
return { deps: { settings, registry, model: targetModel }, completeSimpleMock };
|
||||
}
|
||||
|
||||
function buildLadderModel(id: string, efforts: Effort[]): Model {
|
||||
return buildModel({
|
||||
id,
|
||||
name: id,
|
||||
api: "openai-completions",
|
||||
provider: "mock",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
thinking: { mode: "effort", efforts },
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 4096,
|
||||
});
|
||||
}
|
||||
|
||||
const MAX_LADDER = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh, Effort.Max];
|
||||
const XHIGH_LADDER = [Effort.Low, Effort.Medium, Effort.High, Effort.XHigh];
|
||||
|
||||
it("offers the max label only when opted in on a model that exposes the tier", async () => {
|
||||
const optedIn = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "high", "max");
|
||||
await classifyDifficulty("refactor the scheduler", optedIn.deps);
|
||||
const optedInRequest = optedIn.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
|
||||
// The label alone is inert: the criteria and the tie-break exception are
|
||||
// what make the tier reachable, so all three must ship together.
|
||||
expect(optedInRequest.systemPrompt[0]).toContain("`max`");
|
||||
expect(optedInRequest.systemPrompt[0]).toContain("no reproduction to work from");
|
||||
expect(optedInRequest.systemPrompt[0]).toContain("except between `xhigh` and `max`");
|
||||
|
||||
vi.restoreAllMocks();
|
||||
|
||||
const defaulted = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "high");
|
||||
await classifyDifficulty("refactor the scheduler", defaulted.deps);
|
||||
const defaultedRequest = defaulted.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
|
||||
expect(defaultedRequest.systemPrompt[0]).not.toMatch(/\bmax\b/);
|
||||
expect(defaultedRequest.systemPrompt[0]).toContain("`xhigh`");
|
||||
// The tie-break exception is what makes the top tier reachable, so it must
|
||||
// not leak into the prompt of a user who did not opt in.
|
||||
expect(defaultedRequest.systemPrompt[0]).toContain("choose the lower one.");
|
||||
expect(defaultedRequest.systemPrompt[0]).not.toContain("no reproduction to work from");
|
||||
|
||||
vi.restoreAllMocks();
|
||||
|
||||
const unsupported = createOnlineFixture(buildLadderModel("mock-xhigh", XHIGH_LADDER), "high", "max");
|
||||
await classifyDifficulty("refactor the scheduler", unsupported.deps);
|
||||
const unsupportedRequest = unsupported.completeSimpleMock.mock.calls[0]?.[1] as { systemPrompt: string[] };
|
||||
expect(unsupportedRequest.systemPrompt[0]).not.toMatch(/\bmax\b/);
|
||||
expect(unsupportedRequest.systemPrompt[0]).toContain("choose the lower one.");
|
||||
});
|
||||
|
||||
it("resolves max only when opted in, and snaps it to the ceiling otherwise", async () => {
|
||||
const optedIn = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "max", "max");
|
||||
expect(await classifyDifficulty("untangle this cross-service race", optedIn.deps)).toBe(Effort.Max);
|
||||
|
||||
vi.restoreAllMocks();
|
||||
|
||||
// Hallucinated `max` on a max-capable model must not cross the default ceiling.
|
||||
const defaulted = createOnlineFixture(buildLadderModel("mock-max", MAX_LADDER), "max");
|
||||
expect(await classifyDifficulty("untangle this cross-service race", defaulted.deps)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("resolves the sparse ladder's max tier when opted in", async () => {
|
||||
const fixture = createOnlineFixture(buildLadderModel("mock-sparse", [Effort.High, Effort.Max]), "max", "max");
|
||||
expect(await classifyDifficulty("cut over the storage layer", fixture.deps)).toBe(Effort.Max);
|
||||
});
|
||||
|
||||
it("takes the first label when the classifier echoes several", async () => {
|
||||
// `earliest()` is deliberately conservative: an echoed list resolves to the
|
||||
// lowest-positioned label rather than the model's final word.
|
||||
const fixture = createOnlineFixture(
|
||||
buildLadderModel("mock-max", MAX_LADDER),
|
||||
"low, medium, high, xhigh, max",
|
||||
"max",
|
||||
);
|
||||
expect(await classifyDifficulty("rename a helper", fixture.deps)).toBe(Effort.Low);
|
||||
});
|
||||
|
||||
it("snaps a hallucinated max back to the model's ceiling instead of failing the turn", async () => {
|
||||
const fixture = createOnlineFixture(buildLadderModel("mock-xhigh", XHIGH_LADDER), "max");
|
||||
expect(await classifyDifficulty("untangle this cross-service race", fixture.deps)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("resolves no level on a max-only ladder without opt-in", async () => {
|
||||
// `["max"]` has nothing at or below the default ceiling, so the model clamp
|
||||
// must not snap the request back up — auto yields nothing and the session
|
||||
// keeps its current level.
|
||||
const defaulted = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "xhigh");
|
||||
expect(await classifyDifficulty("cut over the storage layer", defaulted.deps)).toBeUndefined();
|
||||
|
||||
vi.restoreAllMocks();
|
||||
|
||||
const optedIn = createOnlineFixture(buildLadderModel("mock-max-only", [Effort.Max]), "max", "max");
|
||||
expect(await classifyDifficulty("cut over the storage layer", optedIn.deps)).toBe(Effort.Max);
|
||||
});
|
||||
|
||||
it("has no provisional level on a max-only ladder", () => {
|
||||
expect(resolveProvisionalAutoLevel(buildLadderModel("mock-max-only", [Effort.Max]))).toBeUndefined();
|
||||
});
|
||||
|
||||
it("stops at the highest tier under the ceiling on a sparse ladder", async () => {
|
||||
const fixture = createOnlineFixture(buildLadderModel("mock-hm", [Effort.High, Effort.Max]), "max");
|
||||
expect(await classifyDifficulty("cut over the storage layer", fixture.deps)).toBe(Effort.High);
|
||||
});
|
||||
|
||||
it("keeps the provisional auto level below max even when the model defaults to it", () => {
|
||||
const maxDefaultModel = buildModel({
|
||||
id: "mock-max-default",
|
||||
name: "mock-max-default",
|
||||
api: "openai-completions",
|
||||
provider: "mock",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
thinking: { mode: "effort", efforts: MAX_LADDER, defaultLevel: Effort.Max },
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 128_000,
|
||||
maxTokens: 4096,
|
||||
});
|
||||
|
||||
expect(resolveProvisionalAutoLevel(maxDefaultModel)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("clamps auto effort to model support while never resolving below low", () => {
|
||||
const model = getBundledModel("anthropic", "claude-sonnet-4-6");
|
||||
if (!model) throw new Error("Expected bundled Claude Sonnet 4.6 model");
|
||||
|
||||
@@ -71,7 +71,7 @@ describe("autolearn tool gating", () => {
|
||||
expect(noBackend).not.toContain("learn");
|
||||
});
|
||||
|
||||
it("excludes the tools from a subagent even with an explicit list", async () => {
|
||||
it("excludes the tools from a subagent when not in the explicit list", async () => {
|
||||
// taskDepth > 0: the controller never runs here, so a subagent's explicit
|
||||
// whitelist must not be silently widened with write-capable tools.
|
||||
const sub = (
|
||||
@@ -90,6 +90,18 @@ describe("autolearn tool gating", () => {
|
||||
expect(subDiscovered).not.toContain("learn");
|
||||
});
|
||||
|
||||
it("allows the tools in a subagent when explicitly requested in toolNames", async () => {
|
||||
// Frontmatter tools: list overrides the taskDepth gate.
|
||||
const sub = (
|
||||
await createTools(makeSession({ "autolearn.enabled": true, "memory.backend": "mnemopi" }, { taskDepth: 1 }), [
|
||||
"manage_skill",
|
||||
"learn",
|
||||
])
|
||||
).map(t => t.name);
|
||||
expect(sub).toContain("manage_skill");
|
||||
expect(sub).toContain("learn");
|
||||
});
|
||||
|
||||
it("offers learn with the file-based local backend", async () => {
|
||||
const names = (await createTools(makeSession({ "autolearn.enabled": true, "memory.backend": "local" }))).map(
|
||||
t => t.name,
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import * as url from "node:url";
|
||||
import { __buildLegacyPiPackageRootOverrides } from "@oh-my-pi/pi-coding-agent/extensibility/plugins/legacy-pi-compat";
|
||||
@@ -83,6 +84,60 @@ process.stdout.write(JSON.stringify([
|
||||
expect(bundledModuleKeys.has("@oh-my-pi/pi-ai/oauth/openai-codex")).toBe(true);
|
||||
});
|
||||
|
||||
it("actually loads the shim's shared Pi translation through the bundled registry", async () => {
|
||||
// The legacy shim performs the same Pi arg translation as the modern
|
||||
// bridge and imports the shared helpers rather than copying them. Those
|
||||
// live in a single-segment `providers/` module on purpose: `./providers/*`
|
||||
// cannot match a nested `providers/<dir>/<mod>` specifier, which would
|
||||
// fall through to `Bun.resolveSync` and fail under bunfs (issue #3442).
|
||||
//
|
||||
// Executing the generated registry is the contract — a key present in the
|
||||
// override map still proves nothing if the module cannot be imported.
|
||||
const key = "@oh-my-pi/pi-ai/providers/cursor-pi-args";
|
||||
const entry = (await collectBundledPiEntries()).find(candidate => candidate.key === key);
|
||||
expect(entry).toBeDefined();
|
||||
|
||||
// The rendered registry imports by bare specifier, exactly as the real
|
||||
// bundle does, so it must run somewhere those specifiers resolve — the
|
||||
// package itself. A temp dir has no workspace links and would fail for
|
||||
// a reason unrelated to the export map.
|
||||
const packageRoot = path.join(path.dirname(url.fileURLToPath(import.meta.url)), "..", "..");
|
||||
const registryPath = path.join(packageRoot, `.probe-legacy-pi-args-${Bun.randomUUIDv7()}.ts`);
|
||||
await Bun.write(
|
||||
registryPath,
|
||||
`${__renderLegacyPiVirtualModule([entry!])}
|
||||
const mod = await BUNDLED_PI_MODULE_LOADERS[${JSON.stringify(key)}]();
|
||||
process.stdout.write(JSON.stringify([
|
||||
mod.piEscapeRegexLiteral("a.b*c"),
|
||||
mod.piJoinPath("src", "*.ts"),
|
||||
]));
|
||||
`,
|
||||
);
|
||||
let exitCode: number;
|
||||
let stdout: string;
|
||||
let stderr: string;
|
||||
try {
|
||||
const proc = Bun.spawn([process.execPath, registryPath], {
|
||||
cwd: packageRoot,
|
||||
stdout: "pipe",
|
||||
stderr: "pipe",
|
||||
});
|
||||
[exitCode, stdout, stderr] = await Promise.all([
|
||||
proc.exited,
|
||||
new Response(proc.stdout).text(),
|
||||
new Response(proc.stderr).text(),
|
||||
]);
|
||||
} finally {
|
||||
await fs.rm(registryPath, { force: true });
|
||||
}
|
||||
expect(stderr).toBe("");
|
||||
expect(exitCode).toBe(0);
|
||||
expect(JSON.parse(stdout)).toEqual(["a\\.b\\*c", path.join("src", "*.ts")]);
|
||||
|
||||
const overrides = __buildLegacyPiPackageRootOverrides(true, bundledModuleKeys);
|
||||
expect(overrides[key]).toBe(`omp-legacy-pi-bundled:${key}`);
|
||||
});
|
||||
|
||||
it("expands web search provider wildcard exports for compiled plugin imports", () => {
|
||||
const overrides = __buildLegacyPiPackageRootOverrides(true, bundledModuleKeys);
|
||||
const providerKeys = [
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/runner";
|
||||
import type { ExtensionRuntime } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/types";
|
||||
import type { AsyncJobSnapshot } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
|
||||
function createRunner(getAsyncJobSnapshot?: () => AsyncJobSnapshot | null): ExtensionRunner {
|
||||
const runtime = {
|
||||
flagValues: new Map(),
|
||||
pendingProviderRegistrations: [],
|
||||
} as unknown as ExtensionRuntime;
|
||||
return new ExtensionRunner(
|
||||
[],
|
||||
runtime,
|
||||
"/tmp",
|
||||
{} as never,
|
||||
{} as never,
|
||||
undefined,
|
||||
undefined,
|
||||
undefined,
|
||||
getAsyncJobSnapshot,
|
||||
);
|
||||
}
|
||||
|
||||
describe("ExtensionRunner async job context", () => {
|
||||
it("defaults to null outside a session", () => {
|
||||
expect(createRunner().createContext().getAsyncJobSnapshot()).toBeNull();
|
||||
});
|
||||
|
||||
it("exposes the owning session snapshot", () => {
|
||||
const snapshot: AsyncJobSnapshot = {
|
||||
running: [{ id: "bg-1", type: "bash", status: "running", label: "sleep 30", startTime: 1 }],
|
||||
recent: [],
|
||||
delivery: { queued: 0, delivering: false, pendingJobIds: [] },
|
||||
};
|
||||
expect(
|
||||
createRunner(() => snapshot)
|
||||
.createContext()
|
||||
.getAsyncJobSnapshot(),
|
||||
).toBe(snapshot);
|
||||
});
|
||||
});
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user