diff --git a/Cargo.lock b/Cargo.lock index 79930c842..d04f2f533 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -549,9 +549,9 @@ dependencies = [ [[package]] name = "chrono" -version = "0.4.44" +version = "0.4.45" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "c673075a2e0e5f4a1dde27ce9dee1ea4558c7ffe648f576438a20ca1d2acc4b0" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" dependencies = [ "iana-time-zone", "js-sys", @@ -1527,9 +1527,9 @@ checksum = "b9e0384b61958566e926dc50660321d12159025e767c18e043daf26b70104c39" [[package]] name = "ignore" -version = "0.4.25" +version = "0.4.26" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d3d782a365a015e0f5c04902246139249abf769125006fbe7649e2ee88169b4a" +checksum = "b915661dd01db3f05050265b2477bcc6527b3792388e2749b41623cc592be67d" dependencies = [ "crossbeam-deque", "globset", @@ -1748,9 +1748,9 @@ dependencies = [ [[package]] name = "log" -version = "0.4.31" +version = "0.4.32" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "113b30b4cd05f7c06868fdb2854f66a7b9fece9a48425351cd532e810d74024f" +checksum = "953f07c43838f8e6f9758cab68bf5bed85465e7587ebe0b823f1bcd81978ad3a" [[package]] name = "lru" @@ -2331,7 +2331,7 @@ dependencies = [ [[package]] name = "pi-ast" -version = "15.9.1" +version = "15.9.2" dependencies = [ "anyhow", "ast-grep-core", @@ -2399,7 +2399,7 @@ dependencies = [ [[package]] name = "pi-iso" -version = "15.9.1" +version = "15.9.2" dependencies = [ "async-trait", "libc", @@ -2411,7 +2411,7 @@ dependencies = [ [[package]] name = "pi-natives" -version = "15.9.1" +version = "15.9.2" dependencies = [ "anyhow", "arboard", @@ -2457,7 +2457,7 @@ dependencies = [ [[package]] name = "pi-shell" -version = "15.9.1" +version = "15.9.2" dependencies = [ "anyhow", "brush-builtins", diff --git a/Cargo.toml b/Cargo.toml index 02f333bbf..68aa16696 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"] resolver = "3" [workspace.package] -version = "15.9.1" +version = "15.9.2" edition = "2024" license = "MIT" authors = ["Can Boluk"] diff --git a/bun.lock b/bun.lock index ebffae31e..538d79fd5 100644 --- a/bun.lock +++ b/bun.lock @@ -15,7 +15,7 @@ }, "packages/agent": { "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@oh-my-pi/pi-ai": "catalog:", "@oh-my-pi/pi-natives": "catalog:", @@ -30,7 +30,7 @@ }, "packages/ai": { "name": "@oh-my-pi/pi-ai", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@bufbuild/protobuf": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -44,7 +44,7 @@ }, "packages/coding-agent": { "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.1", + "version": "15.9.2", "bin": { "omp": "src/cli.ts", }, @@ -90,7 +90,7 @@ }, "packages/hashline": { "name": "@oh-my-pi/hashline", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "diff": "catalog:", "lru-cache": "catalog:", @@ -101,7 +101,7 @@ }, "packages/mnemopi": { "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.1", + "version": "15.9.2", "bin": { "mnemopi": "src/cli.ts", }, @@ -118,7 +118,7 @@ }, "packages/natives": { "name": "@oh-my-pi/pi-natives", - "version": "15.9.1", + "version": "15.9.2", "devDependencies": { "@napi-rs/cli": "catalog:", "@types/bun": "catalog:", @@ -126,7 +126,7 @@ }, "packages/stats": { "name": "@oh-my-pi/omp-stats", - "version": "15.9.1", + "version": "15.9.2", "bin": { "omp-stats": "./src/index.ts", }, @@ -151,7 +151,7 @@ }, "packages/swarm-extension": { "name": "@oh-my-pi/swarm-extension", - "version": "15.9.1", + "version": "15.9.2", "bin": { "omp-swarm": "src/cli.ts", }, @@ -167,7 +167,7 @@ }, "packages/tui": { "name": "@oh-my-pi/pi-tui", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "@oh-my-pi/pi-utils": "catalog:", @@ -208,7 +208,7 @@ }, "packages/utils": { "name": "@oh-my-pi/pi-utils", - "version": "15.9.1", + "version": "15.9.2", "dependencies": { "@oh-my-pi/pi-natives": "catalog:", "beautiful-mermaid": "catalog:", @@ -248,15 +248,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.1", - "@oh-my-pi/omp-stats": "15.9.1", - "@oh-my-pi/pi-agent-core": "15.9.1", - "@oh-my-pi/pi-ai": "15.9.1", - "@oh-my-pi/pi-coding-agent": "15.9.1", - "@oh-my-pi/pi-mnemopi": "15.9.1", - "@oh-my-pi/pi-natives": "15.9.1", - "@oh-my-pi/pi-tui": "15.9.1", - "@oh-my-pi/pi-utils": "15.9.1", + "@oh-my-pi/hashline": "15.9.2", + "@oh-my-pi/omp-stats": "15.9.2", + "@oh-my-pi/pi-agent-core": "15.9.2", + "@oh-my-pi/pi-ai": "15.9.2", + "@oh-my-pi/pi-coding-agent": "15.9.2", + "@oh-my-pi/pi-mnemopi": "15.9.2", + "@oh-my-pi/pi-natives": "15.9.2", + "@oh-my-pi/pi-tui": "15.9.2", + "@oh-my-pi/pi-utils": "15.9.2", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", @@ -787,7 +787,7 @@ "@types/node": ["@types/node@25.9.1", "", { "dependencies": { "undici-types": ">=7.24.0 <7.24.7" } }, "sha512-xfrlY7UD5rMJk3ZVJP8BNzS28J36YJg+xp+LPXV1TdWxr8uMH5A860QNxYDGQe/ylDSgjxE52Q9VnO7p75tJxg=="], - "@types/react": ["@types/react@19.2.15", "", { "dependencies": { "csstype": "^3.2.2" } }, "sha512-eRwcGNHve+E8qtEQSSRl6urh+rFop4v8gm6O8rGv25CodbvFdLjA1vVQ1KkiFE0w0UPOnb8tDiFKL5lp0rtY5Q=="], + "@types/react": ["@types/react@19.2.16", "", { "dependencies": { "csstype": "^3.2.2" } }, "sha512-esJiCAnl0kfpNdE69f3So4WJUXy95dLZydX0KwK46riIHDzHM7O9Vtf9xCHW0PXIqvgqNrswl522kA/5yx+F4w=="], "@types/react-dom": ["@types/react-dom@19.2.3", "", { "peerDependencies": { "@types/react": "^19.2.0" } }, "sha512-jp2L/eY6fn+KgVVQAOqYItbF0VY/YApe5Mz2F0aykSO8gx31bYCZyvSeYxCHKvzHG5eZjc+zyaS5BrBWya2+kQ=="], @@ -1139,7 +1139,7 @@ "neo-async": ["neo-async@2.6.2", "", {}, "sha512-Yd3UES5mWCSqR+qNT93S3UoYUkqAZ9lLg8a7g9rimsWmYGK8cVToA4/sF3RrshdyV3sAGMXVUmpMYOw+dLpOuw=="], - "node-releases": ["node-releases@2.0.46", "", {}, "sha512-GYVXHE2KnrzAfsAjl4uP++evGFCrAU1jta4ubEjIG7YWt/64Gqv66a30yKwWczVjA6j3bM4nBwH7Pk1JmDHaxQ=="], + "node-releases": ["node-releases@2.0.47", "", {}, "sha512-Uzmd6LXpouKo8EUK68IjH4+E01w/hXyV3R3g/geCJo+rXLNfh1xucB+LOzYEOQPSiUK3h/xZf0cQGcSsmyL2Og=="], "nth-check": ["nth-check@2.1.1", "", { "dependencies": { "boolbase": "^1.0.0" } }, "sha512-lqjrjmaOoAnWfMmBPL+XNnynZh2+swxiX3WUE0s4yEHI6m+AwrK2UZOimIRl3X/4QctVqS8AiZjFqyOGrMXb/w=="], @@ -1259,7 +1259,7 @@ "string-width": ["string-width@4.2.3", "", { "dependencies": { "emoji-regex": "^8.0.0", "is-fullwidth-code-point": "^3.0.0", "strip-ansi": "^6.0.1" } }, "sha512-wKyQRQpjJ0sIp62ErSZdGsjMJWsap5oRNihHhu6G7JVO/9jIB6UyevL+tXuOqrng8j/cxKTWyWUwvSTriiZz/g=="], - "string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], + "string_decoder": ["string_decoder@1.3.0", "", { "dependencies": { "safe-buffer": "~5.2.0" } }, "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA=="], "strip-ansi": ["strip-ansi@7.2.0", "", { "dependencies": { "ansi-regex": "^6.2.2" } }, "sha512-yDPMNjp4WyfYBkHnjIRLfca1i6KMyGCtsVgoKe/z1+6vukgaENdgGBZt+ZmKPc4gavvEZ5OgHfHdrazhgNyG7w=="], @@ -1413,6 +1413,8 @@ "string-width/strip-ansi": ["strip-ansi@6.0.1", "", { "dependencies": { "ansi-regex": "^5.0.1" } }, "sha512-Y38VPSHcqkFrCpFnQ9vuSXmquuv5oXOKpGeT6aGrr3o3Gc9AlVa6JBfUSOCnbxGGZF+/0ooI7KrPuUSztUdU5A=="], + "string_decoder/safe-buffer": ["safe-buffer@5.2.1", "", {}, "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="], + "wrap-ansi/string-width": ["string-width@8.2.1", "", { "dependencies": { "get-east-asian-width": "^1.5.0", "strip-ansi": "^7.1.2" } }, "sha512-IIaP0g3iy9Cyy18w3M9YcaDudujEAVHKt3a3QJg1+sr/oX96TbaGUubG0hJyCjCBThFH+tFpcIyoUHUn1ogaLA=="], "xml2js/xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="], @@ -1427,6 +1429,8 @@ "fastembed/onnxruntime-node/tar": ["tar@7.5.16", "", { "dependencies": { "@isaacs/fs-minipass": "^4.0.0", "chownr": "^3.0.0", "minipass": "^7.1.2", "minizlib": "^3.1.0", "yallist": "^5.0.0" } }, "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w=="], + "jszip/readable-stream/string_decoder": ["string_decoder@1.1.1", "", { "dependencies": { "safe-buffer": "~5.1.0" } }, "sha512-n/ShnvDi6FHbbVfviro+WojiFzv+s8MPMHBczVePfUpDJLwoLT0ht1l4YwBCbi8pJAveEEdnkHyPyTP/mzRfwg=="], + "log-update/slice-ansi/is-fullwidth-code-point": ["is-fullwidth-code-point@5.1.0", "", { "dependencies": { "get-east-asian-width": "^1.3.1" } }, "sha512-5XHYaSyiqADb4RnZ1Bdad6cPp8Toise4TzEjcOYDHZkTCbKgiUl7WTUCpNWHuxmDt91wnsZBc9xinNzopv3JMQ=="], "log-update/wrap-ansi/string-width": ["string-width@7.2.0", "", { "dependencies": { "emoji-regex": "^10.3.0", "get-east-asian-width": "^1.0.0", "strip-ansi": "^7.1.0" } }, "sha512-tsaTIkKW9b4N+AEj+SVA+WhJzV7/zMhcSu78mLKWSk7cXMOSHsBKFWUs0fWwq8QyK3MgJBQRX6Gbi4kYbdvGkQ=="], diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 4ed19386d..231322545 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -68,5 +68,5 @@ use napi_derive::napi; /// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in /// `packages/natives/native/index.js` (which derives the name from /// `package.json#version`). -#[napi(js_name = "__piNativesV15_9_1")] +#[napi(js_name = "__piNativesV15_9_2")] pub const fn pi_natives_version_sentinel() {} diff --git a/docs/auth-broker-gateway.md b/docs/auth-broker-gateway.md index ec7948d1b..1fccca164 100644 --- a/docs/auth-broker-gateway.md +++ b/docs/auth-broker-gateway.md @@ -140,6 +140,14 @@ When the gateway (or any other broker client) calls `fetchUsageReports()` / `get The 15 s client window deliberately sits below the broker’s 5 min server cache, so almost every client poll is served from the broker’s already-cached value; the client cache exists to absorb the parallel fan-out generated by `AuthStorage.#rankOAuthSelections` into a single broker round-trip. +## Client snapshot cache + +`discoverAuthStorage()` persists the broker snapshot to `~/.omp/cache/auth-broker-snapshot.enc` after the initial `/v1/snapshot` fetch and after later broker-sourced full snapshots. The file is AES-256-GCM encrypted with `SHA-256(OMP_AUTH_BROKER_TOKEN)` and authenticated with the broker URL as additional data, so changing either the token or URL makes the cache unreadable. The file is written atomically with mode `0600`. + +Freshness is anchored to the broker-stamped `snapshot.generatedAt`, not local write time. Default TTL is 1 h (`OMP_AUTH_BROKER_SNAPSHOT_TTL_MS`); `0` disables the cache and restores the old always-fetch boot path. When the cached snapshot is still fresh, `omp` boots from it and skips the blocking `/v1/snapshot` query. `RemoteAuthCredentialStore` still starts its normal SSE / long-poll background sync immediately, so deleted or rotated credentials reconcile after startup, and expired OAuth access tokens still refresh through `POST /v1/credential/:id/refresh`. + +If the broker is down at boot and a fresh cache exists, startup now succeeds from the cached snapshot. If the cache is missing, expired, corrupt, written for a different URL, or encrypted with a different token, startup falls back to the live fetch and fails the same way it did before if the broker is unreachable. + ## Operator opt-in The broker is **off** unless `OMP_AUTH_BROKER_URL` (or `auth.broker.url` in `config.yml`) is set. When set, `discoverAuthStorage` in `packages/coding-agent/src/sdk.ts` swaps the local SQLite credential store for `RemoteAuthCredentialStore` and every API call resolves credentials through the broker. @@ -150,6 +158,8 @@ The broker is **off** unless `OMP_AUTH_BROKER_URL` (or `auth.broker.url` in `con | ----------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | | `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`). Selecting this puts the client in broker mode — local SQLite is bypassed. | Any time the omp client should resolve credentials through a broker (and required by `omp auth-gateway serve`). | | `OMP_AUTH_BROKER_TOKEN` | Bearer token used for every broker endpoint except `/v1/healthz`. | When `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `/auth-broker.token`. | +| `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` | Freshness window for the encrypted local snapshot cache. Default `3600000` (1 h); `0` disables cache reads and writes. | Optional in broker mode. | +| `OMP_AUTH_BROKER_SNAPSHOT_CACHE` | Path override for the encrypted local snapshot cache. Default `~/.omp/cache/auth-broker-snapshot.enc` (or XDG cache equivalent). | Optional in broker mode. | Resolution order in `resolveAuthBrokerConfig()`: diff --git a/docs/environment-variables.md b/docs/environment-variables.md index 131d9170c..e556301e8 100644 --- a/docs/environment-variables.md +++ b/docs/environment-variables.md @@ -97,6 +97,8 @@ When the broker is enabled, the local SQLite credential store is bypassed and al | ----------------------- | -------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | | `OMP_AUTH_BROKER_URL` | Base URL of the remote auth-broker (e.g. `https://broker.tailnet:8765`); selects broker mode | Resolving credentials through a broker; also required by `omp auth-gateway serve` (the gateway is itself a broker client) | Wins over `auth.broker.url` in `config.yml`. When set with no resolvable token, `resolveAuthBrokerConfig()` hard-errors instead of falling back to local SQLite. | | `OMP_AUTH_BROKER_TOKEN` | Bearer token sent on every broker endpoint except `/v1/healthz` | `OMP_AUTH_BROKER_URL` is set and no token is available from `auth.broker.token` or `/auth-broker.token` | Resolution: this env → `auth.broker.token` (`$ENV_NAME` indirection supported) → `/auth-broker.token` (mode `0600`). `` is `~/.omp/` (respecting `PI_CONFIG_DIR`). | +| `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` | Freshness window for the encrypted local broker snapshot cache | Optional in broker mode | Default `3600000` (1 h). Freshness is based on broker `snapshot.generatedAt`; `0` disables cache reads/writes and forces the old blocking fetch every startup. | +| `OMP_AUTH_BROKER_SNAPSHOT_CACHE` | Path to the encrypted local broker snapshot cache | Optional in broker mode | Defaults to `~/.omp/cache/auth-broker-snapshot.enc` (or XDG cache equivalent). Useful for tests, ephemeral hosts, or relocating the `0600` cache file. | The gateway has no dedicated env vars — it inherits `OMP_AUTH_BROKER_*`. Its own inbound bearer token lives at `/auth-gateway.token` and is managed via `omp auth-gateway token`. diff --git a/docs/skills.md b/docs/skills.md index 845671d16..beca954bd 100644 --- a/docs/skills.md +++ b/docs/skills.md @@ -89,6 +89,7 @@ Current registered skill providers: - `agents` - `codex` 5. `opencode` (priority 55) +6. `github` (priority 30) — `.github/skills//SKILL.md` (GitHub Agent Skills layout, project-only) Dedup key is skill name. First item with a given name wins. diff --git a/docs/task-agent-discovery.md b/docs/task-agent-discovery.md index 502bb482e..82156f310 100644 --- a/docs/task-agent-discovery.md +++ b/docs/task-agent-discovery.md @@ -24,7 +24,7 @@ It covers runtime behavior as implemented today, including precedence, invalid-d Task agents normalize into `AgentDefinition` (`src/task/types.ts`): - `name`, `description`, `systemPrompt` (required for a valid loaded agent) -- optional `tools`, `spawns`, `model`, `thinkingLevel`, `output`, `blocking` +- optional `tools`, `spawns`, `model`, `thinkingLevel`, `output`, `blocking`, `autoloadSkills`, `readSummarize` - `source`: `"bundled" | "user" | "project"` - optional `filePath` @@ -35,6 +35,7 @@ Parsing comes from frontmatter via `parseAgentFields()` (`src/discovery/helpers. - `spawns` accepts `*`, CSV, or array - backward-compat behavior: if `spawns` missing but `tools` includes `task`, `spawns` becomes `*` - `output` is passed through as opaque schema data +- `read-summarize: false` (parsed as `readSummarize`) forces the subagent's `read` tool to return verbatim file content instead of structural summaries — `runSubprocess` applies it as a `read.summarize.enabled: false` override on the subagent's isolated settings (`src/task/executor.ts`). `explore` and `librarian` ship with it disabled. Defaults to enabled when the field is absent. ## Bundled agents diff --git a/package.json b/package.json index 4cce29fa6..a34d1e667 100644 --- a/package.json +++ b/package.json @@ -20,15 +20,15 @@ "@huggingface/transformers": "^4.2.0", "@mozilla/readability": "^0.6.0", "@napi-rs/cli": "3.7.0", - "@oh-my-pi/hashline": "15.9.1", - "@oh-my-pi/omp-stats": "15.9.1", - "@oh-my-pi/pi-agent-core": "15.9.1", - "@oh-my-pi/pi-ai": "15.9.1", - "@oh-my-pi/pi-coding-agent": "15.9.1", - "@oh-my-pi/pi-mnemopi": "15.9.1", - "@oh-my-pi/pi-natives": "15.9.1", - "@oh-my-pi/pi-tui": "15.9.1", - "@oh-my-pi/pi-utils": "15.9.1", + "@oh-my-pi/hashline": "15.9.2", + "@oh-my-pi/omp-stats": "15.9.2", + "@oh-my-pi/pi-agent-core": "15.9.2", + "@oh-my-pi/pi-ai": "15.9.2", + "@oh-my-pi/pi-coding-agent": "15.9.2", + "@oh-my-pi/pi-mnemopi": "15.9.2", + "@oh-my-pi/pi-natives": "15.9.2", + "@oh-my-pi/pi-tui": "15.9.2", + "@oh-my-pi/pi-utils": "15.9.2", "@opentelemetry/api": "^1.9.1", "@opentelemetry/context-async-hooks": "^2.7.1", "@opentelemetry/exporter-trace-otlp-proto": "^0.218.0", diff --git a/packages/agent/package.json b/packages/agent/package.json index d58b4a8fb..9eb31590f 100644 --- a/packages/agent/package.json +++ b/packages/agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-agent-core", - "version": "15.9.1", + "version": "15.9.2", "description": "General-purpose agent with transport abstraction, state management, and attachment support", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 2789f64c8..83afdf41e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,21 @@ ## [Unreleased] +## [15.9.2] - 2026-06-05 + +### Added + +- Added an AES-256-GCM auth-broker snapshot cache module and `RemoteAuthCredentialStoreOptions.onSnapshot` so broker clients can persist broker-sourced full snapshots without blocking startup on every run. +- Added `Model.omitMaxOutputTokens` so providers (notably Ollama proxies fronting cloud catalogs) can suppress `max_output_tokens` (Responses) and `max_tokens`/`max_completion_tokens` (Completions) on the wire while still using the catalog `maxTokens` for local budgeting. Without it, `applyCommonResponsesSamplingParams` unconditionally sent the catalog cap and HTTP-400'd against upstream APIs whose true output limit was unknown to OMP. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881)) + +### Changed + +- Changed usage-ranked OAuth credential selection to pick deterministic session-sticky weighted buckets instead of always choosing the top-ranked account, capping the best account at 2x the baseline session likelihood while keeping equal-priority accounts evenly balanced. + +### Fixed + +- Fixed parallel `function_call` items on the OpenAI Responses API losing arguments on every call except the last when the upstream server interleaves their stream events (observed against llama.cpp and other local Responses-compat hosts). `processResponsesStream` no longer routes `function_call_arguments.{delta,done}`, `output_item.done`, content_part/text/refusal/reasoning events through a singleton `currentItem`/`currentBlock` reference; it now tracks every open item in registries keyed by `output_index` and `item_id` so each event is folded into the matching block and the emitted `toolcall_end` carries the correct `contentIndex`. ([#1880](https://github.com/can1357/oh-my-pi/issues/1880)) + ## [15.9.1] - 2026-06-04 ### Added diff --git a/packages/ai/package.json b/packages/ai/package.json index f80274b8b..a68ebf4c0 100644 --- a/packages/ai/package.json +++ b/packages/ai/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-ai", - "version": "15.9.1", + "version": "15.9.2", "description": "Unified LLM API with automatic model discovery and provider configuration", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/ai/src/auth-broker/index.ts b/packages/ai/src/auth-broker/index.ts index 4858fbfdf..189d377c5 100644 --- a/packages/ai/src/auth-broker/index.ts +++ b/packages/ai/src/auth-broker/index.ts @@ -2,4 +2,5 @@ export * from "./client"; export * from "./refresher"; export * from "./remote-store"; export * from "./server"; +export * from "./snapshot-cache"; export * from "./types"; diff --git a/packages/ai/src/auth-broker/remote-store.ts b/packages/ai/src/auth-broker/remote-store.ts index 27388cf66..c0f9f17a3 100644 --- a/packages/ai/src/auth-broker/remote-store.ts +++ b/packages/ai/src/auth-broker/remote-store.ts @@ -73,11 +73,17 @@ export interface RemoteAuthCredentialStoreOptions { * to long-poll permanently when the broker returns 404. Default `true`. */ streamSnapshots?: boolean; + /** + * Called after broker-sourced full snapshots are applied. The constructor's + * initial snapshot intentionally does not trigger this hook. + */ + onSnapshot?: (snapshot: SnapshotResponse, generation: number) => void; } export class RemoteAuthCredentialStore implements AuthCredentialStore { readonly #client: AuthBrokerClient; readonly #streamSnapshots: boolean; + readonly #onSnapshot?: (snapshot: SnapshotResponse, generation: number) => void; #snapshot: SnapshotResponse = emptySnapshot(); #snapshotReceivedAt = Date.now(); #generation = 0; @@ -100,6 +106,7 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { this.#client = opts.client; this.#streamSnapshots = opts.streamSnapshots ?? true; this.#applySnapshot(opts.initialSnapshot ?? emptySnapshot(), opts.initialSnapshot?.generation ?? 0); + this.#onSnapshot = opts.onSnapshot; void this.#runBackground(); } @@ -115,6 +122,13 @@ export class RemoteAuthCredentialStore implements AuthCredentialStore { this.#snapshot = snapshot; this.#generation = generation; this.#snapshotReceivedAt = Date.now(); + const onSnapshot = this.#onSnapshot; + if (!onSnapshot) return; + try { + onSnapshot(snapshot, generation); + } catch (error) { + logger.debug("auth-broker snapshot callback failed", { error: String(error) }); + } } async #runBackground(): Promise { diff --git a/packages/ai/src/auth-broker/snapshot-cache.ts b/packages/ai/src/auth-broker/snapshot-cache.ts new file mode 100644 index 000000000..db806e185 --- /dev/null +++ b/packages/ai/src/auth-broker/snapshot-cache.ts @@ -0,0 +1,174 @@ +/** + * AES-GCM encrypted local cache for auth-broker snapshots. + * + * The cache is defense-in-depth for at-rest snapshots: a copied cache file is + * useless without the matching broker bearer token and URL. The token itself is + * still the trust boundary; a process that can read both the token and this file + * can decrypt the snapshot. + */ +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { isEnoent, logger } from "@oh-my-pi/pi-utils"; +import type { SnapshotResponse } from "./types"; +import { snapshotResponseSchema } from "./wire-schemas"; + +const MAGIC = new Uint8Array([0x4f, 0x4d, 0x50, 0x53]); // "OMPS" +const VERSION = 1; +const VERSION_OFFSET = MAGIC.byteLength; +const IV_OFFSET = VERSION_OFFSET + 1; +const IV_LENGTH = 12; +const HEADER_LENGTH = IV_OFFSET + IV_LENGTH; +const AES_ALGORITHM = "AES-GCM"; +const TEXT_ENCODER = new TextEncoder(); +const TEXT_DECODER = new TextDecoder(); +const HEX = "0123456789abcdef"; + +export interface ReadAuthBrokerSnapshotCacheOptions { + path: string; + token: string; + url: string; + ttlMs: number; + /** Override clock for deterministic tests. */ + now?: () => number; +} + +export interface WriteAuthBrokerSnapshotCacheOptions { + path: string; + token: string; + url: string; + snapshot: SnapshotResponse; +} + +export async function readAuthBrokerSnapshotCache( + opts: ReadAuthBrokerSnapshotCacheOptions, +): Promise { + if (opts.ttlMs <= 0) return null; + let data: Uint8Array; + try { + data = await fs.readFile(opts.path); + } catch (error) { + if (isEnoent(error)) return null; + throw error; + } + + try { + const plaintext = await decryptCachePayload(data, opts.token, opts.url); + if (!plaintext) return null; + const parsed: unknown = JSON.parse(TEXT_DECODER.decode(plaintext)); + const result = snapshotResponseSchema.safeParse(parsed); + if (!result.success) { + logger.debug("auth-broker snapshot cache schema invalid", { path: opts.path }); + return null; + } + const snapshot = result.data; + const now = opts.now?.() ?? Date.now(); + if (now - snapshot.generatedAt > opts.ttlMs) return null; + return snapshot; + } catch (error) { + logger.debug("auth-broker snapshot cache read failed", { path: opts.path, error: String(error) }); + return null; + } +} + +export async function writeAuthBrokerSnapshotCache(opts: WriteAuthBrokerSnapshotCacheOptions): Promise { + const payload = await encryptCachePayload(opts.snapshot, opts.token, opts.url); + await fs.mkdir(path.dirname(opts.path), { recursive: true }); + const tmpPath = `${opts.path}.${process.pid}.${randomHex(8)}.tmp`; + let removeTemp = false; + try { + const handle = await fs.open(tmpPath, "wx", 0o600); + removeTemp = true; + try { + await handle.writeFile(payload); + } finally { + await handle.close(); + } + await fs.chmod(tmpPath, 0o600); + await fs.rename(tmpPath, opts.path); + removeTemp = false; + } finally { + if (removeTemp) await fs.rm(tmpPath, { force: true }).catch(() => {}); + } +} + +async function encryptCachePayload(snapshot: SnapshotResponse, token: string, url: string): Promise { + const key = await deriveAesKey(token, ["encrypt"]); + const iv = new Uint8Array(IV_LENGTH); + globalThis.crypto.getRandomValues(iv); + const plaintext = TEXT_ENCODER.encode(JSON.stringify(snapshot)); + const ciphertext = new Uint8Array( + await globalThis.crypto.subtle.encrypt( + { + name: AES_ALGORITHM, + iv, + additionalData: TEXT_ENCODER.encode(url), + }, + key, + plaintext, + ), + ); + const payload = new Uint8Array(HEADER_LENGTH + ciphertext.byteLength); + payload.set(MAGIC, 0); + payload[VERSION_OFFSET] = VERSION; + payload.set(iv, IV_OFFSET); + payload.set(ciphertext, HEADER_LENGTH); + return payload; +} + +async function decryptCachePayload(data: Uint8Array, token: string, url: string): Promise { + if (data.byteLength <= HEADER_LENGTH) { + logger.debug("auth-broker snapshot cache file too short"); + return null; + } + for (let i = 0; i < MAGIC.byteLength; i++) { + if (data[i] !== MAGIC[i]) { + logger.debug("auth-broker snapshot cache magic mismatch"); + return null; + } + } + if (data[VERSION_OFFSET] !== VERSION) { + logger.debug("auth-broker snapshot cache version mismatch", { version: data[VERSION_OFFSET] }); + return null; + } + const key = await deriveAesKey(token, ["decrypt"]); + const iv = asStrict(data.subarray(IV_OFFSET, HEADER_LENGTH)); + const ciphertext = asStrict(data.subarray(HEADER_LENGTH)); + try { + return new Uint8Array( + await globalThis.crypto.subtle.decrypt( + { + name: AES_ALGORITHM, + iv, + additionalData: TEXT_ENCODER.encode(url), + }, + key, + ciphertext, + ), + ); + } catch (error) { + logger.debug("auth-broker snapshot cache decrypt failed", { error: String(error) }); + return null; + } +} + +async function deriveAesKey(token: string, usages: Array<"encrypt" | "decrypt">): Promise { + const digest = await globalThis.crypto.subtle.digest("SHA-256", TEXT_ENCODER.encode(token)); + return globalThis.crypto.subtle.importKey("raw", digest, AES_ALGORITHM, false, usages); +} + +function asStrict(bytes: Uint8Array): Uint8Array { + if (bytes.buffer instanceof ArrayBuffer && bytes.byteOffset === 0 && bytes.byteLength === bytes.buffer.byteLength) { + return bytes as Uint8Array; + } + const copy = new Uint8Array(bytes.byteLength); + copy.set(bytes); + return copy; +} + +function randomHex(byteLength: number): string { + const bytes = new Uint8Array(byteLength); + globalThis.crypto.getRandomValues(bytes); + let out = ""; + for (const byte of bytes) out += HEX[byte >> 4] + HEX[byte & 15]; + return out; +} diff --git a/packages/ai/src/auth-broker/types.ts b/packages/ai/src/auth-broker/types.ts index 3cfd1bbc8..1387b6473 100644 --- a/packages/ai/src/auth-broker/types.ts +++ b/packages/ai/src/auth-broker/types.ts @@ -117,6 +117,9 @@ export const DEFAULT_REFRESH_SKEW_MS = 5 * 60_000; /** Default broker refresh-loop cadence. */ export const DEFAULT_REFRESH_INTERVAL_MS = 60_000; +/** Default freshness window for the encrypted local broker snapshot cache. */ +export const DEFAULT_SNAPSHOT_CACHE_TTL_MS = 60 * 60_000; + /** Keepalive cadence for `GET /v1/snapshot/stream` SSE comments. */ export const DEFAULT_STREAM_KEEPALIVE_MS = 20_000; diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index 907c7b7f0..f03f7b810 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -557,11 +557,23 @@ type OAuthResolutionResult = { apiKey: string; credential: OAuthCredential }; */ export interface OAuthAccess { accessToken: string; + credentialId?: number; accountId?: string; email?: string; projectId?: string; enterpriseUrl?: string; } + +export interface OAuthAccessFailure { + credentialId?: number; + accountId?: string; + email?: string; + projectId?: string; + enterpriseUrl?: string; + error: string; +} + +export type OAuthAccessResolution = ({ ok: true } & OAuthAccess) | ({ ok: false } & OAuthAccessFailure); export interface InvalidateCredentialMatchingOptions { signal?: AbortSignal; sessionId?: string; @@ -725,6 +737,25 @@ class AuthStorageUsageCache implements UsageCache { // ───────────────────────────────────────────────────────────────────────────── type StoredCredential = { id: number; credential: AuthCredential }; +type OAuthSelection = { credential: OAuthCredential; index: number }; + +type OAuthCandidate = { + selection: OAuthSelection; + usage: UsageReport | null; + usageChecked: boolean; +}; + +type RankedOAuthCandidate = OAuthCandidate & { + blocked: boolean; + blockedUntil?: number; + hasPriorityBoost: boolean; + planPriority: number; + secondaryUsed: number; + secondaryDrainRate: number; + primaryUsed: number; + primaryDrainRate: number; + orderPos: number; +}; // ───────────────────────────────────────────────────────────────────────────── // AuthStorage Class @@ -2733,35 +2764,126 @@ export class AuthStorage { return usedFraction / elapsedHours; } + #compareRankedOAuthCandidatePriority( + left: RankedOAuthCandidate, + right: RankedOAuthCandidate, + provider: string, + modelId: string | undefined, + ): number { + if (left.blocked !== right.blocked) return left.blocked ? 1 : -1; + if (left.blocked && right.blocked) { + const leftBlockedUntil = left.blockedUntil ?? Number.POSITIVE_INFINITY; + const rightBlockedUntil = right.blockedUntil ?? Number.POSITIVE_INFINITY; + if (leftBlockedUntil !== rightBlockedUntil) return leftBlockedUntil - rightBlockedUntil; + return 0; + } + if (requiresOpenAICodexProModel(provider, modelId) && left.planPriority !== right.planPriority) { + return left.planPriority - right.planPriority; + } + if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; + if (left.secondaryDrainRate !== right.secondaryDrainRate) { + return left.secondaryDrainRate - right.secondaryDrainRate; + } + if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed; + if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate; + if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed; + return 0; + } + + #compareRankedOAuthCandidates( + left: RankedOAuthCandidate, + right: RankedOAuthCandidate, + provider: string, + modelId: string | undefined, + ): number { + const priority = this.#compareRankedOAuthCandidatePriority(left, right, provider, modelId); + return priority !== 0 ? priority : left.orderPos - right.orderPos; + } + + #orderRankedOAuthCandidates( + candidates: RankedOAuthCandidate[], + sessionId: string | undefined, + provider: string, + modelId: string | undefined, + ): OAuthCandidate[] { + candidates.sort((left, right) => this.#compareRankedOAuthCandidates(left, right, provider, modelId)); + if (!sessionId) { + return candidates.map(candidate => ({ + selection: candidate.selection, + usage: candidate.usage, + usageChecked: candidate.usageChecked, + })); + } + + const unblocked = candidates.filter(candidate => !candidate.blocked); + if (unblocked.length <= 1) { + return candidates.map(candidate => ({ + selection: candidate.selection, + usage: candidate.usage, + usageChecked: candidate.usageChecked, + })); + } + + const priorityByCandidate = new Map(); + let bucketIndex = 0; + let previous = unblocked[0]; + const bucketByCandidate = new Map(); + for (const candidate of unblocked) { + if ( + candidate !== previous && + this.#compareRankedOAuthCandidatePriority(previous, candidate, provider, modelId) !== 0 + ) { + bucketIndex += 1; + } + bucketByCandidate.set(candidate, bucketIndex); + previous = candidate; + } + const maxBucket = bucketIndex; + for (const candidate of unblocked) { + const bucket = bucketByCandidate.get(candidate) ?? 0; + priorityByCandidate.set(candidate, maxBucket === 0 ? 0 : 1 - bucket / maxBucket); + } + + let totalWeight = 0; + for (const candidate of unblocked) { + totalWeight += 1 + (priorityByCandidate.get(candidate) ?? 0); + } + + const hit = ((Bun.hash.xxHash32(sessionId) >>> 0) / 2 ** 32) * totalWeight; + let cursor = 0; + let selected = unblocked[unblocked.length - 1]; + for (const candidate of unblocked) { + cursor += 1 + (priorityByCandidate.get(candidate) ?? 0); + if (hit < cursor) { + selected = candidate; + break; + } + } + + const ordered = [ + selected, + ...unblocked.filter(candidate => candidate !== selected), + ...candidates.filter(candidate => candidate.blocked), + ]; + return ordered.map(candidate => ({ + selection: candidate.selection, + usage: candidate.usage, + usageChecked: candidate.usageChecked, + })); + } + async #rankOAuthSelections(args: { providerKey: string; provider: string; order: number[]; - credentials: Array<{ credential: OAuthCredential; index: number }>; + credentials: OAuthSelection[]; options?: AuthApiKeyOptions; + sessionId?: string; strategy: CredentialRankingStrategy; - }): Promise< - Array<{ - selection: { credential: OAuthCredential; index: number }; - usage: UsageReport | null; - usageChecked: boolean; - }> - > { + }): Promise { const nowMs = Date.now(); const { strategy } = args; - const ranked: Array<{ - selection: { credential: OAuthCredential; index: number }; - usage: UsageReport | null; - usageChecked: boolean; - blocked: boolean; - blockedUntil?: number; - hasPriorityBoost: boolean; - secondaryUsed: number; - secondaryDrainRate: number; - primaryUsed: number; - primaryDrainRate: number; - orderPos: number; - }> = []; + const ranked: RankedOAuthCandidate[] = []; // Pre-fetch usage reports in parallel for non-blocked credentials. // Wrap with a timeout so slow/429'd fetches don't indefinitely block // credential selection — better to pick a credential without usage data @@ -2821,6 +2943,7 @@ export class AuthStorage { blocked, blockedUntil, hasPriorityBoost: strategy.hasPriorityBoost?.(primary) ?? false, + planPriority: getOpenAICodexPlanPriority(usage), secondaryUsed: this.#normalizeUsageFraction(secondaryTarget), secondaryDrainRate: this.#computeWindowDrainRate( secondaryTarget, @@ -2832,32 +2955,7 @@ export class AuthStorage { orderPos, }); } - ranked.sort((left, right) => { - if (left.blocked !== right.blocked) return left.blocked ? 1 : -1; - if (left.blocked && right.blocked) { - const leftBlockedUntil = left.blockedUntil ?? Number.POSITIVE_INFINITY; - const rightBlockedUntil = right.blockedUntil ?? Number.POSITIVE_INFINITY; - if (leftBlockedUntil !== rightBlockedUntil) return leftBlockedUntil - rightBlockedUntil; - return left.orderPos - right.orderPos; - } - if (requiresOpenAICodexProModel(args.provider, args.options?.modelId)) { - const leftPlanPriority = getOpenAICodexPlanPriority(left.usage); - const rightPlanPriority = getOpenAICodexPlanPriority(right.usage); - if (leftPlanPriority !== rightPlanPriority) return leftPlanPriority - rightPlanPriority; - } - if (left.hasPriorityBoost !== right.hasPriorityBoost) return left.hasPriorityBoost ? -1 : 1; - if (left.secondaryDrainRate !== right.secondaryDrainRate) - return left.secondaryDrainRate - right.secondaryDrainRate; - if (left.secondaryUsed !== right.secondaryUsed) return left.secondaryUsed - right.secondaryUsed; - if (left.primaryDrainRate !== right.primaryDrainRate) return left.primaryDrainRate - right.primaryDrainRate; - if (left.primaryUsed !== right.primaryUsed) return left.primaryUsed - right.primaryUsed; - return left.orderPos - right.orderPos; - }); - return ranked.map(candidate => ({ - selection: candidate.selection, - usage: candidate.usage, - usageChecked: candidate.usageChecked, - })); + return this.#orderRankedOAuthCandidates(ranked, args.sessionId, args.provider, args.options?.modelId); } /** @@ -2894,8 +2992,17 @@ export class AuthStorage { const sessionPreferredIsAvailable = sessionPreferredIndex !== undefined && !this.#isCredentialBlocked(providerKey, sessionPreferredIndex); const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel); + const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order; const candidates = shouldRank - ? await this.#rankOAuthSelections({ providerKey, provider, order, credentials, options, strategy: strategy! }) + ? await this.#rankOAuthSelections({ + providerKey, + provider, + order: rankingOrder, + credentials, + options, + sessionId, + strategy: strategy!, + }) : order .map(idx => credentials[idx]) .filter((selection): selection is { credential: OAuthCredential; index: number } => Boolean(selection)) @@ -3399,6 +3506,75 @@ export class AuthStorage { }; } + /** + * Resolve every stored OAuth credential for `provider` independently. + * + * Refreshes credentials through the same broker/local path as + * {@link AuthStorage.getOAuthAccess}, but does not rank, round-robin, or + * stop after the first usable account. Intended for diagnostics that must + * exercise each stored account exactly once. + */ + async getOAuthAccesses(provider: string, options?: AuthApiKeyOptions): Promise { + if (this.#runtimeOverrides.has(provider) || this.#configOverrides.has(provider)) { + return []; + } + const providerKey = this.#getProviderTypeKey(provider, "oauth"); + const selections = this.#getStoredCredentials(provider) + .map((entry, index) => ({ credentialId: entry.id, credential: entry.credential, index })) + .filter( + (entry): entry is { credentialId: number; credential: OAuthCredential; index: number } => + entry.credential.type === "oauth", + ); + return Promise.all( + selections.map(async (selection): Promise => { + try { + const resolved = await this.#tryOAuthCredential( + provider, + { credential: selection.credential, index: selection.index }, + providerKey, + undefined, + options, + { + checkUsage: false, + allowBlocked: true, + }, + ); + if (!resolved) { + return { + ok: false, + credentialId: selection.credentialId, + accountId: selection.credential.accountId, + email: selection.credential.email, + projectId: selection.credential.projectId, + enterpriseUrl: selection.credential.enterpriseUrl, + error: "OAuth access unavailable", + }; + } + const { credential } = resolved; + return { + ok: true, + credentialId: selection.credentialId, + accessToken: credential.access, + accountId: credential.accountId, + email: credential.email, + projectId: credential.projectId, + enterpriseUrl: credential.enterpriseUrl, + }; + } catch (error) { + return { + ok: false, + credentialId: selection.credentialId, + accountId: selection.credential.accountId, + email: selection.credential.email, + projectId: selection.credential.projectId, + enterpriseUrl: selection.credential.enterpriseUrl, + error: error instanceof Error ? error.message : String(error), + }; + } + }), + ); + } + #extractStructuredApiKeyToken(apiKey: string): string | undefined { if (!apiKey.startsWith("{")) return undefined; try { diff --git a/packages/ai/src/providers/azure-openai-responses.ts b/packages/ai/src/providers/azure-openai-responses.ts index f9d3a2bed..04027d02a 100644 --- a/packages/ai/src/providers/azure-openai-responses.ts +++ b/packages/ai/src/providers/azure-openai-responses.ts @@ -296,7 +296,7 @@ function buildParams( prompt_cache_key: normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId), }; - applyCommonResponsesSamplingParams(params, options, model.provider); + applyCommonResponsesSamplingParams(params, options, model); if (context.tools) { params.tools = convertTools(context.tools); diff --git a/packages/ai/src/providers/openai-completions.ts b/packages/ai/src/providers/openai-completions.ts index 93fbf86df..b8b5e591c 100644 --- a/packages/ai/src/providers/openai-completions.ts +++ b/packages/ai/src/providers/openai-completions.ts @@ -1180,7 +1180,7 @@ function buildParams( params.store = false; } - if (effectiveMaxTokens) { + if (effectiveMaxTokens && !model.omitMaxOutputTokens) { if (compat.maxTokensField === "max_tokens") { params.max_tokens = effectiveMaxTokens; } else { diff --git a/packages/ai/src/providers/openai-responses-shared.ts b/packages/ai/src/providers/openai-responses-shared.ts index 2125ce726..135f92589 100644 --- a/packages/ai/src/providers/openai-responses-shared.ts +++ b/packages/ai/src/providers/openai-responses-shared.ts @@ -395,19 +395,54 @@ export async function processResponsesStream( model: Model, options?: ProcessResponsesStreamOptions, ): Promise { - let currentItem: - | ResponseReasoningItem - | ResponseOutputMessage - | ResponseFunctionToolCall - | ResponseCustomToolCall - | null = null; - let currentBlock: - | ThinkingContent - | TextContent - | (ToolCall & { partialJson: string; lastParseLen?: number }) - | null = null; - const blocks = output.content; - const blockIndex = () => blocks.length - 1; + type StreamingToolCallBlock = ToolCall & { partialJson: string; lastParseLen?: number }; + interface StreamingItem { + item: ResponseReasoningItem | ResponseOutputMessage | ResponseFunctionToolCall | ResponseCustomToolCall; + block: ThinkingContent | TextContent | StreamingToolCallBlock; + } + + // Multiple items (parallel function_calls in particular) can be open at the same + // time. OpenAI's spec routes every per-item event by `output_index`/`item_id`; + // see https://github.com/can1357/oh-my-pi/issues/1880 — llama.cpp emits parallel + // function_call deltas interleaved, and a singleton `current` reference would + // fold them into the wrong block and drop arguments on every call but the last. + const openItemsByOutputIndex = new Map(); + const openItemsByItemId = new Map(); + let lastOpenItem: StreamingItem | null = null; + + const registerOpenItem = ( + outputIndex: number | undefined, + itemId: string | undefined, + entry: StreamingItem, + ): void => { + if (typeof outputIndex === "number") openItemsByOutputIndex.set(outputIndex, entry); + if (itemId) openItemsByItemId.set(itemId, entry); + lastOpenItem = entry; + }; + const lookupOpenItem = (event: { output_index?: number; item_id?: string }): StreamingItem | undefined => { + if (typeof event.output_index === "number") { + const found = openItemsByOutputIndex.get(event.output_index); + if (found) return found; + } + if (event.item_id) { + const found = openItemsByItemId.get(event.item_id); + if (found) return found; + } + // Fallback for tests / mock providers that omit identifiers on stream events. + return lastOpenItem ?? undefined; + }; + const closeOpenItem = ( + outputIndex: number | undefined, + itemId: string | undefined, + entry: StreamingItem | undefined, + ): void => { + if (typeof outputIndex === "number") openItemsByOutputIndex.delete(outputIndex); + if (itemId) openItemsByItemId.delete(itemId); + if (entry && lastOpenItem === entry) lastOpenItem = null; + }; + const contentIndexOf = (block: ThinkingContent | TextContent | StreamingToolCallBlock): number => + output.content.indexOf(block); + let sawFirstToken = false; for await (const event of openaiStream) { @@ -420,29 +455,28 @@ export async function processResponsesStream( } const item = event.item; if (item.type === "reasoning") { - currentItem = item; - currentBlock = { type: "thinking", thinking: "", itemId: item.id }; - output.content.push(currentBlock); - stream.push({ type: "thinking_start", contentIndex: blockIndex(), partial: output }); + const block: ThinkingContent = { type: "thinking", thinking: "", itemId: item.id }; + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "thinking_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "message") { - currentItem = item; - currentBlock = { type: "text", text: "" }; - output.content.push(currentBlock); - stream.push({ type: "text_start", contentIndex: blockIndex(), partial: output }); + const block: TextContent = { type: "text", text: "" }; + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "text_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "function_call") { - currentItem = item; - currentBlock = { + const block: StreamingToolCallBlock = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), name: item.name, arguments: {}, partialJson: item.arguments || "", }; - output.content.push(currentBlock); - stream.push({ type: "toolcall_start", contentIndex: blockIndex(), partial: output }); + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } else if (item.type === "custom_tool_call") { - currentItem = item; - currentBlock = { + const block: StreamingToolCallBlock = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), // Preserve the raw wire name (e.g. `apply_patch`). The agent-loop @@ -456,39 +490,43 @@ export async function processResponsesStream( // accumulation buffer so later code that inspects the field still works. partialJson: item.input ?? "", }; - output.content.push(currentBlock); - stream.push({ type: "toolcall_start", contentIndex: blockIndex(), partial: output }); + output.content.push(block); + registerOpenItem(event.output_index, item.id, { item, block }); + stream.push({ type: "toolcall_start", contentIndex: contentIndexOf(block), partial: output }); } } else if (event.type === "response.reasoning_summary_part.added") { - if (currentItem?.type === "reasoning") { - currentItem.summary = currentItem.summary || []; - currentItem.summary.push(event.part); + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning") { + entry.item.summary = entry.item.summary || []; + entry.item.summary.push(event.part); } } else if (event.type === "response.reasoning_summary_text.delta") { - if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") { - currentItem.summary = currentItem.summary || []; - const lastPart = currentItem.summary[currentItem.summary.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning" && entry.block.type === "thinking") { + entry.item.summary = entry.item.summary || []; + const lastPart = entry.item.summary[entry.item.summary.length - 1]; if (lastPart) { - currentBlock.thinking += event.delta; + entry.block.thinking += event.delta; lastPart.text += event.delta; stream.push({ type: "thinking_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } } else if (event.type === "response.reasoning_summary_part.done") { - if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") { - currentItem.summary = currentItem.summary || []; - const lastPart = currentItem.summary[currentItem.summary.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning" && entry.block.type === "thinking") { + entry.item.summary = entry.item.summary || []; + const lastPart = entry.item.summary[entry.item.summary.length - 1]; if (lastPart) { - currentBlock.thinking += "\n\n"; + entry.block.thinking += "\n\n"; lastPart.text += "\n\n"; stream.push({ type: "thinking_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: "\n\n", partial: output, }); @@ -497,91 +535,103 @@ export async function processResponsesStream( } else if (event.type === "response.reasoning_text.delta") { // Raw reasoning text delta from local providers that stream thinking // directly rather than via the OpenAI summary tracking protocol. - if (currentItem?.type === "reasoning" && currentBlock?.type === "thinking") { - currentBlock.thinking += event.delta; + const entry = lookupOpenItem(event); + if (entry?.item.type === "reasoning" && entry.block.type === "thinking") { + entry.block.thinking += event.delta; stream.push({ type: "thinking_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } else if (event.type === "response.content_part.added") { - if (currentItem?.type === "message") { - currentItem.content = currentItem.content || []; + const entry = lookupOpenItem(event); + if (entry?.item.type === "message") { + entry.item.content = entry.item.content || []; if (event.part.type === "output_text" || event.part.type === "refusal") { - currentItem.content.push(event.part); + entry.item.content.push(event.part); } } } else if (event.type === "response.output_text.delta") { - if (currentItem?.type === "message" && currentBlock?.type === "text") { - const lastPart = currentItem.content?.[currentItem.content.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "message" && entry.block.type === "text") { + const lastPart = entry.item.content?.[entry.item.content.length - 1]; if (lastPart?.type === "output_text") { - currentBlock.text += event.delta; + entry.block.text += event.delta; lastPart.text += event.delta; stream.push({ type: "text_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } } else if (event.type === "response.refusal.delta") { - if (currentItem?.type === "message" && currentBlock?.type === "text") { - const lastPart = currentItem.content?.[currentItem.content.length - 1]; + const entry = lookupOpenItem(event); + if (entry?.item.type === "message" && entry.block.type === "text") { + const lastPart = entry.item.content?.[entry.item.content.length - 1]; if (lastPart?.type === "refusal") { - currentBlock.text += event.delta; + entry.block.text += event.delta; lastPart.refusal += event.delta; stream.push({ type: "text_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(entry.block), delta: event.delta, partial: output, }); } } } else if (event.type === "response.function_call_arguments.delta") { - if (currentItem?.type === "function_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson += event.delta; - const throttled = parseStreamingJsonThrottled(currentBlock.partialJson, currentBlock.lastParseLen ?? 0); + const entry = lookupOpenItem(event); + if (entry?.item.type === "function_call" && entry.block.type === "toolCall") { + const block = entry.block; + block.partialJson += event.delta; + const throttled = parseStreamingJsonThrottled(block.partialJson, block.lastParseLen ?? 0); if (throttled) { - currentBlock.arguments = throttled.value; - currentBlock.lastParseLen = throttled.parsedLen; + block.arguments = throttled.value; + block.lastParseLen = throttled.parsedLen; } stream.push({ type: "toolcall_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(block), delta: event.delta, partial: output, }); } } else if (event.type === "response.function_call_arguments.done") { - if (currentItem?.type === "function_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson = event.arguments; - currentBlock.arguments = parseStreamingJson(currentBlock.partialJson); - delete (currentBlock as { partialJson?: string }).partialJson; - delete (currentBlock as { lastParseLen?: number }).lastParseLen; + const entry = lookupOpenItem(event); + if (entry?.item.type === "function_call" && entry.block.type === "toolCall") { + const block = entry.block; + block.partialJson = event.arguments; + block.arguments = parseStreamingJson(block.partialJson); + delete (block as { partialJson?: string }).partialJson; + delete (block as { lastParseLen?: number }).lastParseLen; } } else if (event.type === "response.custom_tool_call_input.delta") { - if (currentItem?.type === "custom_tool_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson += event.delta; - currentBlock.arguments = { input: currentBlock.partialJson }; + const entry = lookupOpenItem(event); + if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") { + const block = entry.block; + block.partialJson += event.delta; + block.arguments = { input: block.partialJson }; stream.push({ type: "toolcall_delta", - contentIndex: blockIndex(), + contentIndex: contentIndexOf(block), delta: event.delta, partial: output, }); } } else if (event.type === "response.custom_tool_call_input.done") { - if (currentItem?.type === "custom_tool_call" && currentBlock?.type === "toolCall") { - currentBlock.partialJson = event.input; - currentBlock.arguments = { input: event.input }; + const entry = lookupOpenItem(event); + if (entry?.item.type === "custom_tool_call" && entry.block.type === "toolCall") { + entry.block.partialJson = event.input; + entry.block.arguments = { input: event.input }; } } else if (event.type === "response.output_item.done") { const item = structuredCloneJSON(event.item); options?.onOutputItemDone?.(item); + const entry = lookupOpenItem({ output_index: event.output_index, item_id: item.id }); if (item.type === "reasoning") { const thinking = item.summary?.length > 0 @@ -595,54 +645,53 @@ export async function processResponsesStream( if (reasoningBlock) { reasoningBlock.thinking = thinking; reasoningBlock.thinkingSignature = JSON.stringify(item); - const reasoningBlockIndex = output.content.indexOf(reasoningBlock); stream.push({ type: "thinking_end", - contentIndex: reasoningBlockIndex, + contentIndex: contentIndexOf(reasoningBlock), content: thinking, partial: output, }); } - if ((currentBlock as ThinkingContent | null)?.itemId === item.id) currentBlock = null; - } else if (item.type === "message" && currentBlock?.type === "text") { - currentBlock.text = item.content + closeOpenItem(event.output_index, item.id, entry); + } else if (item.type === "message" && entry?.block.type === "text") { + const block = entry.block; + block.text = item.content .map(part => (part.type === "output_text" ? (part.text ?? "") : (part.refusal ?? ""))) .join(""); - currentBlock.textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined); + block.textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined); stream.push({ type: "text_end", - contentIndex: blockIndex(), - content: currentBlock.text, + contentIndex: contentIndexOf(block), + content: block.text, partial: output, }); - currentBlock = null; + closeOpenItem(event.output_index, item.id, entry); } else if (item.type === "function_call") { - const args = - currentBlock?.type === "toolCall" && currentBlock.partialJson - ? parseStreamingJson(currentBlock.partialJson) - : parseStreamingJson(item.arguments || "{}"); + const block = entry?.block.type === "toolCall" ? entry.block : undefined; + const args = block?.partialJson + ? parseStreamingJson(block.partialJson) + : parseStreamingJson(item.arguments || "{}"); const toolCall: ToolCall = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), name: item.name, arguments: args, }; - if (currentBlock?.type === "toolCall") { + if (block) { // Persist the authoritative final args on the stored block. The // throttled delta parser may have skipped the last partial parse, - // leaving currentBlock.arguments stale (often `{}`); the emitted - // toolCall and the persisted block must agree. - currentBlock.arguments = args; - delete (currentBlock as { partialJson?: string }).partialJson; - delete (currentBlock as { lastParseLen?: number }).lastParseLen; + // leaving block.arguments stale (often `{}`); the emitted toolCall + // and the persisted block must agree. + block.arguments = args; + delete (block as { partialJson?: string }).partialJson; + delete (block as { lastParseLen?: number }).lastParseLen; } - currentBlock = null; - stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output }); + const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; + closeOpenItem(event.output_index, item.id, entry); + stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } else if (item.type === "custom_tool_call") { - const rawInput = - currentBlock?.type === "toolCall" && currentBlock.partialJson - ? currentBlock.partialJson - : (item.input ?? ""); + const block = entry?.block.type === "toolCall" ? entry.block : undefined; + const rawInput = block?.partialJson ? block.partialJson : (item.input ?? ""); const toolCall: ToolCall = { type: "toolCall", id: encodeResponsesToolCallId(item.call_id, item.id), @@ -650,8 +699,9 @@ export async function processResponsesStream( arguments: { input: rawInput }, customWireName: item.name, }; - currentBlock = null; - stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output }); + const contentIndex = block ? contentIndexOf(block) : output.content.length - 1; + closeOpenItem(event.output_index, item.id, entry); + stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } } else if (event.type === "response.completed") { const response = event.response; @@ -752,21 +802,26 @@ type CommonSamplingOptions = Pick< /** * Apply the common `StreamOptions` → Responses sampling-parameter mapping (max output tokens, * temperature, top-p/k, min-p, presence/repetition penalties, service tier). Mutates `params`. + * + * `max_output_tokens` is suppressed when {@link Model.omitMaxOutputTokens} is `true`, so + * proxies (notably Ollama) that forward to upstream APIs with an unknown output-token cap + * can let the upstream apply its own default instead of 400-ing on `maxTokens` values that + * reflect the model's context window rather than the upstream output limit. */ export function applyCommonResponsesSamplingParams

( params: P, options: CommonSamplingOptions | undefined, - provider: string, + model: Pick, ): void { - if (options?.maxTokens) params.max_output_tokens = options.maxTokens; + if (options?.maxTokens && !model.omitMaxOutputTokens) params.max_output_tokens = options.maxTokens; if (options?.temperature !== undefined) params.temperature = options.temperature; if (options?.topP !== undefined) params.top_p = options.topP; if (options?.topK !== undefined) params.top_k = options.topK; if (options?.minP !== undefined) params.min_p = options.minP; if (options?.presencePenalty !== undefined) params.presence_penalty = options.presencePenalty; if (options?.repetitionPenalty !== undefined) params.repetition_penalty = options.repetitionPenalty; - if (shouldSendServiceTier(options?.serviceTier, provider)) { - const resolved = resolveServiceTier(options?.serviceTier, provider); + if (shouldSendServiceTier(options?.serviceTier, model.provider)) { + const resolved = resolveServiceTier(options?.serviceTier, model.provider); if (resolved === "flex" || resolved === "scale" || resolved === "priority") { params.service_tier = resolved; } diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index f9647128d..ac1684b43 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -463,7 +463,7 @@ function buildParams( stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined, }; - applyCommonResponsesSamplingParams(params, options, model.provider); + applyCommonResponsesSamplingParams(params, options, model); // TODO: openai responses has no top-level `stop`/`stop_sequences`; surface via reasoning.stop? // `StreamOptions.stopSequences` is intentionally dropped for this provider. // TODO: openai responses has no top-level `frequency_penalty` field as of the current SDK; diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index 95bbd72e6..9b03d99d8 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -905,6 +905,18 @@ export interface Model { premiumMultiplier?: number; contextWindow: number; maxTokens: number; + /** + * When `true`, providers MUST omit `max_output_tokens` (Responses) / + * `max_tokens` / `max_completion_tokens` (Completions) from the outbound + * request and let the upstream API decide the per-response cap. `maxTokens` + * is still used locally for budgeting (compaction, context promotion); only + * the wire field is suppressed. + * + * Use this for proxies (notably Ollama) that forward to a backend whose true + * output limit OMP cannot discover — sending the wrong value triggers 400s + * from the upstream provider. + */ + omitMaxOutputTokens?: boolean; headers?: Record; /** * Streaming transport override. When `"pi-native"`, `streamSimple` routes diff --git a/packages/ai/test/auth-broker-remote-store.test.ts b/packages/ai/test/auth-broker-remote-store.test.ts index 206d0fc71..50be18938 100644 --- a/packages/ai/test/auth-broker-remote-store.test.ts +++ b/packages/ai/test/auth-broker-remote-store.test.ts @@ -8,6 +8,7 @@ import { AuthStorage, REMOTE_REFRESH_SENTINEL, RemoteAuthCredentialStore, + type SnapshotResponse, SqliteAuthCredentialStore, startAuthBroker, } from "../src"; @@ -109,4 +110,26 @@ describe("RemoteAuthCredentialStore SSE integration", () => { await waitUntil(() => remote!.snapshot.credentials.length === 1); expect(remote!.snapshot.credentials[0].id).not.toBe(bId); }); + + test("calls onSnapshot for broker snapshots but not the constructor snapshot", async () => { + const client = new AuthBrokerClient({ url: handle!.url, token }); + const initialResult = await client.fetchSnapshot(); + if (initialResult.status !== 200) throw new Error("expected initial snapshot"); + const callbacks: Array<{ snapshot: SnapshotResponse; generation: number }> = []; + remote = new RemoteAuthCredentialStore({ + client, + initialSnapshot: initialResult.snapshot, + streamSnapshots: false, + onSnapshot: (snapshot, generation) => { + callbacks.push({ snapshot, generation }); + }, + }); + expect(callbacks).toHaveLength(0); + + const refreshed = await remote.refreshSnapshot(); + + expect(callbacks).toHaveLength(1); + expect(callbacks[0].generation).toBe(refreshed.generation); + expect(callbacks[0].snapshot).toEqual(refreshed); + }); }); diff --git a/packages/ai/test/auth-broker-snapshot-cache.test.ts b/packages/ai/test/auth-broker-snapshot-cache.test.ts new file mode 100644 index 000000000..9480472c6 --- /dev/null +++ b/packages/ai/test/auth-broker-snapshot-cache.test.ts @@ -0,0 +1,180 @@ +import { describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { readAuthBrokerSnapshotCache, type SnapshotResponse, writeAuthBrokerSnapshotCache } from "../src"; + +const TOKEN = "broker-cache-token"; +const URL = "http://127.0.0.1:8765"; + +function makeSnapshot(generatedAt: number): SnapshotResponse { + return { + generation: 7, + generatedAt, + serverNowMs: generatedAt, + refresher: { + enabled: true, + intervalMs: 60_000, + skewMs: 300_000, + nextSweepInMs: 10_000, + }, + credentials: [ + { + id: 1, + provider: "anthropic", + credential: { type: "api_key", key: "secret-api-key" }, + identityKey: null, + rotatesInMs: null, + }, + ], + }; +} + +async function withCachePath(run: (cachePath: string) => Promise): Promise { + const tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "auth-broker-snapshot-cache-")); + try { + await run(path.join(tempDir, "snapshot.enc")); + } finally { + await fs.rm(tempDir, { recursive: true, force: true }); + } +} + +describe("auth-broker snapshot cache", () => { + test("round-trips an encrypted snapshot and writes mode 0600", async () => { + await withCachePath(async cachePath => { + const snapshot = makeSnapshot(1_000_000); + await writeAuthBrokerSnapshotCache({ path: cachePath, token: TOKEN, url: URL, snapshot }); + + const stat = await fs.stat(cachePath); + expect(stat.mode & 0o777).toBe(0o600); + const payload = await fs.readFile(cachePath); + expect(new TextDecoder().decode(payload)).not.toContain("secret-api-key"); + + const decoded = await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }); + expect(decoded).toEqual(snapshot); + }); + }); + + test("returns null when token, url binding, or ciphertext integrity do not match", async () => { + await withCachePath(async cachePath => { + const snapshot = makeSnapshot(1_000_000); + await writeAuthBrokerSnapshotCache({ path: cachePath, token: TOKEN, url: URL, snapshot }); + + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: "wrong-token", + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: "http://127.0.0.1:9999", + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + const tampered = await fs.readFile(cachePath); + tampered[tampered.byteLength - 1] ^= 0xff; + await fs.writeFile(cachePath, tampered); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + }); + }); + + test("enforces generatedAt-based TTL", async () => { + await withCachePath(async cachePath => { + const snapshot = makeSnapshot(10_000); + await writeAuthBrokerSnapshotCache({ path: cachePath, token: TOKEN, url: URL, snapshot }); + + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 100, + now: () => 10_100, + }), + ).toEqual(snapshot); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 100, + now: () => 10_101, + }), + ).toBeNull(); + }); + }); + + test("returns null for missing, short, unencrypted, and schema-invalid files", async () => { + await withCachePath(async cachePath => { + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + await fs.writeFile(cachePath, new Uint8Array([0x4f, 0x4d])); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + await fs.writeFile(cachePath, JSON.stringify(makeSnapshot(1_000_000))); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + + await writeAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + snapshot: { generation: 1 } as unknown as SnapshotResponse, + }); + expect( + await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: URL, + ttlMs: 60_000, + now: () => 1_001_000, + }), + ).toBeNull(); + }); + }); +}); diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 6402e39c6..7e23abae9 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -103,6 +103,33 @@ function createCredential(accountId: string, email: string): OAuthCredentials { }; } +async function countApiKeySelections( + authStorage: AuthStorage, + provider: string, + sessionPrefix: string, + samples = 150, +): Promise> { + const counts = new Map(); + for (let index = 0; index < samples; index += 1) { + const apiKey = await authStorage.getApiKey(provider, `${sessionPrefix}-${index}`); + if (!apiKey) continue; + counts.set(apiKey, (counts.get(apiKey) ?? 0) + 1); + } + return counts; +} + +function countFor(counts: Map, apiKey: string): number { + return counts.get(apiKey) ?? 0; +} + +function expectWeightedPreference(counts: Map, preferred: string, fallback: string): void { + const preferredCount = countFor(counts, preferred); + const fallbackCount = countFor(counts, fallback); + expect(preferredCount).toBeGreaterThan(fallbackCount); + expect(preferredCount / fallbackCount).toBeGreaterThan(1.4); + expect(preferredCount / fallbackCount).toBeLessThan(2.4); +} + describe("AuthStorage codex oauth ranking", () => { let tempDir = ""; let store: AuthCredentialStore | null = null; @@ -146,7 +173,7 @@ describe("AuthStorage codex oauth ranking", () => { } }); - test("prefers near-reset weekly account over lower-used far-reset account", async () => { + test("weights near-reset weekly account over lower-used far-reset account", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("openai-codex", [ @@ -171,11 +198,11 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-weekly-reset"); - expect(apiKey).toBe("api-acct-near"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-near"); + expectWeightedPreference(counts, "api-acct-near", "api-acct-far"); }); - test("prioritizes fresh 5h ticker account at 0% usage", async () => { + test("weights fresh 5h ticker account at 0% usage", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("openai-codex", [ @@ -208,8 +235,8 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-five-hour-start"); - expect(apiKey).toBe("api-acct-zero"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-zero"); + expectWeightedPreference(counts, "api-acct-zero", "api-acct-progress"); }); test("skips exhausted weekly account even when reset is near", async () => { if (!authStorage) throw new Error("test setup failed"); @@ -399,7 +426,7 @@ describe("AuthStorage codex oauth ranking", () => { expect(elapsedMs).toBeLessThan(1_000); }); - test("sorts 3 accounts by weekly drain rate", async () => { + test("weights 3 accounts by weekly drain rate", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("openai-codex", [ @@ -433,8 +460,9 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-three-accounts"); - expect(apiKey).toBe("api-acct-slow"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-three"); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-medium")); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-fast")); }); test("handles usage fetch failure gracefully (null report)", async () => { @@ -455,8 +483,8 @@ describe("AuthStorage codex oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("openai-codex", "session-null-usage"); - expect(apiKey).toBe("api-acct-known"); + const counts = await countApiKeySelections(authStorage, "openai-codex", "weighted-codex-known", 300); + expectWeightedPreference(counts, "api-acct-known", "api-acct-null"); }); test("refreshes expired oauth candidates in parallel before selection", async () => { if (!authStorage) throw new Error("test setup failed"); @@ -623,7 +651,7 @@ describe("AuthStorage claude oauth ranking", () => { } }); - test("prefers lower secondary drain rate account", async () => { + test("weights lower secondary drain rate account", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("anthropic", [ @@ -648,8 +676,67 @@ describe("AuthStorage claude oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("anthropic", "session-claude-drain"); - expect(apiKey).toBe("api-acct-near"); + const counts = await countApiKeySelections(authStorage, "anthropic", "weighted-claude-near"); + expectWeightedPreference(counts, "api-acct-near", "api-acct-far"); + }); + + test("balances equal-priority accounts evenly", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("anthropic", [ + { type: "oauth", ...createCredential("acct-a", "a@example.com") }, + { type: "oauth", ...createCredential("acct-b", "b@example.com") }, + ]); + + for (const accountId of ["acct-a", "acct-b"]) { + usageByAccount.set( + accountId, + createClaudeUsageReport({ + accountId, + primary: { usedFraction: 0.25, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.25, resetInMs: 4 * 24 * HOUR_MS }, + }), + ); + } + + const counts = await countApiKeySelections(authStorage, "anthropic", "weighted-claude-equal", 200); + expect(Math.abs(countFor(counts, "api-acct-a") - countFor(counts, "api-acct-b"))).toBeLessThanOrEqual(25); + }); + + test("caps the strongest priority bucket at about 2x baseline weight", async () => { + if (!authStorage) throw new Error("test setup failed"); + + await authStorage.set("anthropic", [ + { type: "oauth", ...createCredential("acct-best", "best@example.com") }, + { type: "oauth", ...createCredential("acct-base-a", "base-a@example.com") }, + { type: "oauth", ...createCredential("acct-base-b", "base-b@example.com") }, + ]); + + usageByAccount.set( + "acct-best", + createClaudeUsageReport({ + accountId: "acct-best", + primary: { usedFraction: 0.05, resetInMs: 4 * HOUR_MS }, + secondary: { usedFraction: 0.05, resetInMs: 6 * 24 * HOUR_MS }, + }), + ); + for (const accountId of ["acct-base-a", "acct-base-b"]) { + usageByAccount.set( + accountId, + createClaudeUsageReport({ + accountId, + primary: { usedFraction: 0.7, resetInMs: 2 * HOUR_MS }, + secondary: { usedFraction: 0.7, resetInMs: 2 * 24 * HOUR_MS }, + }), + ); + } + + const counts = await countApiKeySelections(authStorage, "anthropic", "claude-cap", 300); + expectWeightedPreference(counts, "api-acct-best", "api-acct-base-a"); + expectWeightedPreference(counts, "api-acct-best", "api-acct-base-b"); + expect(Math.abs(countFor(counts, "api-acct-base-a") - countFor(counts, "api-acct-base-b"))).toBeLessThanOrEqual( + 15, + ); }); test("skips exhausted account and picks healthy", async () => { @@ -710,7 +797,7 @@ describe("AuthStorage claude oauth ranking", () => { expect(apiKey).toBe("api-acct-soon"); }); - test("sorts 3 accounts by secondary drain rate", async () => { + test("weights 3 accounts by secondary drain rate", async () => { if (!authStorage) throw new Error("test setup failed"); await authStorage.set("anthropic", [ @@ -744,8 +831,9 @@ describe("AuthStorage claude oauth ranking", () => { }), ); - const apiKey = await authStorage.getApiKey("anthropic", "session-claude-three"); - expect(apiKey).toBe("api-acct-slow"); + const counts = await countApiKeySelections(authStorage, "anthropic", "weighted-claude-three"); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-medium")); + expect(countFor(counts, "api-acct-slow")).toBeGreaterThan(countFor(counts, "api-acct-fast")); }); test("single credential works without ranking", async () => { diff --git a/packages/ai/test/openai-responses-omit-max-output-tokens.test.ts b/packages/ai/test/openai-responses-omit-max-output-tokens.test.ts new file mode 100644 index 000000000..a5dcb1537 --- /dev/null +++ b/packages/ai/test/openai-responses-omit-max-output-tokens.test.ts @@ -0,0 +1,73 @@ +import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { getBundledModel } from "../src/models"; +import { streamSimple } from "../src/stream"; +import type { Context, Model } from "../src/types"; + +const originalFetch = global.fetch; + +const baseModel = getBundledModel("openai", "gpt-4o-mini") as Model<"openai-responses">; + +function mockSseFetch(): Record { + const captured: Record = {}; + const fetchMock = vi.fn(async (_url: string | URL | Request, init?: RequestInit) => { + const body = typeof init?.body === "string" ? (JSON.parse(init.body) as Record) : {}; + Object.assign(captured, body); + const event = { + type: "response.completed", + response: { + status: "completed", + usage: { + input_tokens: 1, + output_tokens: 1, + total_tokens: 2, + input_tokens_details: { cached_tokens: 0 }, + }, + }, + }; + return new Response(`data: ${JSON.stringify(event)}\n\n`, { + status: 200, + headers: { "content-type": "text/event-stream" }, + }); + }); + global.fetch = Object.assign(fetchMock, { preconnect: originalFetch.preconnect }) as typeof fetch; + return captured; +} + +const ctx: Context = { + systemPrompt: ["hi"], + messages: [{ role: "user", content: "ping", timestamp: Date.now() }], +}; + +async function drain(model: Model<"openai-responses">): Promise> { + const captured = mockSseFetch(); + const stream = streamSimple(model, ctx, { apiKey: "k" }); + for await (const event of stream) { + if (event.type === "done" || event.type === "error") break; + } + return captured; +} + +beforeEach(() => { + expect(baseModel.maxTokens).toBeGreaterThan(0); +}); + +afterEach(() => { + global.fetch = originalFetch; + vi.restoreAllMocks(); +}); + +describe("openai-responses max_output_tokens opt-out", () => { + it("sends max_output_tokens = model.maxTokens by default", async () => { + const body = await drain(baseModel); + expect(body.max_output_tokens).toBe(baseModel.maxTokens); + }); + + it("omits max_output_tokens when model.omitMaxOutputTokens is true", async () => { + const model: Model<"openai-responses"> = { ...baseModel, omitMaxOutputTokens: true }; + const body = await drain(model); + expect(body).not.toHaveProperty("max_output_tokens"); + // maxTokens is still populated locally for budgeting, even though we + // don't put it on the wire. + expect(model.maxTokens).toBe(baseModel.maxTokens); + }); +}); diff --git a/packages/ai/test/openai-responses-parallel-tool-calls.test.ts b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts new file mode 100644 index 000000000..56e076e6d --- /dev/null +++ b/packages/ai/test/openai-responses-parallel-tool-calls.test.ts @@ -0,0 +1,205 @@ +// Regression for https://github.com/can1357/oh-my-pi/issues/1880. +// +// llama.cpp (and any OpenAI-Responses-compatible host that interleaves +// multiple function_call items) emits `output_item.added` for every parallel +// call before the deltas arrive, then routes deltas via `item_id`/`output_index` +// instead of relying on a single in-flight item. `processResponsesStream` +// previously kept a singleton `currentBlock` reference and ignored those +// identifiers, so deltas for the first call were folded into the buffer of the +// most-recently-added block. The dispatcher then received empty `{}` arguments +// for every call except the last one. +// +// These tests pin the contract: each `function_call_arguments.{delta,done}` and +// `output_item.done` event must be routed by `output_index`/`item_id`, not by +// arrival order. +import { describe, expect, test } from "bun:test"; +import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared"; +import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types"; +import type { ResponseStreamEvent } from "openai/resources/responses/responses"; + +function makeModel(): Model<"openai-responses"> { + return { + api: "openai-responses", + name: "Llama", + id: "llama-3", + provider: "llama.cpp", + baseUrl: "http://127.0.0.1:8080/v1", + contextWindow: 8192, + maxTokens: 2048, + input: ["text"], + reasoning: false, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + }; +} + +function makeOutput(): AssistantMessage { + return { + role: "assistant", + content: [], + timestamp: Date.now(), + provider: "llama.cpp", + model: "llama-3", + api: "openai-responses", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "stop", + }; +} + +async function* makeStream(events: unknown[]): AsyncIterable { + for (const e of events) yield e as ResponseStreamEvent; +} + +type EmittedEvent = { type?: string } & Record; + +describe("processResponsesStream: parallel function_call items", () => { + test("routes deltas to the correct block when both items are added before any delta", async () => { + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never; + + const argsA = JSON.stringify({ _i: "Reading test", path: "test.txt" }); + const argsB = JSON.stringify({ _i: "Reading test", path: "test.md" }); + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: "" }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: "" }, + }, + { + type: "response.function_call_arguments.delta", + output_index: 0, + item_id: "fc_a", + delta: argsA, + }, + { + type: "response.function_call_arguments.delta", + output_index: 1, + item_id: "fc_b", + delta: argsB, + }, + { + type: "response.function_call_arguments.done", + output_index: 0, + item_id: "fc_a", + arguments: argsA, + }, + { + type: "response.function_call_arguments.done", + output_index: 1, + item_id: "fc_b", + arguments: argsB, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.done", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: argsB }, + }, + ]), + output, + stream, + makeModel(), + ); + + expect(output.content).toHaveLength(2); + const [blockA, blockB] = output.content; + expect(blockA?.type).toBe("toolCall"); + expect(blockB?.type).toBe("toolCall"); + if (blockA?.type !== "toolCall" || blockB?.type !== "toolCall") throw new Error("expected toolCalls"); + expect(blockA.arguments).toEqual({ _i: "Reading test", path: "test.txt" }); + expect(blockB.arguments).toEqual({ _i: "Reading test", path: "test.md" }); + + const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{ + toolCall: { id: string; arguments: Record }; + contentIndex: number; + }>; + expect(ends).toHaveLength(2); + const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e])); + expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ _i: "Reading test", path: "test.txt" }); + expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ _i: "Reading test", path: "test.md" }); + expect(byCallId.get("call_a")?.contentIndex).toBe(0); + expect(byCallId.get("call_b")?.contentIndex).toBe(1); + + // Delta events must also carry the per-block contentIndex — otherwise the + // streaming UI updates the wrong block while args are still arriving. + const deltas = emitted.filter(e => e.type === "toolcall_delta") as Array<{ + delta: string; + contentIndex: number; + }>; + expect(deltas).toHaveLength(2); + const deltaForA = deltas.find(d => d.delta === argsA); + const deltaForB = deltas.find(d => d.delta === argsB); + expect(deltaForA?.contentIndex).toBe(0); + expect(deltaForB?.contentIndex).toBe(1); + }); + + test("routes done-only finalization to the correct block when arguments stream as a single chunk on each item", async () => { + // Some local Responses-compat hosts skip the per-delta protocol entirely + // and stash the full arguments string on `output_item.added`/`done`. + const output = makeOutput(); + const emitted: EmittedEvent[] = []; + const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never; + + const argsA = JSON.stringify({ path: "test.txt" }); + const argsB = JSON.stringify({ path: "test.md" }); + + await processResponsesStream( + makeStream([ + { + type: "response.output_item.added", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.added", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: argsB }, + }, + { + type: "response.output_item.done", + output_index: 0, + item: { type: "function_call", id: "fc_a", call_id: "call_a", name: "read", arguments: argsA }, + }, + { + type: "response.output_item.done", + output_index: 1, + item: { type: "function_call", id: "fc_b", call_id: "call_b", name: "read", arguments: argsB }, + }, + ]), + output, + stream, + makeModel(), + ); + + const [blockA, blockB] = output.content; + if (blockA?.type !== "toolCall" || blockB?.type !== "toolCall") throw new Error("expected toolCalls"); + expect(blockA.arguments).toEqual({ path: "test.txt" }); + expect(blockB.arguments).toEqual({ path: "test.md" }); + + const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{ + toolCall: { id: string; arguments: Record }; + }>; + expect(ends).toHaveLength(2); + const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e])); + expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ path: "test.txt" }); + expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ path: "test.md" }); + }); +}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 39fb7333e..ffea14606 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -11,6 +11,30 @@ - Fixed profile bootstrap parsing so stripped `--profile`/`--alias` values no longer make optional or extension flags consume following prompt text, preserved standalone `--` as end-of-options after extension string flags, and made profile aliases respect `ZDOTDIR` while rejecting shell reserved words. - Made native user-level config discovery follow the active profile. Skills, rules, slash commands, prompts, instructions, hooks, tools, settings, extensions, MCP servers, and the top-level `SYSTEM.md`/`RULES.md`/`AGENTS.md` now resolve the user scope through `getAgentDir()`, so a named profile sees only its own `~/.omp/profiles//agent` config instead of the default profile's `~/.omp/agent` leaking into every profile. This matches the `/mcp` config writer and `getMCPConfigPath("user")`. - Fixed symlinked extension directories being skipped by native auto-discovery. The glob walker runs with `follow_links=false`, so a symlinked directory under `extensions/` was yielded as a symlink but never descended into — its `index.{ts,js}`/`package.json` stayed invisible while real directories loaded normally. `discoverExtensionModulePaths` now detects top-level symlinked directories and resolves their entry points, so an extension shared across profiles via a symlink loads like a real directory (symlinked extension *files* were already handled). + +## [15.9.2] - 2026-06-05 + +### Added + +- Added an encrypted local auth-broker snapshot cache for `discoverAuthStorage`, with `OMP_AUTH_BROKER_SNAPSHOT_TTL_MS` and `OMP_AUTH_BROKER_SNAPSHOT_CACHE`, so fresh cached broker credentials can boot without a blocking `/v1/snapshot` fetch and survive broker-down startup windows. +- Added `dry-balance` CLI command to perform a dry-run OAuth account balancing check across configurable random session IDs, with sample and concurrency options, JSON output, and success/failure summary reporting +- Added `--json` output mode and machine-readable result format to `omp dry-balance` for automated use +- Added `omitMaxOutputTokens` to `models.yml` model definitions and `modelOverrides`, so users can opt a model out of the on-the-wire `max_output_tokens` / `max_tokens` cap while keeping the catalog `maxTokens` for local budgeting. Intended for Ollama-style proxies whose upstream output limit OMP cannot discover. ([#1881](https://github.com/can1357/oh-my-pi/issues/1881)) + +### Fixed + +- Fixed TTSR rule-violation injections leaking the absolute home directory to the model: the `ttsr-interrupt` / `ttsr-tool-reminder` blocks rendered the matched rule's `path` as its absolute on-disk path (e.g. `/Users/me/Projects/app/.omp/rules/no-any.md`). The path is now relativized to the session cwd when the rule lives in the project (`.omp/rules/no-any.md`), or `~`-relative when it lives under home, so no absolute path is fed into the agent's context outside the system prompt. +- Fixed `AsyncJobManager.instance()` being cleared while the owning top-level session was still live, which broke the `task` async path with "Async execution is enabled but no async job manager is available" until process restart. Any in-process secondary top-level `createAgentSession()` call (e.g. the Agent Control Center's create flow in `agent-dashboard.ts`) constructed a fresh `AsyncJobManager`, overwrote the singleton, and then cleared it on its own dispose. Secondary sessions now leave the live singleton untouched, and their dispose-time cleanup is scoped so it can no longer cancel the primary session's running bash/task jobs. `bash` / `task` / `job` tools and session job snapshots now resolve the manager through session-scoped async manager wiring rather than `AsyncJobManager.instance()`, so a secondary in-process top-level session cannot accidentally register background work on the owning session's manager or report the owning session's jobs; subagents still inherit the parent's manager via their scoped async manager. Startup failures after a top-level session installs its manager now clear and dispose that manager before the next session decides whether it can create its own ([#1923](https://github.com/can1357/oh-my-pi/issues/1923)). +- Fixed the `task` tool returning a hard `Async execution is enabled but no async job manager is available.` error when `async.enabled` was true but `AsyncJobManager.instance()` returned `undefined`, leaving `task` non-functional for the rest of the session. The tool now falls back to the existing synchronous execution path (which still runs subagents concurrently via `mapWithConcurrencyLimit`), and logs a warning so the missing-manager state stays diagnosable ([#1922](https://github.com/can1357/oh-my-pi/issues/1922)). +- Fixed Hindsight retain/recall/reflect calls staying pinned to the bank that was selected when the session started after the operator edited `hindsight.bankId`, `hindsight.bankIdPrefix`, or `hindsight.scoping` mid-session. The backend now subscribes to those settings via `onHindsightScopeChanged` and rebuilds the active `HindsightSessionState` against the recomputed scope, disposing the old state after flushing its queue so in-flight tool-initiated retains still land in the bank they were enqueued for. Also renamed `ensureBankMission` to `ensureBankExists` so a blank `bankMission` no longer skips bank creation entirely, and called it before mental-model bootstrap so `createMentalModel` is never the first POST against a missing bank. `AgentSession.dispose` now flushes the retain queue before clearing `#hindsightSessionState`, since the queue's identity guard would otherwise drop the spliced batch ([#1902](https://github.com/can1357/oh-my-pi/issues/1902)). +- Fixed `/tree` rendering a bare "No entries found" line on a fresh session where the only persisted entries are the `model_change` + `thinking_level_change` written by `sdk.ts` at startup — both are hidden by the tree-selector's default filter, so `#filteredNodes.length === 0` while `tree.length === 2` and the controller's `tree.length === 0` short-circuit never fired. The selector now splits the empty-state into three distinct shapes — truly empty tree, search query with no matches, and filter mode rejecting every entry — surfacing the cause and the recovery key (`Alt+A` to show all, `Backspace` to clear a stale search) so users on a fresh session can see immediately that the panel isn't broken ([#1909](https://github.com/can1357/oh-my-pi/issues/1909)). +- Fixed remote MCP OAuth refresh failures leaving stale credentials in `agent.db`: when the token endpoint returns a definitive failure (`invalid_grant`, `invalid_token`, `revoked`, plain 401/403 not classified as transient), `MCPManager#resolveAuthConfig` now drops the credential via `AuthStorage.remove(credentialId)` and skips re-attaching the dead `Authorization: Bearer …` header. Previously a revoked refresh token kept producing `401 invalid_token` on every MCP request and survived restarts, so users had to hand-clear the credential row to recover; the next connect now surfaces a clean auth error and `/mcp reauth ` (or `/mcp unauth`) recovers without restarting. Transient refresh failures (network/`fetch failed`/`ECONNREFUSED`) still fall back to the existing access token ([#1908](https://github.com/can1357/oh-my-pi/issues/1908)). +- Fixed `omp://docs` and `omp://docs/...` internal documentation URLs in the distributed package to resolve through the embedded documentation index instead of failing with `Documentation file not found` ([#1898](https://github.com/can1357/oh-my-pi/issues/1898)). +- Fixed the `github` discovery provider silently ignoring `.github/skills//SKILL.md`, GitHub's documented Agent Skills layout. The provider now registers a `skills` capability (priority 30, project-only) that scans `.github/skills/` non-recursively via `scanSkillsFromDir` with `requireDescription: true`, matching the Agent Skills spec and the sibling `native`/`omp-plugins` providers ([#1906](https://github.com/can1357/oh-my-pi/issues/1906)). +- Fixed inline images rendering as a wall of empty PUA box glyphs with laggy scrolling on Kitty-protocol terminals that do not honor Unicode placeholders (most notably WezTerm and tmux/screen passthrough to a non-Kitty outer terminal). The 15.9 placeholder rollout enabled the `U=1`/U+10EEEE grid for every Kitty-protocol path; it now defaults on only for `kitty` and `ghostty`, with `PI_NO_KITTY_PLACEHOLDERS=1` as a hard opt-out and `PI_KITTY_PLACEHOLDERS=1` as opt-in for terminals (e.g. wezterm nightlies) that have since added support ([#1877](https://github.com/can1357/oh-my-pi/issues/1877)). +- Fixed auto session-title generation failures being swallowed without an actionable diagnostic. Title generation now logs structured start, missing-model/API-key, provider-error, empty-result, and exception outcomes with the session id and resolved title model; the interactive auto-title caller also logs uncaught persistence/generation errors instead of dropping them. ([#1892](https://github.com/can1357/oh-my-pi/issues/1892)) +- Fixed `TranscriptContainer` reporting the live block boundary to the TUI again, so ED3-risk foreground streaming can append newly sealed transcript blocks to native scrollback once while deferring only the active live block. + ## [15.9.1] - 2026-06-04 ### Added @@ -29,7 +53,6 @@ ### Fixed - Fixed a streamed assistant message freezing at a partial prefix (e.g. only "Nat" of "Natives built, now…") on ED3-risk terminals (Ghostty/kitty/iTerm2/Alacritty), with the final text appearing only after a resize. `TranscriptContainer` freezes each non-live block by replaying its last live render, but render coalescing can finalize a block's content and append the next block within the same throttled frame — so the block was sealed at its stale mid-stream snapshot and never repainted until the next `thaw`. The block that was live on the previous render is now recomputed once on the live→frozen transition, sealing it at its final content. - - Fixed ACP/RPC stdio startup so protocol frames are no longer consumed as one-shot piped prompt input before the JSON-RPC transport starts. - Fixed `omp completions` to await the completion script write before exiting. - Fixed `AssistantMessageComponent` exposing its stable-prefix completion API again so streamed assistant messages remain unstable until explicitly completed. diff --git a/packages/coding-agent/package.json b/packages/coding-agent/package.json index b87565e10..d983390e5 100644 --- a/packages/coding-agent/package.json +++ b/packages/coding-agent/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-coding-agent", - "version": "15.9.1", + "version": "15.9.2", "description": "Coding agent CLI with read, bash, edit, write tools and session management", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/coding-agent/src/cli-commands.ts b/packages/coding-agent/src/cli-commands.ts index 8fa568001..c9ac00741 100644 --- a/packages/coding-agent/src/cli-commands.ts +++ b/packages/coding-agent/src/cli-commands.ts @@ -20,6 +20,7 @@ export const commands: CommandEntry[] = [ { name: "completions", load: () => import("./commands/completions").then(m => m.default) }, { name: "__complete", load: () => import("./commands/complete").then(m => m.default) }, { name: "config", load: () => import("./commands/config").then(m => m.default) }, + { name: "dry-balance", load: () => import("./commands/dry-balance").then(m => m.default) }, { name: "grep", load: () => import("./commands/grep").then(m => m.default) }, { name: "grievances", load: () => import("./commands/grievances").then(m => m.default) }, { name: "install", load: () => import("./commands/install").then(m => m.default) }, diff --git a/packages/coding-agent/src/cli/dry-balance-cli.ts b/packages/coding-agent/src/cli/dry-balance-cli.ts new file mode 100644 index 000000000..dc35d254f --- /dev/null +++ b/packages/coding-agent/src/cli/dry-balance-cli.ts @@ -0,0 +1,823 @@ +import type { + Api, + AssistantMessage, + AssistantMessageEvent, + AssistantMessageEventStream, + Context, + Model, + OAuthAccess, + OAuthAccessResolution, + SimpleStreamOptions, +} from "@oh-my-pi/pi-ai"; +import { streamSimple } from "@oh-my-pi/pi-ai"; +import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui"; +import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils"; +import chalk from "chalk"; +import type { CanonicalModelVariant } from "../config/model-equivalence"; +import { type CanonicalModelQueryOptions, ModelRegistry } from "../config/model-registry"; +import { + formatModelString, + type ModelMatchPreferences, + resolveAllowedModels, + resolveCliModel, + resolveModelRoleValue, +} from "../config/model-resolver"; +import { Settings } from "../config/settings"; +import dryBalanceBenchPrompt from "../prompts/dry-balance-bench.md" with { type: "text" }; +import { discoverAuthStorage } from "../sdk"; + +const DEFAULT_SAMPLE_COUNT = 100; +const DEFAULT_CONCURRENCY = 32; +const BENCH_MAX_TOKENS = 512; +const BENCH_RENDER_INTERVAL_MS = 80; +const BENCH_ACCOUNT_WIDTH = 60; +const BENCH_ERROR_WIDTH = 110; +const BENCH_SPINNER_FRAMES = ["⠋", "⠙", "⠹", "⠸", "⠼", "⠴", "⠦", "⠧", "⠇", "⠏"] as const; +const DRY_BALANCE_BENCH_PROMPT = dryBalanceBenchPrompt.trim(); + +export interface DryBalanceCommandArgs { + model?: string; + flags: { + model?: string; + count?: number; + concurrency?: number; + json?: boolean; + bench?: boolean; + }; +} + +export interface DryBalanceAuthOptions { + baseUrl?: string; + modelId?: string; + signal?: AbortSignal; +} + +export interface DryBalanceAuthStorage { + getOAuthAccess( + provider: string, + sessionId?: string, + options?: DryBalanceAuthOptions, + ): Promise; + getOAuthAccesses?(provider: string, options?: DryBalanceAuthOptions): Promise; +} + +export interface DryBalanceModelRegistry { + authStorage: DryBalanceAuthStorage; + getAll(): Model[]; + getAvailable(): Model[]; + getApiKey(model: Model, sessionId?: string): Promise; + getCanonicalVariants(canonicalId: string, options?: CanonicalModelQueryOptions): CanonicalModelVariant[]; + resolveCanonicalModel?(canonicalId: string, options?: CanonicalModelQueryOptions): Model | undefined; + getCanonicalId?(model: Model): string | undefined; +} + +export interface DryBalanceRuntime { + modelRegistry: DryBalanceModelRegistry; + settings?: Settings; + close?: () => void; +} + +export interface DryBalanceAccountStat { + account: string; + count: number; + percent: number; +} + +export interface DryBalanceFailureStat { + reason: string; + count: number; + percent: number; +} + +export interface DryBalanceBenchSuccessResult { + ok: true; + account: string; + ttftMs: number; + durationMs: number; + outputTokens: number; + tokensPerSecond: number; +} + +export interface DryBalanceBenchFailureResult { + ok: false; + account?: string; + error: string; +} + +export type DryBalanceBenchResult = DryBalanceBenchSuccessResult | DryBalanceBenchFailureResult; + +export interface DryBalanceBenchSummary { + total: number; + success: { + total: number; + averageTtftMs: number | null; + averageTokensPerSecond: number | null; + }; + failure: { + total: number; + reasons: DryBalanceFailureStat[]; + }; + results: DryBalanceBenchResult[]; +} + +export interface DryBalanceSummary { + model: string; + provider: string; + samples: number; + concurrency: number; + success: { + total: number; + accounts: DryBalanceAccountStat[]; + }; + failure: { + total: number; + reasons: DryBalanceFailureStat[]; + }; + bench?: DryBalanceBenchSummary; +} + +type DryBalanceStreamSimple = ( + model: Model, + context: Context, + options?: SimpleStreamOptions, +) => AssistantMessageEventStream; + +export interface DryBalanceDependencies { + createRuntime?: () => Promise; + randomSessionId?: () => string; + writeStdout?: (text: string) => void; + writeStderr?: (text: string) => void; + setExitCode?: (code: number) => void; + streamSimple?: DryBalanceStreamSimple; + now?: () => number; + stdoutIsTTY?: boolean; + stderrIsTTY?: boolean; +} + +type DryBalanceAttemptResult = + | { + ok: true; + account: string; + } + | { + ok: false; + reason: string; + }; + +type DryBalanceBenchProgressStatus = + | { state: "waiting" } + | { state: "running"; account: string } + | { state: "success"; result: DryBalanceBenchSuccessResult } + | { state: "failure"; result: DryBalanceBenchFailureResult }; + +interface DryBalanceBenchProgressSink { + markRunning(index: number, account: string): void; + complete(index: number, result: DryBalanceBenchResult): void; + close(): void; +} + +type DryBalanceBenchTarget = + | { + ok: true; + account: string; + accessToken: string; + } + | { + ok: false; + account: string; + error: string; + }; + +function normalizePositiveInteger(name: string, value: number | undefined, fallback: number): number { + const resolved = value ?? fallback; + if (!Number.isInteger(resolved) || resolved <= 0) { + throw new Error(`--${name} must be a positive integer`); + } + return resolved; +} + +function getErrorMessage(error: unknown): string { + if (error instanceof Error && error.message) return error.message; + const message = String(error); + return message ? message : "Unknown error"; +} + +function extractAccount(access: { + email?: string; + accountId?: string; + projectId?: string; + enterpriseUrl?: string; +}): string { + return access.email ?? access.accountId ?? access.projectId ?? access.enterpriseUrl ?? "(unknown oauth account)"; +} + +function getBenchTargetKey(access: { + credentialId?: number; + email?: string; + accountId?: string; + projectId?: string; + enterpriseUrl?: string; + accessToken?: string; +}): string { + return ( + access.email ?? + access.accountId ?? + access.projectId ?? + access.enterpriseUrl ?? + (access.credentialId === undefined ? access.accessToken : `credential:${access.credentialId}`) ?? + "(unknown oauth account)" + ); +} + +function sanitizeBenchText(text: string, width: number): string { + return truncateToWidth(replaceTabs(text).replace(/\r?\n/g, " "), width); +} + +function formatBenchIndex(index: number, total: number): string { + return `#${String(index + 1).padStart(String(total).length, "0")}`; +} + +function formatBenchAccount(account: string | undefined): string { + return account ? sanitizeBenchText(account, BENCH_ACCOUNT_WIDTH) : chalk.dim("(no account)"); +} + +function formatBenchDuration(ms: number): string { + return formatDuration(Math.max(0, Math.round(ms))); +} + +function formatBenchTps(tokensPerSecond: number): string { + return `${tokensPerSecond.toFixed(1)}/s`; +} + +function isBenchSuccess(result: DryBalanceBenchResult): result is DryBalanceBenchSuccessResult { + return result.ok; +} + +function isBenchFirstTokenEvent(event: AssistantMessageEvent): boolean { + switch (event.type) { + case "text_delta": + case "thinking_delta": + case "toolcall_delta": + return event.delta.length > 0; + case "text_end": + case "thinking_end": + return event.content.length > 0; + default: + return false; + } +} + +function resolveBenchMaxTokens(model: Model): number { + return Number.isFinite(model.maxTokens) && model.maxTokens > 0 + ? Math.min(BENCH_MAX_TOKENS, model.maxTokens) + : BENCH_MAX_TOKENS; +} + +function normalizeBenchMs(value: number): number { + return Number.isFinite(value) && value > 0 ? value : 0; +} + +function renderBenchResultLine(index: number, total: number, result: DryBalanceBenchResult): string { + const prefix = formatBenchIndex(index, total); + if (result.ok) { + return `${chalk.green("✓")} ${prefix} ${formatBenchAccount(result.account)} ${chalk.dim("TTFT")} ${formatBenchDuration( + result.ttftMs, + )} ${chalk.dim("TPS")} ${formatBenchTps(result.tokensPerSecond)}`; + } + return `${chalk.red("✗")} ${prefix} ${formatBenchAccount(result.account)} ${chalk.red( + sanitizeBenchText(result.error, BENCH_ERROR_WIDTH), + )}`; +} + +function renderBenchStatusLine( + status: DryBalanceBenchProgressStatus, + index: number, + total: number, + frame: number, +): string { + const prefix = formatBenchIndex(index, total); + switch (status.state) { + case "waiting": + return `${chalk.dim("○")} ${prefix} ${chalk.dim("waiting")}`; + case "running": { + const spinner = BENCH_SPINNER_FRAMES[frame % BENCH_SPINNER_FRAMES.length] ?? "*"; + return `${chalk.yellow(spinner)} ${prefix} ${formatBenchAccount(status.account)} ${chalk.dim("sending request")}`; + } + case "success": + return renderBenchResultLine(index, total, status.result); + case "failure": + return renderBenchResultLine(index, total, status.result); + } +} + +function createBenchProgressSink( + total: number, + write: (text: string) => void, + interactive: boolean, +): DryBalanceBenchProgressSink { + const statuses: DryBalanceBenchProgressStatus[] = Array.from({ length: total }, () => ({ state: "waiting" })); + if (!interactive) { + return { + markRunning(index, account) { + statuses[index] = { state: "running", account }; + write(`${renderBenchStatusLine(statuses[index], index, total, 0)}\n`); + }, + complete(index, result) { + statuses[index] = result.ok ? { state: "success", result } : { state: "failure", result }; + write(`${renderBenchResultLine(index, total, result)}\n`); + }, + close() {}, + }; + } + + let frame = 0; + let lineCount = 0; + let timer: NodeJS.Timeout | undefined; + const render = (): void => { + const lines = [ + chalk.bold("bench requests"), + ...statuses.map((status, index) => renderBenchStatusLine(status, index, total, frame)), + ]; + if (lineCount > 0) write(`\x1b[${lineCount}A`); + write(`${lines.map(line => `\x1b[2K${line}`).join("\n")}\n`); + lineCount = lines.length; + }; + render(); + timer = setInterval(() => { + frame += 1; + render(); + }, BENCH_RENDER_INTERVAL_MS); + timer.unref?.(); + return { + markRunning(index, account) { + statuses[index] = { state: "running", account }; + render(); + }, + complete(index, result) { + statuses[index] = result.ok ? { state: "success", result } : { state: "failure", result }; + render(); + }, + close() { + if (timer) { + clearInterval(timer); + timer = undefined; + } + render(); + }, + }; +} + +async function runBenchRequest( + model: Model, + sessionId: string, + account: string, + accessToken: string, + streamFn: DryBalanceStreamSimple, + now: () => number, +): Promise { + const startedAt = now(); + let firstTokenAt: number | undefined; + try { + const context: Context = { + messages: [ + { + role: "user", + content: DRY_BALANCE_BENCH_PROMPT, + timestamp: Date.now(), + attribution: "user", + }, + ], + }; + const stream = streamFn(model, context, { + apiKey: accessToken, + sessionId, + maxTokens: resolveBenchMaxTokens(model), + temperature: 0.2, + disableReasoning: true, + hideThinkingSummary: true, + }); + let message: AssistantMessage | undefined; + for await (const event of stream) { + if (firstTokenAt === undefined && isBenchFirstTokenEvent(event)) { + firstTokenAt = now(); + } + if (event.type === "error") { + return { ok: false, account, error: event.error.errorMessage ?? "request failed" }; + } + if (event.type === "done") { + message = event.message; + } + } + message ??= await stream.result(); + if (message.stopReason === "error" || message.errorMessage) { + return { ok: false, account, error: message.errorMessage ?? "request failed" }; + } + const durationMs = normalizeBenchMs(message.duration ?? now() - startedAt); + const ttftMs = normalizeBenchMs( + message.ttft ?? (firstTokenAt === undefined ? durationMs : firstTokenAt - startedAt), + ); + const outputTokens = Number.isFinite(message.usage.output) && message.usage.output > 0 ? message.usage.output : 0; + const tokensPerSecond = durationMs > 0 ? (outputTokens * 1000) / durationMs : 0; + return { + ok: true, + account, + ttftMs, + durationMs, + outputTokens, + tokensPerSecond, + }; + } catch (error) { + return { ok: false, account, error: getErrorMessage(error) }; + } +} + +async function resolveBenchTargets( + model: Model, + authStorage: DryBalanceAuthStorage, +): Promise { + const resolved = authStorage.getOAuthAccesses + ? await authStorage.getOAuthAccesses(model.provider, { + baseUrl: model.baseUrl, + modelId: model.id, + }) + : await authStorage + .getOAuthAccess(model.provider, undefined, { + baseUrl: model.baseUrl, + modelId: model.id, + }) + .then(access => (access ? [{ ok: true as const, ...access }] : [])); + const targets: DryBalanceBenchTarget[] = []; + const seen = new Set(); + for (const entry of resolved) { + const key = getBenchTargetKey(entry); + if (seen.has(key)) continue; + seen.add(key); + const account = extractAccount(entry); + if (entry.ok) { + targets.push({ ok: true, account, accessToken: entry.accessToken }); + } else { + targets.push({ ok: false, account, error: entry.error }); + } + } + return targets; +} + +async function runBenchTargets( + model: Model, + targets: DryBalanceBenchTarget[], + randomSessionId: () => string, + progress: DryBalanceBenchProgressSink | undefined, + streamFn: DryBalanceStreamSimple, + now: () => number, +): Promise { + return Promise.all( + targets.map(async (target, index) => { + if (!target.ok) { + const result: DryBalanceBenchFailureResult = { + ok: false, + account: target.account, + error: target.error, + }; + progress?.complete(index, result); + return result; + } + progress?.markRunning(index, target.account); + const result = await runBenchRequest( + model, + randomSessionId(), + target.account, + target.accessToken, + streamFn, + now, + ); + progress?.complete(index, result); + return result; + }), + ); +} + +async function createDefaultRuntime(): Promise { + const authStorage = await discoverAuthStorage(); + try { + const settings = await Settings.init({ cwd: getProjectDir() }); + const modelRegistry = new ModelRegistry(authStorage); + return { + modelRegistry, + settings, + close: () => authStorage.close(), + }; + } catch (error) { + authStorage.close(); + throw error; + } +} + +async function resolveDryBalanceModel( + modelSelector: string | undefined, + modelRegistry: DryBalanceModelRegistry, + settings: Settings | undefined, + randomSessionId: () => string, +): Promise<{ model: Model; warning?: string }> { + const preferences: ModelMatchPreferences = { + usageOrder: settings?.getStorage()?.getModelUsageOrder(), + }; + if (modelSelector) { + const resolved = resolveCliModel({ + cliModel: modelSelector, + modelRegistry, + preferences, + }); + if (resolved.error) throw new Error(resolved.error); + if (!resolved.model) throw new Error(`Model "${modelSelector}" not found`); + return { model: resolved.model, warning: resolved.warning }; + } + + const allowedModels = await resolveAllowedModels(modelRegistry, settings, preferences); + if (allowedModels.length === 0) { + throw new Error( + "No models available. Use --model to select a model or configure enabledModels/default model settings.", + ); + } + + const defaultRoleSpec = resolveModelRoleValue(settings?.getModelRole("default"), allowedModels, { + settings, + matchPreferences: preferences, + modelRegistry, + }); + if (defaultRoleSpec.model) { + return { model: defaultRoleSpec.model, warning: defaultRoleSpec.warning }; + } + + for (const candidate of allowedModels) { + const apiKey = await modelRegistry.getApiKey(candidate, randomSessionId()); + if (apiKey) return { model: candidate }; + } + + return { + model: allowedModels[0], + warning: + "No allowed model had usable credentials during default resolution; dry-balance will report OAuth failures for the first allowed model.", + }; +} + +async function runOneAttempt( + model: Model, + modelRegistry: DryBalanceModelRegistry, + sessionId: string, +): Promise { + try { + // AuthStorage.getOAuthAccess shares the OAuth credential ranking, refresh, + // usage-limit, broker, and session-sticky path used by getApiKey(), while + // returning the selected account metadata instead of bearer bytes. + const access = await modelRegistry.authStorage.getOAuthAccess(model.provider, sessionId, { + baseUrl: model.baseUrl, + modelId: model.id, + }); + if (!access) return { ok: false, reason: "no OAuth access resolved" }; + return { ok: true, account: extractAccount(access) }; + } catch (error) { + return { ok: false, reason: getErrorMessage(error) }; + } +} + +async function mapConcurrent( + items: T[], + concurrency: number, + fn: (item: T, index: number) => Promise, +): Promise { + const results = new Array(items.length); + let nextIndex = 0; + const workerCount = Math.min(concurrency, items.length); + await Promise.all( + Array.from({ length: workerCount }, async () => { + while (true) { + const index = nextIndex; + nextIndex += 1; + if (index >= items.length) return; + results[index] = await fn(items[index], index); + } + }), + ); + return results; +} + +function sortedStats( + map: Map, + samples: number, +): Array<{ label: string; count: number; percent: number }> { + return [...map.entries()] + .map(([label, count]) => ({ label, count, percent: (count / samples) * 100 })) + .sort((left, right) => right.count - left.count || left.label.localeCompare(right.label)); +} + +function summarizeBenchResults(results: DryBalanceBenchResult[]): DryBalanceBenchSummary | undefined { + if (results.length === 0) return undefined; + const successes = results.filter(isBenchSuccess); + const failureReasons = new Map(); + for (const result of results) { + if (!result.ok) { + failureReasons.set(result.error, (failureReasons.get(result.error) ?? 0) + 1); + } + } + const average = (values: number[]): number | null => + values.length === 0 ? null : values.reduce((sum, value) => sum + value, 0) / values.length; + return { + total: results.length, + success: { + total: successes.length, + averageTtftMs: average(successes.map(result => result.ttftMs)), + averageTokensPerSecond: average(successes.map(result => result.tokensPerSecond)), + }, + failure: { + total: results.length - successes.length, + reasons: sortedStats(failureReasons, results.length).map(stat => ({ + reason: stat.label, + count: stat.count, + percent: stat.percent, + })), + }, + results, + }; +} + +function summarizeResults( + model: Model, + samples: number, + concurrency: number, + results: DryBalanceAttemptResult[], +): DryBalanceSummary { + const accounts = new Map(); + const reasons = new Map(); + for (const result of results) { + if (result.ok) { + accounts.set(result.account, (accounts.get(result.account) ?? 0) + 1); + } else { + reasons.set(result.reason, (reasons.get(result.reason) ?? 0) + 1); + } + } + const accountStats: DryBalanceAccountStat[] = sortedStats(accounts, samples).map(stat => ({ + account: stat.label, + count: stat.count, + percent: stat.percent, + })); + const failureStats: DryBalanceFailureStat[] = sortedStats(reasons, samples).map(stat => ({ + reason: stat.label, + count: stat.count, + percent: stat.percent, + })); + const summary: DryBalanceSummary = { + model: formatModelString(model), + provider: model.provider, + samples, + concurrency, + success: { + total: results.filter(result => result.ok).length, + accounts: accountStats, + }, + failure: { + total: results.filter(result => !result.ok).length, + reasons: failureStats, + }, + }; + return summary; +} + +function formatRows(rows: Array<{ count: number; percent: number; label: string }>): string[] { + if (rows.length === 0) return [` ${chalk.dim("(none)")}`]; + const maxCountWidth = Math.max(...rows.map(row => row.count.toString().length)); + return rows.map(row => { + const count = row.count.toString().padStart(maxCountWidth); + const percent = `${row.percent.toFixed(1)}%`.padStart(6); + return ` ${count} ${percent} ${row.label}`; + }); +} + +export function formatDryBalanceText(summary: DryBalanceSummary): string { + const accountRows = summary.success.accounts.map(row => ({ + count: row.count, + percent: row.percent, + label: row.account, + })); + const failureRows = summary.failure.reasons.map(row => ({ + count: row.count, + percent: row.percent, + label: row.reason, + })); + const lines = [ + chalk.bold("dry-balance"), + `model: ${summary.model}`, + `provider: ${summary.provider}`, + `samples: ${summary.samples}`, + `concurrency: ${summary.concurrency}`, + "", + `${chalk.green("success")} ${summary.success.total}`, + ...formatRows(accountRows), + "", + `${summary.failure.total > 0 ? chalk.red("failure") : chalk.dim("failure")} ${summary.failure.total}`, + ...formatRows(failureRows), + ]; + if (summary.bench) { + const avgTtft = + summary.bench.success.averageTtftMs === null ? "-" : formatBenchDuration(summary.bench.success.averageTtftMs); + const avgTps = + summary.bench.success.averageTokensPerSecond === null + ? "-" + : formatBenchTps(summary.bench.success.averageTokensPerSecond); + const benchFailureRows = summary.bench.failure.reasons.map(row => ({ + count: row.count, + percent: row.percent, + label: row.reason, + })); + lines.push( + "", + chalk.bold("bench"), + `requests: ${summary.bench.total}`, + `${chalk.green("success")} ${summary.bench.success.total}`, + `avg TTFT: ${avgTtft}`, + `avg TPS: ${avgTps}`, + "", + `${summary.bench.failure.total > 0 ? chalk.red("failure") : chalk.dim("failure")} ${summary.bench.failure.total}`, + ...formatRows(benchFailureRows), + ); + } + return `${lines.join("\n")}\n`; +} + +export async function runDryBalanceCommand( + command: DryBalanceCommandArgs, + deps: DryBalanceDependencies = {}, +): Promise { + const isBench = command.flags.bench === true; + const samples = isBench ? 0 : normalizePositiveInteger("count", command.flags.count, DEFAULT_SAMPLE_COUNT); + const concurrency = isBench + ? 0 + : Math.min(samples, normalizePositiveInteger("concurrency", command.flags.concurrency, DEFAULT_CONCURRENCY)); + const randomSessionId = deps.randomSessionId ?? (() => Bun.randomUUIDv7()); + const writeStdout = deps.writeStdout ?? ((text: string) => process.stdout.write(text)); + const writeStderr = deps.writeStderr ?? ((text: string) => process.stderr.write(text)); + const setExitCode = + deps.setExitCode ?? + ((code: number) => { + process.exitCode = code; + }); + const streamFn = deps.streamSimple ?? streamSimple; + const now = deps.now ?? (() => performance.now()); + const runtime = await (deps.createRuntime ?? createDefaultRuntime)(); + let progress: DryBalanceBenchProgressSink | undefined; + let progressClosed = false; + const closeProgress = (): void => { + if (progressClosed) return; + progressClosed = true; + progress?.close(); + }; + try { + const modelSelector = command.flags.model ?? command.model; + const { model, warning } = await resolveDryBalanceModel( + modelSelector, + runtime.modelRegistry, + runtime.settings, + randomSessionId, + ); + if (warning) writeStderr(`${chalk.yellow(`Warning: ${warning}`)}\n`); + let results: DryBalanceAttemptResult[]; + let benchResults: DryBalanceBenchResult[] | undefined; + let summarySamples = samples; + let summaryConcurrency = concurrency; + if (isBench) { + const targets = await resolveBenchTargets(model, runtime.modelRegistry.authStorage); + if (targets.length === 0) throw new Error(`No OAuth accounts resolved for provider ${model.provider}`); + summarySamples = targets.length; + summaryConcurrency = targets.length; + const progressWrite = command.flags.json ? writeStderr : writeStdout; + const progressInteractive = command.flags.json + ? (deps.stderrIsTTY ?? process.stderr.isTTY === true) + : (deps.stdoutIsTTY ?? process.stdout.isTTY === true); + progress = createBenchProgressSink(targets.length, progressWrite, progressInteractive); + benchResults = await runBenchTargets(model, targets, randomSessionId, progress, streamFn, now); + results = targets.map(target => + target.ok ? { ok: true, account: target.account } : { ok: false, reason: target.error }, + ); + } else { + const sessionIds = Array.from({ length: samples }, () => randomSessionId()); + results = await mapConcurrent(sessionIds, concurrency, sessionId => + runOneAttempt(model, runtime.modelRegistry, sessionId), + ); + } + closeProgress(); + const summary = summarizeResults(model, summarySamples, summaryConcurrency, results); + if (benchResults) { + const benchSummary = summarizeBenchResults(benchResults); + if (benchSummary) summary.bench = benchSummary; + } + if (command.flags.json) { + writeStdout(`${JSON.stringify(summary, null, 2)}\n`); + } else { + writeStdout(formatDryBalanceText(summary)); + } + if (summary.failure.total > 0 || (summary.bench?.failure.total ?? 0) > 0) setExitCode(1); + return summary; + } finally { + closeProgress(); + runtime.close?.(); + } +} diff --git a/packages/coding-agent/src/commands/dry-balance.ts b/packages/coding-agent/src/commands/dry-balance.ts new file mode 100644 index 000000000..e27763014 --- /dev/null +++ b/packages/coding-agent/src/commands/dry-balance.ts @@ -0,0 +1,43 @@ +import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli"; +import { runDryBalanceCommand } from "../cli/dry-balance-cli"; + +export default class DryBalance extends Command { + static description = "Dry-run OAuth account balancing across random session ids"; + + static args = { + model: Args.string({ + description: "Model selector (provider/model or fuzzy id). Defaults to the configured default model.", + required: false, + }), + }; + + static flags = { + model: Flags.string({ description: "Model selector (same syntax as --model on omp)" }), + count: Flags.integer({ description: "Number of random session ids to try", default: 100 }), + concurrency: Flags.integer({ description: "Maximum concurrent credential resolutions", default: 32 }), + json: Flags.boolean({ description: "Output JSON" }), + bench: Flags.boolean({ description: "Send one live benchmark request per OAuth account" }), + }; + + static examples = [ + "# Dry-run the configured default model with 100 random session ids\n omp dry-balance", + "# Dry-run a specific model\n omp dry-balance anthropic/claude-sonnet-4-5", + "# Larger run with bounded concurrency\n omp dry-balance --model openai-codex/gpt-5-codex --count 1000 --concurrency 64", + "# Benchmark every OAuth account in parallel\n omp dry-balance --bench", + "# Machine-readable output\n omp dry-balance --json", + ]; + + async run(): Promise { + const { args, flags } = await this.parse(DryBalance); + await runDryBalanceCommand({ + model: args.model, + flags: { + model: flags.model, + count: flags.count, + concurrency: flags.concurrency, + json: flags.json, + bench: flags.bench, + }, + }); + } +} diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index 20f723336..3f0008b43 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -547,6 +547,7 @@ function applyModelOverride(model: Model, override: ModelOverride): Model; compat?: Model["compat"]; contextPromotionTarget?: string; @@ -597,6 +599,7 @@ type CustomModelOverlay = { cost?: { input: number; output: number; cacheRead: number; cacheWrite: number }; contextWindow?: number; maxTokens?: number; + omitMaxOutputTokens?: boolean; headers?: Record; compat?: Model["compat"]; contextPromotionTarget?: string; @@ -667,6 +670,7 @@ function buildCustomModelOverlay( cost: modelDef.cost, contextWindow: modelDef.contextWindow, maxTokens: modelDef.maxTokens, + omitMaxOutputTokens: modelDef.omitMaxOutputTokens, headers: mergeCustomModelHeaders(providerHeaders, modelDef.headers, authHeader, providerApiKey), compat: mergeCompat(providerCompat, modelDef.compat), contextPromotionTarget: modelDef.contextPromotionTarget, @@ -823,6 +827,7 @@ function finalizeCustomModel(model: CustomModelOverlay, options: CustomModelBuil resolvedModel.contextWindow ?? reference?.contextWindow ?? (options.useDefaults ? 128000 : undefined), maxTokens: resolvedModel.maxTokens ?? reference?.maxTokens ?? (options.useDefaults ? 16384 : undefined), headers: resolvedModel.headers, + omitMaxOutputTokens: resolvedModel.omitMaxOutputTokens ?? reference?.omitMaxOutputTokens, compat: mergeCompat(reference?.compat, resolvedModel.compat), contextPromotionTarget: resolvedModel.contextPromotionTarget, premiumMultiplier: resolvedModel.premiumMultiplier, @@ -1124,6 +1129,7 @@ export class ModelRegistry { cost: customModel.cost ?? existingModel.cost, contextWindow: customModel.contextWindow ?? existingModel.contextWindow, maxTokens: customModel.maxTokens ?? existingModel.maxTokens, + omitMaxOutputTokens: customModel.omitMaxOutputTokens ?? existingModel.omitMaxOutputTokens, // Same-id custom definitions replace bundled transport behavior. Provider-level // headers/compat were already folded into customModel during parsing; do not // re-merge bundled transport metadata here. diff --git a/packages/coding-agent/src/config/models-config-schema.ts b/packages/coding-agent/src/config/models-config-schema.ts index 965d85d33..1911651bb 100644 --- a/packages/coding-agent/src/config/models-config-schema.ts +++ b/packages/coding-agent/src/config/models-config-schema.ts @@ -93,6 +93,7 @@ const ModelDefinitionSchema = z.object({ premiumMultiplier: z.number().optional(), contextWindow: z.number().optional(), maxTokens: z.number().optional(), + omitMaxOutputTokens: z.boolean().optional(), headers: z.record(z.string(), z.string()).optional(), compat: OpenAICompatSchema.optional(), contextPromotionTarget: z.string().min(1).optional(), @@ -114,6 +115,7 @@ export const ModelOverrideSchema = z.object({ premiumMultiplier: z.number().optional(), contextWindow: z.number().optional(), maxTokens: z.number().optional(), + omitMaxOutputTokens: z.boolean().optional(), headers: z.record(z.string(), z.string()).optional(), compat: OpenAICompatSchema.optional(), contextPromotionTarget: z.string().min(1).optional(), diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index c735aa770..f5dc33299 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -907,6 +907,9 @@ const SETTING_HOOKS: Partial>> = { for (const cb of appendOnlyModeCallbacks) cb(value); } }, + "hindsight.bankId": () => fireHindsightScopeChanged(), + "hindsight.bankIdPrefix": () => fireHindsightScopeChanged(), + "hindsight.scoping": () => fireHindsightScopeChanged(), }; /** Callbacks invoked when `provider.appendOnlyContext` changes at runtime. */ const appendOnlyModeCallbacks = new Set<(value: string) => void>(); @@ -923,6 +926,41 @@ export function onAppendOnlyModeChanged(cb: (value: string) => void): () => void }; } +/** Callbacks fired when any `hindsight.bankId` / `bankIdPrefix` / `scoping` value changes. */ +const hindsightScopeCallbacks = new Set<() => void>(); + +function fireHindsightScopeChanged(): void { + // Snapshot the callback set before invoking — a callback's body is allowed + // to subscribe a NEW callback (the Hindsight backend re-registers the + // fresh state's listener on every rebuild). Iterating the live Set would + // re-invoke those just-added callbacks within the same fire, which spins + // in place: subscribe → invoke → subscribe → invoke → … + for (const cb of [...hindsightScopeCallbacks]) { + try { + cb(); + } catch (err) { + logger.warn("Settings: hindsight scope hook failed", { error: String(err) }); + } + } +} + +/** + * Subscribe to changes in the Hindsight bank-scoping settings. Lets the + * Hindsight backend rebuild the active `HindsightSessionState` when the + * operator switches `hindsight.bankId`, `hindsight.bankIdPrefix`, or + * `hindsight.scoping` mid-session so subsequent retain/recall calls land in + * the new bank instead of the one selected at session start. + * + * Returns an unsubscribe function. The callback receives no arguments — the + * caller is expected to re-read the relevant settings via `Settings.get`. + */ +export function onHindsightScopeChanged(cb: () => void): () => void { + hindsightScopeCallbacks.add(cb); + return () => { + hindsightScopeCallbacks.delete(cb); + }; +} + // ═══════════════════════════════════════════════════════════════════════════ // Global Singleton // ═══════════════════════════════════════════════════════════════════════════ diff --git a/packages/coding-agent/src/discovery/github.ts b/packages/coding-agent/src/discovery/github.ts index 0a92df7ff..bb2a200c5 100644 --- a/packages/coding-agent/src/discovery/github.ts +++ b/packages/coding-agent/src/discovery/github.ts @@ -10,6 +10,7 @@ * Capabilities: * - context-files: copilot-instructions.md in .github/ * - instructions: *.instructions.md in .github/instructions/ with applyTo frontmatter + * - skills: /SKILL.md in .github/skills/ (GitHub Agent Skills layout) */ import * as path from "node:path"; import { parseFrontmatter } from "@oh-my-pi/pi-utils"; @@ -17,9 +18,10 @@ import { registerProvider } from "../capability"; import { type ContextFile, contextFileCapability } from "../capability/context-file"; import { readFile } from "../capability/fs"; import { type Instruction, instructionCapability } from "../capability/instruction"; +import { type Skill, skillCapability } from "../capability/skill"; import type { LoadContext, LoadResult, SourceMeta } from "../capability/types"; -import { calculateDepth, createSourceMeta, getProjectPath, loadFilesFromDir } from "./helpers"; +import { calculateDepth, createSourceMeta, getProjectPath, loadFilesFromDir, scanSkillsFromDir } from "./helpers"; const PROVIDER_ID = "github"; const DISPLAY_NAME = "GitHub Copilot"; @@ -97,6 +99,32 @@ function transformInstruction(name: string, content: string, filePath: string, s }; } +// ============================================================================= +// Skills +// ============================================================================= + +/** + * Load skills from `.github/skills//SKILL.md`. + * + * GitHub documents this layout for Copilot Agent Skills and matches the + * non-recursive shape `scanSkillsFromDir` already expects. `requireDescription` + * is on to match the Agent Skills spec (name + description are mandatory) and + * the sibling `native`/`omp-plugins` providers. + * + * @see https://docs.github.com/en/copilot/how-tos/copilot-on-github/customize-copilot/customize-cloud-agent/add-skills + */ +async function loadSkills(ctx: LoadContext): Promise> { + const skillsDir = getProjectPath(ctx, "github", "skills"); + if (!skillsDir) return { items: [], warnings: [] }; + + return scanSkillsFromDir(ctx, { + dir: skillsDir, + providerId: PROVIDER_ID, + level: "project", + requireDescription: true, + }); +} + // ============================================================================= // Provider Registration // ============================================================================= @@ -116,3 +144,11 @@ registerProvider(instructionCapability.id, { priority: PRIORITY, load: loadInstructions, }); + +registerProvider(skillCapability.id, { + id: PROVIDER_ID, + displayName: DISPLAY_NAME, + description: "Load skills from .github/skills/*/SKILL.md", + priority: PRIORITY, + load: loadSkills, +}); diff --git a/packages/coding-agent/src/discovery/helpers.ts b/packages/coding-agent/src/discovery/helpers.ts index 29168523e..dc6eba68f 100644 --- a/packages/coding-agent/src/discovery/helpers.ts +++ b/packages/coding-agent/src/discovery/helpers.ts @@ -218,6 +218,7 @@ export interface ParsedAgentFields { output?: unknown; thinkingLevel?: ThinkingLevel; autoloadSkills?: string[]; + readSummarize?: boolean; blocking?: boolean; } @@ -271,10 +272,11 @@ export function parseAgentFields(frontmatter: Record): ParsedAg const thinkingLevel = parseThinkingLevel(rawThinkingLevel); const model = parseModelList(frontmatter.model); const blocking = parseBoolean(frontmatter.blocking); + const readSummarize = parseBoolean(frontmatter.readSummarize); const autoloadSkills = parseArrayOrCSV(frontmatter.autoloadSkills) ?.map(s => s.trim()) .filter(Boolean); - return { name, description, tools, spawns, model, output, thinkingLevel, blocking, autoloadSkills }; + return { name, description, tools, spawns, model, output, thinkingLevel, blocking, autoloadSkills, readSummarize }; } async function globIf( diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index 90def20a2..70817b916 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -41,10 +41,15 @@ const PI_SUBPATH_REMAPS: ReadonlyMap = new Map([ const LEGACY_PI_SPECIFIER_FILTER = new RegExp(`^@(?:${PI_SCOPE_ALTERNATION})/(?:${PI_PACKAGE_ALTERNATION})(?:/.*)?$`); const LEGACY_PI_IMPORT_SPECIFIER_REGEX = new RegExp( - `((?:from\\s+|import\\s*\\(\\s*)["'])(@(?:${PI_SCOPE_ALTERNATION})/(?:${PI_PACKAGE_ALTERNATION})(?:/[^"'()\\s]+)?)(["'])`, + `((?:from\\s+|import\\s+|import\\s*\\(\\s*)["'])(@(?:${PI_SCOPE_ALTERNATION})/(?:${PI_PACKAGE_ALTERNATION})(?:/[^"'()\\s]+)?)(["'])`, "g", ); const resolvedSpecifierFallbacks = new Map(); +const SOURCE_MODULE_EXTENSIONS = [".ts", ".tsx", ".mts", ".cts", ".js", ".jsx", ".mjs", ".cjs"] as const; +const SUPPORTED_PACKAGE_IMPORT_CONDITIONS = new Set(["bun", "node", "import", "default"]); +const packageRootCache = new Map(); +const packageImportsCache = new Map | null>(); +const PACKAGE_IMPORT_EXCLUDED = Symbol("packageImportExcluded"); // Extensions that imported `@sinclair/typebox` directly used to resolve against a // real `@sinclair/typebox` install. The runtime dep was replaced with the Zod-backed @@ -221,33 +226,245 @@ function rewriteLegacyPiImports(source: string): string { // Match the bare `@sinclair/typebox` import specifier (static + dynamic). // Subpath imports like `@sinclair/typebox/compiler` are intentionally excluded — // they expose TypeBox-only APIs the Zod-backed shim does not provide. -const TYPEBOX_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s*\(\s*)["'])(@sinclair\/typebox)(["'])/g; +const TYPEBOX_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s+|import\s*\(\s*)["'])(@sinclair\/typebox)(["'])/g; /** - * Rewrite the legacy specifiers a Pi extension may import — `@(scope)/pi-*` and - * the bare `@sinclair/typebox` root — to absolute `file://` URLs pointing at the - * bundled package or compat shim. Every other specifier (relative siblings, the - * extension's own bare dependencies) is left untouched so Bun resolves it + * Rewrite the extension-owned specifiers OMP must host-resolve — legacy + * `@(scope)/pi-*`, bare `@sinclair/typebox`, and package `imports` aliases like + * `#src/*` — to absolute `file://` URLs. Every other specifier (relative + * siblings and third-party dependencies) is left untouched so Bun resolves it * natively from the extension's real on-disk location. */ -function rewriteLegacyExtensionSource(source: string): string { +async function rewriteLegacyExtensionSource(source: string, importerPath: string): Promise { const withPi = rewriteLegacyPiImports(source); - return withPi.replace( + const withTypeBox = withPi.replace( TYPEBOX_IMPORT_SPECIFIER_REGEX, (_match, prefix: string, _specifier: string, suffix: string) => { return `${prefix}${toImportSpecifier(TYPEBOX_SHIM_PATH)}${suffix}`; }, ); + return rewriteExtensionPackageImports(withTypeBox, importerPath); +} + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +async function pathExists(p: string): Promise { + try { + await fs.stat(p); + return true; + } catch { + return false; + } +} + +function hasSourceModuleExtension(p: string): boolean { + const ext = path.extname(p).toLowerCase(); + return (SOURCE_MODULE_EXTENSIONS as readonly string[]).includes(ext); +} + +async function resolveSourceModuleFile(basePath: string): Promise { + try { + const stats = await fs.stat(basePath); + if (stats.isFile()) { + // Non-source files (JSON, WASM, text assets, etc.) bypass the on-load + // rewrite hook so Bun's native loaders handle them; our hook would + // otherwise pass them through `getLoader()` which falls back to `js`. + return hasSourceModuleExtension(basePath) ? realpathOrSelf(basePath) : null; + } + if (stats.isDirectory()) { + for (const extension of SOURCE_MODULE_EXTENSIONS) { + const resolved = await resolveSourceModuleFile(path.join(basePath, `index${extension}`)); + if (resolved) return resolved; + } + } + } catch { + // Fall through to extension candidates below. + } + + if (path.extname(basePath)) { + return null; + } + + for (const extension of SOURCE_MODULE_EXTENSIONS) { + const resolved = await resolveSourceModuleFile(`${basePath}${extension}`); + if (resolved) return resolved; + } + return null; +} + +async function findPackageRoot(importerPath: string): Promise { + let dir = path.dirname(importerPath); + while (true) { + const cached = packageRootCache.get(dir); + if (cached !== undefined) { + return cached; + } + + if (await pathExists(path.join(dir, "package.json"))) { + packageRootCache.set(path.dirname(importerPath), dir); + return dir; + } + + const parent = path.dirname(dir); + if (parent === dir) { + packageRootCache.set(path.dirname(importerPath), null); + return null; + } + dir = parent; + } +} + +async function readPackageImports(packageRoot: string): Promise | null> { + const cached = packageImportsCache.get(packageRoot); + if (cached !== undefined) { + return cached; + } + + let imports: Record | null = null; + try { + const pkg = await Bun.file(path.join(packageRoot, "package.json")).json(); + if (isRecord(pkg) && isRecord(pkg.imports)) { + imports = pkg.imports; + } + } catch { + imports = null; + } + packageImportsCache.set(packageRoot, imports); + return imports; +} + +type PackageImportTargetSelection = string | typeof PACKAGE_IMPORT_EXCLUDED | null; +type ResolvedPackageImportTargetSelection = string | typeof PACKAGE_IMPORT_EXCLUDED; + +function selectPackageImportTarget(entry: unknown): PackageImportTargetSelection { + if (entry === null) { + return PACKAGE_IMPORT_EXCLUDED; + } + if (typeof entry === "string") { + return entry; + } + if (Array.isArray(entry)) { + for (const item of entry) { + const target = selectPackageImportTarget(item); + if (target !== null) return target; + } + return null; + } + if (!isRecord(entry)) { + return null; + } + for (const [condition, value] of Object.entries(entry)) { + if (!SUPPORTED_PACKAGE_IMPORT_CONDITIONS.has(condition)) { + continue; + } + const target = selectPackageImportTarget(value); + if (target !== null) return target; + } + return null; +} + +async function resolvePackageImportTarget( + packageRoot: string, + target: string, + wildcard: string | null, +): Promise { + if (!target.startsWith("./")) { + return null; + } + const substituted = wildcard === null ? target : target.replaceAll("*", wildcard); + return resolveSourceModuleFile(path.resolve(packageRoot, substituted)); +} + +async function resolvePackageImportSpecifier(specifier: string, importerPath: string): Promise { + if (!specifier.startsWith("#")) { + return null; + } + + const packageRoot = await findPackageRoot(importerPath); + if (!packageRoot) { + return null; + } + + const imports = await readPackageImports(packageRoot); + if (!imports) { + return null; + } + + const exactTarget = selectPackageImportTarget(imports[specifier]); + if (exactTarget === PACKAGE_IMPORT_EXCLUDED) { + return null; + } + if (exactTarget !== null) { + return resolvePackageImportTarget(packageRoot, exactTarget, null); + } + + let bestMatch: { keyLength: number; target: ResolvedPackageImportTargetSelection; wildcard: string } | null = null; + for (const [key, entry] of Object.entries(imports)) { + const starIndex = key.indexOf("*"); + if (starIndex === -1) continue; + + const prefix = key.slice(0, starIndex); + const suffix = key.slice(starIndex + 1); + if (!specifier.startsWith(prefix) || !specifier.endsWith(suffix)) { + continue; + } + + const target = selectPackageImportTarget(entry); + if (target === null) { + continue; + } + + if (!bestMatch || key.length > bestMatch.keyLength) { + bestMatch = { + keyLength: key.length, + target, + wildcard: specifier.slice(prefix.length, specifier.length - suffix.length), + }; + } + } + + if (!bestMatch || bestMatch.target === PACKAGE_IMPORT_EXCLUDED) { + return null; + } + return resolvePackageImportTarget(packageRoot, bestMatch.target, bestMatch.wildcard); +} + +const PACKAGE_IMPORT_SPECIFIER_REGEX = /((?:from\s+|import\s+|import\s*\(\s*)["'])(#[^"'()\s]+)(["'])/g; + +async function rewriteExtensionPackageImports(source: string, importerPath: string): Promise { + let rewritten = ""; + let lastIndex = 0; + for (const match of source.matchAll(PACKAGE_IMPORT_SPECIFIER_REGEX)) { + const matchIndex = match.index; + if (matchIndex === undefined) continue; + + const [fullMatch, prefix, specifier, suffix] = match; + if (!prefix || !specifier || !suffix) continue; + + const resolved = await resolvePackageImportSpecifier(specifier, importerPath); + if (!resolved) continue; + + rewritten += source.slice(lastIndex, matchIndex); + rewritten += `${prefix}${toImportSpecifier(resolved)}${suffix}`; + lastIndex = matchIndex + fullMatch.length; + } + + if (lastIndex === 0) { + return source; + } + return `${rewritten}${source.slice(lastIndex)}`; } function escapeRegExp(value: string): string { return value.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); } -// Match relative import specifiers (static `from "./…"` and dynamic -// `import("./…")`). Used to walk an extension's own module graph; bare and -// absolute specifiers are deliberately excluded. -const RELATIVE_IMPORT_SPECIFIER_REGEX = /(?:from\s+|import\s*\(\s*)["'](\.\.?\/[^"']+)["']/g; +// Match source modules in an extension graph (relative imports and package +// `imports` aliases such as `#src/*`). Bare third-party dependencies remain +// native Bun resolutions. +const EXTENSION_GRAPH_SPECIFIER_REGEX = /(?:from\s+|import\s+|import\s*\(\s*)["']((?:\.\.?\/|#)[^"']+)["']/g; // Extension entry realpaths that already have a load-time rewrite hook // installed. Each `Bun.plugin()` registration is process-global and permanent, @@ -287,10 +504,14 @@ async function collectExtensionModules(entryRealPath: string): Promise { if (hookedExtensionEntries.has(entryRealPath)) { @@ -322,9 +544,8 @@ async function ensureExtensionGraphHook(entryRealPath: string): Promise { name: `omp:legacy-pi-ext:${Bun.hash(entryRealPath).toString(36)}`, setup(build) { build.onLoad({ filter, namespace: "file" }, async args => { - // Re-read on every load so a `?mtime` reload picks up edited source. const raw = await Bun.file(args.path).text(); - return { contents: rewriteLegacyExtensionSource(raw), loader: getLoader(args.path) }; + return { contents: await rewriteLegacyExtensionSource(raw, args.path), loader: getLoader(args.path) }; }); }, }); @@ -337,9 +558,8 @@ async function ensureExtensionGraphHook(entryRealPath: string): Promise { * and `__dirname`-relative `readFileSync` asset loads (HTML/CSS bundled next to * the entry) resolve exactly as they do under the original Pi runtime — no * temp-directory mirroring and no asset copying. An `onLoad` hook scoped to the - * entry's relative-import graph rewrites only the legacy `@(scope)/pi-*` and - * `@sinclair/typebox` imports in the extension's own source; everything else - * resolves natively. + * entry's source graph rewrites only host-resolved compatibility imports in the + * extension's own source; everything else resolves natively. */ export async function loadLegacyPiModule(resolvedPath: string): Promise { // Bun reports the realpath of a loaded module to `onLoad` and exposes it as diff --git a/packages/coding-agent/src/hindsight/backend.ts b/packages/coding-agent/src/hindsight/backend.ts index 2d00dee27..f01c92002 100644 --- a/packages/coding-agent/src/hindsight/backend.ts +++ b/packages/coding-agent/src/hindsight/backend.ts @@ -9,10 +9,10 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import { logger } from "@oh-my-pi/pi-utils"; -import type { Settings } from "../config/settings"; +import { onHindsightScopeChanged, type Settings } from "../config/settings"; import type { MemoryBackend, MemoryBackendStartOptions } from "../memory-backend/types"; import type { AgentSession } from "../session/agent-session"; -import { computeBankScope } from "./bank"; +import { type BankScope, computeBankScope } from "./bank"; import { createHindsightClient } from "./client"; import { isHindsightConfigured, loadHindsightConfig } from "./config"; import type { HindsightMessage } from "./content"; @@ -60,12 +60,16 @@ export const hindsightBackend: MemoryBackend = { recallTagsMatch: parent.recallTagsMatch, config: parent.config, session, - missionsSet: parent.missionsSet, + banksSet: parent.banksSet, lastRetainedTurn: 0, hasRecalledForFirstTurn: true, aliasOf: parent, }), ); + // Aliases don't run auto-recall/auto-retain, so any pending retain + // queue belongs to the previous alias and is safe to drop after a + // best-effort flush (`flushRetainQueue` is no-op when empty). + await previous?.flushRetainQueue(); previous?.dispose(); return; } @@ -76,38 +80,7 @@ export const hindsightBackend: MemoryBackend = { return; } - const client = createHindsightClient(config); - const scope = computeBankScope(config, session.sessionManager.getCwd()); - - const state = new HindsightSessionState({ - sessionId, - client, - bankId: scope.bankId, - retainTags: scope.retainTags, - recallTags: scope.recallTags, - recallTagsMatch: scope.recallTagsMatch, - config, - session, - missionsSet: new Set(), - lastRetainedTurn: 0, - hasRecalledForFirstTurn: false, - }); - - // Cleanup any stale state for this session (defensive — prevents leaks - // when a session is reused without going through dispose). - const previous = session.setHindsightSessionState(state); - previous?.dispose(); - state.attachSessionListeners(); - - // Kick off mental-model bootstrap. Resolves asynchronously; the first - // turn races and is covered in `beforeAgentStartPrompt` via - // `mentalModelsLoadPromise`. Subsequent turns see the populated cache - // because `runMentalModelLoad` calls `refreshBaseSystemPrompt`. - if (config.mentalModelsEnabled) { - state.mentalModelsLoadPromise = state.runMentalModelLoad(scope).catch(err => { - logger.debug("Hindsight: mental-model bootstrap failed", { bankId: state.bankId, error: String(err) }); - }); - } + await installPrimaryState(session, settings, new Set()); }, async buildDeveloperInstructions(_agentDir, settings, session): Promise { @@ -173,6 +146,182 @@ export const hindsightBackend: MemoryBackend = { return await state.recallForCompaction(flat); }, }; +interface PrimaryRebuildTask { + pending: boolean; +} + +const primaryRebuildTasks = new WeakMap(); + +/** + * Coalesce and serialize live scope rebuilds for one session. Cwd reloads fire + * all settings hooks synchronously; running every callback immediately would + * let multiple rebuilds capture the same old state and leak the fresh states + * installed by earlier continuations. + */ +function schedulePrimaryStateRebuild(session: AgentSession): void { + const task = primaryRebuildTasks.get(session); + if (task) { + task.pending = true; + return; + } + + const nextTask: PrimaryRebuildTask = { pending: true }; + primaryRebuildTasks.set(session, nextTask); + void Promise.resolve() + .then(async () => { + while (nextTask.pending) { + nextTask.pending = false; + try { + await rebuildPrimaryStateOnScopeChange(session); + } catch (err) { + logger.warn("Hindsight: scope rebuild failed", { error: String(err) }); + } + } + }) + .finally(() => { + if (primaryRebuildTasks.get(session) === nextTask) { + primaryRebuildTasks.delete(session); + } + }); +} + +/** + * Build (or rebuild) the primary `HindsightSessionState` for `session` from + * the current settings and install it. Disposes any previous primary state + * after flushing its retain queue so in-flight tool-initiated retains land in + * the bank that was selected when they were enqueued, not in the new bank. + * + * The created state takes ownership of the `onHindsightScopeChanged` + * subscription so subsequent `hindsight.bankId` / `bankIdPrefix` / `scoping` + * edits trigger another rebuild from the same wiring. + */ +async function installPrimaryState( + session: AgentSession, + settings: Settings, + banksSet: Set, +): Promise { + const sessionId = session.sessionId; + if (!sessionId) return undefined; + + const config = loadHindsightConfig(settings); + if (!isHindsightConfigured(config)) return undefined; + + const client = createHindsightClient(config); + const scope = computeBankScope(config, session.sessionManager.getCwd()); + + // Cleanup any stale state for this session (defensive — prevents leaks + // when a session is reused without going through dispose). Flush the + // previous state's retain queue BEFORE clearing it, otherwise + // `HindsightRetainQueue.#doFlush` sees `session.getHindsightSessionState() + // !== state` and drops the batch. Re-read after the await so a concurrent + // owner cannot leave the actual current state undisposed. + let previous = session.getHindsightSessionState(); + if (previous) { + await previous.flushRetainQueue(); + } + const latest = session.getHindsightSessionState(); + if (latest && latest !== previous) { + previous?.dispose(); + previous = latest; + await previous.flushRetainQueue(); + } + + const state = new HindsightSessionState({ + sessionId, + client, + bankId: scope.bankId, + retainTags: scope.retainTags, + recallTags: scope.recallTags, + recallTagsMatch: scope.recallTagsMatch, + config, + session, + banksSet, + lastRetainedTurn: 0, + hasRecalledForFirstTurn: false, + }); + + // Subscribe BEFORE installing: if the operator manages to flip another + // setting between install and subscribe, we'd miss the edge. + state.unsubscribeScope = onHindsightScopeChanged(() => { + schedulePrimaryStateRebuild(session); + }); + + const displaced = session.setHindsightSessionState(state); + if (displaced && displaced !== previous) { + await displaced.flushRetainQueue(); + displaced.dispose(); + } + previous?.dispose(); + state.attachSessionListeners(); + + // Kick off mental-model bootstrap. Resolves asynchronously; the first + // turn races and is covered in `beforeAgentStartPrompt` via + // `mentalModelsLoadPromise`. Subsequent turns see the populated cache + // because `runMentalModelLoad` calls `refreshBaseSystemPrompt`. + if (config.mentalModelsEnabled) { + state.mentalModelsLoadPromise = state.runMentalModelLoad(scope).catch(err => { + logger.debug("Hindsight: mental-model bootstrap failed", { bankId: state.bankId, error: String(err) }); + }); + } + + return state; +} + +/** + * `onHindsightScopeChanged` handler: re-evaluate the bank scope from current + * settings and rebuild the primary state when it has actually drifted. No-op + * when the scope is unchanged or the session is no longer hosting a primary + * state (e.g. it was wiped to `undefined`, or this is a subagent alias). + */ +async function rebuildPrimaryStateOnScopeChange(session: AgentSession): Promise { + const current = session.getHindsightSessionState(); + if (!current || current.aliasOf) return; + + const settings = session.settings; + const config = loadHindsightConfig(settings); + if (!isHindsightConfigured(config)) { + // Hindsight effectively unwired mid-session. Flush before clearing so + // queued retains don't get dropped by `HindsightRetainQueue.#doFlush`. + await current.flushRetainQueue(); + const previous = session.setHindsightSessionState(undefined); + previous?.dispose(); + return; + } + + const next = computeBankScope(config, session.sessionManager.getCwd()); + if (bankScopesEqual(next, current)) return; + + // Preserve the banksSet so we don't re-PUT banks we've already confirmed. + await installPrimaryState(session, settings, current.banksSet); +} + +/** Tag-array equality: order matters because we never reorder on the way in. */ +function stringArraysEqual(a: string[] | undefined, b: string[] | undefined): boolean { + if (a === b) return true; + if (!a || !b) return false; + if (a.length !== b.length) return false; + for (let i = 0; i < a.length; i++) { + if (a[i] !== b[i]) return false; + } + return true; +} + +/** + * Structural compare of a freshly resolved `BankScope` against a live state's + * bank routing. Used by the scope-change handler to skip rebuilds that don't + * actually move the bank or its tag filters. + */ +function bankScopesEqual( + scope: BankScope, + state: Pick, +): boolean { + return ( + scope.bankId === state.bankId && + stringArraysEqual(scope.retainTags, state.retainTags) && + stringArraysEqual(scope.recallTags, state.recallTags) && + scope.recallTagsMatch === state.recallTagsMatch + ); +} /** Reduce arbitrary AgentMessages into the Hindsight flat-text shape. */ function flattenMessagesForRecall(messages: AgentMessage[]): HindsightMessage[] { diff --git a/packages/coding-agent/src/hindsight/bank.ts b/packages/coding-agent/src/hindsight/bank.ts index a6f9cc149..32bc28ce4 100644 --- a/packages/coding-agent/src/hindsight/bank.ts +++ b/packages/coding-agent/src/hindsight/bank.ts @@ -1,5 +1,5 @@ /** - * Bank ID derivation, project-tag scoping, and first-use mission setup. + * Bank ID derivation, project-tag scoping, and first-use bank setup. * * Three scoping modes (`HindsightConfig.scoping`): * - `global` — single shared bank, no per-project filter. @@ -11,10 +11,13 @@ * The base bank id is `bankIdPrefix-bankId` (default `omp`). Per-project mode * appends `-`; tagged mode leaves the bank untouched and uses tags. * - * Mission setup is idempotent at module level — a missionsSet keeps track of - * banks we've already POSTed to so each session boundary doesn't fire a fresh - * `createBank` call. Failures are swallowed: missions are an optimisation, not - * a precondition for retain/recall. + * Bank existence is idempotent at module level — a banksSet keeps track of + * banks we've already PUT so each session boundary doesn't fire a fresh + * `createBank` call. The PUT is idempotent server-side, so re-firing on a hot + * path would only burn round-trips. Failures are swallowed: missing the + * mission patch is an optimisation, but the bank ITSELF must exist before + * mental-model bootstrap or the first retain, otherwise the very first POST + * lands against a missing bank. */ import * as path from "node:path"; @@ -93,39 +96,46 @@ export function deriveBankId(config: HindsightConfig, directory: string): string } /** - * Ensure a bank's reflect/retain mission is set, exactly once per process. + * Ensure a bank exists, and patch its reflect/retain mission on first use. * - * Tracked via the supplied set; on overflow we drop the oldest half so the set - * cannot grow unboundedly across long-lived processes. + * Idempotent: skips the PUT when the bank id is already in the supplied set. + * The mission body is optional — when `bankMission` is blank we still PUT to + * make sure the bank itself is created, so mental-model bootstrap and the + * first retain don't land against a non-existent bank. + * + * The set is capped; on overflow we drop the oldest half so it cannot grow + * unboundedly across long-lived processes. */ -export async function ensureBankMission( +export async function ensureBankExists( client: HindsightApi, bankId: string, config: HindsightConfig, - missionsSet: Set, + banksSet: Set, ): Promise { + if (banksSet.has(bankId)) return; + const mission = config.bankMission?.trim(); - if (!mission) return; - if (missionsSet.has(bankId)) return; + const retainMission = config.retainMission?.trim(); try { await client.createBank(bankId, { - reflectMission: mission, - retainMission: config.retainMission?.trim() || undefined, + reflectMission: mission || undefined, + retainMission: retainMission || undefined, }); - missionsSet.add(bankId); - if (missionsSet.size > MISSION_SET_CAP) { - const keys = [...missionsSet].sort(); + banksSet.add(bankId); + if (banksSet.size > MISSION_SET_CAP) { + const keys = [...banksSet].sort(); for (const key of keys.slice(0, keys.length >> 1)) { - missionsSet.delete(key); + banksSet.delete(key); } } if (config.debug) { - logger.debug("Hindsight: set mission for bank", { bankId }); + logger.debug("Hindsight: ensured bank", { bankId, mission: Boolean(mission) }); } } catch (err) { - // Mission set is best-effort; the bank may not exist yet, or the API may - // reject the call. Either way, retain/recall still work, so swallow. - logger.debug("Hindsight: ensureBankMission failed", { bankId, error: String(err) }); + // Bank creation is best-effort; the server may already have it, or the + // API may reject the call. Either way, downstream retain/recall calls + // will surface a clearer error if the bank really is missing. + logger.debug("Hindsight: ensureBankExists failed", { bankId, error: String(err) }); } } diff --git a/packages/coding-agent/src/hindsight/mental-models.ts b/packages/coding-agent/src/hindsight/mental-models.ts index 294fb5fab..210137193 100644 --- a/packages/coding-agent/src/hindsight/mental-models.ts +++ b/packages/coding-agent/src/hindsight/mental-models.ts @@ -112,7 +112,7 @@ function dedupe(items: T[]): T[] { * Idempotently create any seed mental models that don't already exist on the * bank. Best-effort: a list/create failure does not throw — mental models are * an optimization, not a precondition for retain/recall, and we mirror the - * swallow-on-failure pattern used by `ensureBankMission`. + * swallow-on-failure pattern used by `ensureBankExists`. * * Existing models are NEVER modified. See module docstring. */ diff --git a/packages/coding-agent/src/hindsight/state.ts b/packages/coding-agent/src/hindsight/state.ts index 883e5f714..26f3e7d58 100644 --- a/packages/coding-agent/src/hindsight/state.ts +++ b/packages/coding-agent/src/hindsight/state.ts @@ -1,6 +1,6 @@ import { logger } from "@oh-my-pi/pi-utils"; import type { AgentSession } from "../session/agent-session"; -import { type BankScope, ensureBankMission } from "./bank"; +import { type BankScope, ensureBankExists } from "./bank"; import type { HindsightApi, MemoryItemInput } from "./client"; import type { HindsightConfig } from "./config"; import { @@ -45,12 +45,12 @@ export interface HindsightSessionStateOptions { recallTagsMatch?: "any" | "all" | "any_strict" | "all_strict"; config: HindsightConfig; session: AgentSession; - missionsSet: Set; + banksSet: Set; lastRetainedTurn?: number; hasRecalledForFirstTurn?: boolean; /** * When set, this entry is a subagent alias that reuses the parent's bank, - * scope, config, client, and missionsSet. Aliases skip auto-recall and + * scope, config, client, and banksSet. Aliases skip auto-recall and * auto-retain — those run on the parent only — but the recall/retain/reflect * tools resolve via the alias so they persist to the same bank as the parent. */ @@ -148,7 +148,7 @@ export class HindsightRetainQueue { } try { - await ensureBankMission(state.client, state.bankId, state.config, state.missionsSet); + await ensureBankExists(state.client, state.bankId, state.config, state.banksSet); const batch: MemoryItemInput[] = items.map(item => ({ content: item.content, context: item.context ?? state.config.retainContext, @@ -198,7 +198,7 @@ export class HindsightSessionState { recallTagsMatch?: "any" | "all" | "any_strict" | "all_strict"; config: HindsightConfig; session: AgentSession; - missionsSet: Set; + banksSet: Set; lastRetainedTurn: number; hasRecalledForFirstTurn: boolean; lastRecallSnippet?: string; @@ -213,6 +213,12 @@ export class HindsightSessionState { */ mentalModelsLoadPromise?: Promise; unsubscribe?: () => void; + /** + * Releases the `onHindsightScopeChanged` subscription that drives live + * rebuilds when `hindsight.bankId` / `bankIdPrefix` / `scoping` change. + * Only set on primary states; aliases inherit the parent's subscription. + */ + unsubscribeScope?: () => void; /** Alias states delegate persistence config to a primary parent state. */ aliasOf?: HindsightSessionState; readonly retainQueue: HindsightRetainQueue; @@ -226,7 +232,7 @@ export class HindsightSessionState { this.recallTagsMatch = options.recallTagsMatch; this.config = options.config; this.session = options.session; - this.missionsSet = options.missionsSet; + this.banksSet = options.banksSet; this.lastRetainedTurn = options.lastRetainedTurn ?? 0; this.hasRecalledForFirstTurn = options.hasRecalledForFirstTurn ?? false; this.aliasOf = options.aliasOf; @@ -291,7 +297,7 @@ export class HindsightSessionState { const { transcript } = prepareRetentionTranscript(target, true); if (!transcript) return; - await ensureBankMission(this.client, this.bankId, this.config, this.missionsSet); + await ensureBankExists(this.client, this.bankId, this.config, this.banksSet); await this.client.retain(this.bankId, transcript, { documentId, context: this.config.retainContext, @@ -398,6 +404,12 @@ export class HindsightSessionState { async runMentalModelLoad(scope: BankScope): Promise { if (!this.config.mentalModelsEnabled) return; + // Create/ensure the bank BEFORE the first mental-model POST so we don't + // land `createMentalModel` against a bank the server has never seen — + // that surfaces as a FK / 404 on Hindsight's side. `ensureBankExists` + // is idempotent (PUT) and skips after the first call via `banksSet`. + await ensureBankExists(this.client, this.bankId, this.config, this.banksSet); + // Seeding is opt-in (`hindsight.mentalModelAutoSeed`). Default behaviour is // read-only: we surface whatever models the operator has curated on the // bank, but we do NOT POST to create new ones unless they explicitly @@ -456,6 +468,8 @@ export class HindsightSessionState { dispose(): void { this.unsubscribe?.(); this.unsubscribe = undefined; + this.unsubscribeScope?.(); + this.unsubscribeScope = undefined; this.retainQueue.dispose(); } diff --git a/packages/coding-agent/src/internal-urls/omp-protocol.ts b/packages/coding-agent/src/internal-urls/omp-protocol.ts index a36c29edf..ee5e13534 100644 --- a/packages/coding-agent/src/internal-urls/omp-protocol.ts +++ b/packages/coding-agent/src/internal-urls/omp-protocol.ts @@ -64,9 +64,15 @@ export class OmpProtocolHandler implements ProtocolHandler { throw new Error("Path traversal (..) is not allowed in omp:// URLs"); } - const content = EMBEDDED_DOCS[normalized]; + const docPath = + normalized === "docs" ? "" : normalized.startsWith("docs/") ? normalized.slice("docs/".length) : normalized; + if (!docPath) { + return this.#listDocs(url); + } + + const content = EMBEDDED_DOCS[docPath]; if (content === undefined) { - const lookup = normalized.replace(/\.md$/, ""); + const lookup = docPath.replace(/\.md$/, ""); const suggestions = EMBEDDED_DOC_FILENAMES.filter( f => f.includes(lookup) || lookup.includes(f.replace(/\.md$/, "")), ).slice(0, 5); diff --git a/packages/coding-agent/src/mcp/manager.ts b/packages/coding-agent/src/mcp/manager.ts index 8477b02d8..250b43a27 100644 --- a/packages/coding-agent/src/mcp/manager.ts +++ b/packages/coding-agent/src/mcp/manager.ts @@ -6,7 +6,7 @@ */ import * as path from "node:path"; import * as url from "node:url"; -import type { TSchema } from "@oh-my-pi/pi-ai"; +import { isDefinitiveOAuthFailure, type TSchema } from "@oh-my-pi/pi-ai"; import { logger } from "@oh-my-pi/pi-utils"; import type { SourceMeta } from "../capability/types"; import { resolveConfigValue } from "../config/resolve-config-value"; @@ -1184,29 +1184,48 @@ export class MCPManager { await this.#authStorage.set(credentialId, refreshedCredential); credential = refreshedCredential; } catch (refreshError) { - logger.warn("MCP OAuth refresh failed, using existing token", { - credentialId, - error: refreshError, - }); + const errorMsg = refreshError instanceof Error ? refreshError.message : String(refreshError); + if (isDefinitiveOAuthFailure(errorMsg)) { + // `invalid_grant` / `invalid_token` / 401 from the token endpoint means + // the server has retired this credential — keeping the stale access + // token would just re-fail with 401 on every MCP request and leave a + // poisoned row in agent.db that survives restarts. Drop it now so the + // next connect attempt surfaces a clean "needs reauth" failure and + // the user can recover with `/mcp reauth ` (or `/mcp unauth` + // to forget the server entirely). + logger.warn("MCP OAuth refresh failed definitively; cleared credential", { + credentialId, + error: errorMsg, + }); + await this.#authStorage.remove(credentialId); + credential = undefined; + } else { + logger.warn("MCP OAuth refresh failed, using existing token", { + credentialId, + error: refreshError, + }); + } } } - if (resolved.type === "http" || resolved.type === "sse") { - resolved = { - ...resolved, - headers: { - ...resolved.headers, - Authorization: `Bearer ${credential.access}`, - }, - }; - } else { - resolved = { - ...resolved, - env: { - ...resolved.env, - OAUTH_ACCESS_TOKEN: credential.access, - }, - }; + if (credential?.type === "oauth") { + if (resolved.type === "http" || resolved.type === "sse") { + resolved = { + ...resolved, + headers: { + ...resolved.headers, + Authorization: `Bearer ${credential.access}`, + }, + }; + } else { + resolved = { + ...resolved, + env: { + ...resolved.env, + OAUTH_ACCESS_TOKEN: credential.access, + }, + }; + } } } } catch (error) { diff --git a/packages/coding-agent/src/modes/components/transcript-container.ts b/packages/coding-agent/src/modes/components/transcript-container.ts index 64af9ce42..3925a6f27 100644 --- a/packages/coding-agent/src/modes/components/transcript-container.ts +++ b/packages/coding-agent/src/modes/components/transcript-container.ts @@ -1,4 +1,4 @@ -import { type Component, Container, TERMINAL } from "@oh-my-pi/pi-tui"; +import { type Component, Container, type NativeScrollbackLiveRegion, TERMINAL } from "@oh-my-pi/pi-tui"; const kSnapshot = Symbol("transcript.frozenRender"); @@ -34,7 +34,7 @@ interface SnapshotCarrier { * and any drift reconciles safely. On terminals that can rebuild history this * freezing is unnecessary, so it renders every block live for full fidelity. */ -export class TranscriptContainer extends Container { +export class TranscriptContainer extends Container implements NativeScrollbackLiveRegion { // Bumped to invalidate every block's snapshot at once; a snapshot is only // honored when its stored generation still matches. #generation = 0; @@ -43,6 +43,10 @@ export class TranscriptContainer extends Container { // predate content that finalized in the same coalesced frame that appended the // block now below it — so it must recompute once on the live→frozen transition. #prevLiveChild: Component | undefined; + // Local line index where the current bottom-most block begins in the most + // recent render. TUI extends the native-scrollback pinned region from this + // point through the live block and the root chrome rendered below it. + #nativeScrollbackLiveRegionStart: number | undefined; override invalidate(): void { // A theme/global invalidation forces a full recompute on the rebuild that @@ -56,6 +60,10 @@ export class TranscriptContainer extends Container { super.clear(); } + getNativeScrollbackLiveRegionStart(): number | undefined { + return this.#nativeScrollbackLiveRegionStart; + } + /** * Retire all frozen snapshots so the next render reflects each block's current * state. Call at reconciliation checkpoints (prompt submit) where the whole @@ -68,6 +76,7 @@ export class TranscriptContainer extends Container { override render(width: number): string[] { width = Math.max(1, width); + this.#nativeScrollbackLiveRegionStart = undefined; if (!TERMINAL.eagerEraseScrollbackRisk) return super.render(width); const lines: string[] = []; @@ -77,7 +86,9 @@ export class TranscriptContainer extends Container { this.#prevLiveChild = liveChild; for (let i = 0; i < this.children.length; i++) { const child = this.children[i]! as Component & SnapshotCarrier; - if (child !== liveChild) { + if (child === liveChild) { + this.#nativeScrollbackLiveRegionStart = lines.length; + } else { const snapshot = child[kSnapshot]; // Replay the block's last render from while it was live. A stale // generation (post-thaw) or width mismatch (resize in flight, an diff --git a/packages/coding-agent/src/modes/components/tree-selector.ts b/packages/coding-agent/src/modes/components/tree-selector.ts index c2df593aa..c015f8ca4 100644 --- a/packages/coding-agent/src/modes/components/tree-selector.ts +++ b/packages/coding-agent/src/modes/components/tree-selector.ts @@ -441,8 +441,35 @@ class TreeList implements Component { const lines: string[] = []; if (this.#filteredNodes.length === 0) { - lines.push(truncateToWidth(theme.fg("muted", " No entries found"), width)); - lines.push(truncateToWidth(theme.fg("muted", ` (0/0)${this.#getFilterLabel()}`), width)); + // Three empty-state shapes: + // - flatNodes empty → no entries at all (truly fresh session). + // - search query rejects everything → tell the user the search is the cause. + // - filter mode rejects everything → tell the user the filter is the cause and + // how to widen it. Otherwise fresh sessions whose only persisted entries are + // `model_change` + `thinking_level_change` (both hidden by the default filter) + // read as "broken /tree" — see #1909. + if (this.#flatNodes.length === 0) { + lines.push(truncateToWidth(theme.fg("muted", " No entries found"), width)); + lines.push(truncateToWidth(theme.fg("muted", ` (0/0)${this.#getFilterLabel()}`), width)); + } else if (this.#searchQuery.length > 0) { + lines.push(truncateToWidth(theme.fg("muted", ` No entries match search "${this.#searchQuery}"`), width)); + lines.push(truncateToWidth(theme.fg("muted", " Press Backspace to clear the search"), width)); + lines.push( + truncateToWidth(theme.fg("muted", ` (0/${this.#flatNodes.length})${this.#getFilterLabel()}`), width), + ); + } else { + const filterLabel = this.#getFilterLabel().trim() || "[default]"; + lines.push( + truncateToWidth( + theme.fg("muted", ` ${this.#flatNodes.length} entries hidden by the current filter ${filterLabel}`), + width, + ), + ); + lines.push(truncateToWidth(theme.fg("muted", " Press Alt+A to show all, Alt+D for default"), width)); + lines.push( + truncateToWidth(theme.fg("muted", ` (0/${this.#flatNodes.length})${this.#getFilterLabel()}`), width), + ); + } return lines; } diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index e13ef9abf..8cf4086b8 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -1,6 +1,6 @@ import * as fs from "node:fs/promises"; import type { AutocompleteProvider, SlashCommand } from "@oh-my-pi/pi-tui"; -import { $env, sanitizeText } from "@oh-my-pi/pi-utils"; +import { $env, logger, sanitizeText } from "@oh-my-pi/pi-utils"; import { getRoleInfo } from "../../config/model-registry"; import { isSettingsInitialized, settings } from "../../config/settings"; import { renderSegmentTrack } from "../../modes/components/segment-track"; @@ -406,7 +406,13 @@ export class InputController { } } }) - .catch(() => {}); + .catch(err => { + logger.warn("title-generator: uncaught auto-title error", { + sessionId: this.ctx.session.sessionId, + reason: "uncaught-auto-title-error", + error: err instanceof Error ? err.message : String(err), + }); + }); } if (this.ctx.onInputCallback) { diff --git a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts index b854f85cd..7ddc176f4 100644 --- a/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts +++ b/packages/coding-agent/src/modes/setup-wizard/scenes/sign-in.ts @@ -1,6 +1,6 @@ import type { AuthStorage } from "@oh-my-pi/pi-ai"; import type { OAuthProvider } from "@oh-my-pi/pi-ai/utils/oauth/types"; -import { Input, matchesKey, truncateToWidth } from "@oh-my-pi/pi-tui"; +import { Input, matchesKey, wrapTextWithAnsi } from "@oh-my-pi/pi-tui"; import { getAgentDbPath } from "@oh-my-pi/pi-utils"; import { OAuthSelectorComponent } from "../../components/oauth-selector"; import { theme } from "../../theme/theme"; @@ -15,6 +15,10 @@ const CALLBACK_SERVER_PROVIDERS: Partial> = { "google-antigravity": true, }; +function loginUrlLink(url: string): string { + return `\x1b]8;;${url}\x07Open login URL\x1b]8;;\x07`; +} + interface PromptState { message: string; placeholder?: string; @@ -33,6 +37,7 @@ export class SignInTab implements SetupTab { #authStorage: AuthStorage; #selector: OAuthSelectorComponent; #statusLines: string[] = []; + #authUrl: string | undefined; #prompt: PromptState | undefined; #promptResolve: ((value: string) => void) | undefined; #loginAbort: AbortController | undefined; @@ -72,22 +77,31 @@ export class SignInTab implements SetupTab { } render(width: number): string[] { - const lines = [theme.fg("muted", "Pick a provider to sign in — you can connect more than one."), ""]; + const lines: string[] = []; if (this.#loggingInProvider) { - lines.push(theme.bold(`Signing in to ${this.#loggingInProvider}`), ""); + lines.push(theme.bold(`Signing in to ${this.#loggingInProvider}`)); } else { + lines.push(theme.fg("muted", "Pick a provider to sign in — you can connect more than one."), ""); lines.push(...this.#selector.render(width)); } - if (this.#statusLines.length > 0) { - lines.push("", ...this.#statusLines.map(line => truncateToWidth(line, width))); + + const urlLines = this.#authUrl ? wrapTextWithAnsi(theme.fg("dim", this.#authUrl), width) : []; + if (this.#authUrl) { + lines.push(theme.fg("accent", `Browser login: ${loginUrlLink(this.#authUrl)}`), ...urlLines.slice(0, 2)); } if (this.#prompt) { - lines.push("", theme.fg("warning", this.#prompt.message)); + lines.push(theme.fg("warning", this.#prompt.message)); if (this.#prompt.placeholder) { lines.push(theme.fg("dim", this.#prompt.placeholder)); } lines.push(this.#prompt.input.render(width)[0] ?? ""); } + if (urlLines.length > 2) { + lines.push(...urlLines); + } + if (this.#statusLines.length > 0) { + lines.push(...this.#statusLines.flatMap(line => wrapTextWithAnsi(line, width))); + } return lines; } @@ -109,6 +123,7 @@ export class SignInTab implements SetupTab { this.#selector.stopValidation(); this.#loggingInProvider = providerId; this.#statusLines = [theme.fg("dim", "Starting OAuth flow…")]; + this.#authUrl = undefined; this.#loginAbort = new AbortController(); this.host.restoreFocus(); this.host.requestRender(); @@ -116,7 +131,8 @@ export class SignInTab implements SetupTab { await this.#authStorage.login(providerId as OAuthProvider, { signal: this.#loginAbort.signal, onAuth: info => { - this.#statusLines.push(theme.fg("accent", `Open this URL: ${info.url}`)); + this.#authUrl = info.url; + this.#statusLines = []; if (info.instructions) { this.#statusLines.push(theme.fg("warning", info.instructions)); } @@ -140,6 +156,7 @@ export class SignInTab implements SetupTab { theme.fg("success", `${theme.status.success} Signed in to ${providerId}`), theme.fg("dim", `Credentials saved to ${getAgentDbPath()}`), ]; + this.#authUrl = undefined; this.#loggingInProvider = undefined; this.#loginAbort = undefined; this.#selector.stopValidation(); @@ -150,12 +167,14 @@ export class SignInTab implements SetupTab { if (this.#disposed) return; if (this.#loginAbort?.signal.aborted) { this.#statusLines = [theme.fg("dim", "Login cancelled.")]; + this.#authUrl = undefined; } else { const message = error instanceof Error ? error.message : String(error); this.#statusLines = [ theme.fg("error", `Login failed: ${message}`), theme.fg("dim", "Choose another provider or press Esc to continue."), ]; + this.#authUrl = undefined; } this.#loggingInProvider = undefined; this.#loginAbort = undefined; @@ -174,6 +193,7 @@ export class SignInTab implements SetupTab { this.#resolvePrompt(value); }; input.onEscape = () => { + this.#loginAbort?.abort(); this.#resolvePrompt(""); }; this.host.setFocus(input); diff --git a/packages/coding-agent/src/prompts/agents/explore.md b/packages/coding-agent/src/prompts/agents/explore.md index 6ba32f97d..d7ceb117e 100644 --- a/packages/coding-agent/src/prompts/agents/explore.md +++ b/packages/coding-agent/src/prompts/agents/explore.md @@ -4,6 +4,7 @@ description: Fast read-only codebase scout returning compressed context for hand tools: read, search, find, web_search model: pi/smol thinking-level: med +read-summarize: false output: properties: summary: diff --git a/packages/coding-agent/src/prompts/agents/librarian.md b/packages/coding-agent/src/prompts/agents/librarian.md index a805c886c..766aaecfa 100644 --- a/packages/coding-agent/src/prompts/agents/librarian.md +++ b/packages/coding-agent/src/prompts/agents/librarian.md @@ -4,6 +4,7 @@ description: Researches external libraries and APIs by reading source code. Retu tools: read, search, find, bash, lsp, web_search, ast_grep model: pi/smol thinking-level: minimal +read-summarize: false output: properties: answer: diff --git a/packages/coding-agent/src/prompts/dry-balance-bench.md b/packages/coding-agent/src/prompts/dry-balance-bench.md new file mode 100644 index 000000000..c05f5a177 --- /dev/null +++ b/packages/coding-agent/src/prompts/dry-balance-bench.md @@ -0,0 +1,8 @@ +Write a 20-line poem about balancing OAuth accounts across many providers. + +Form: +- Exactly 20 lines, no title, no stanza breaks. +- Each line is terse and image-driven, in the spirit of haiku: 7 words or fewer, no end punctuation. +- Let the imagery carry the theme — tokens, scopes, refresh cycles, expiry, consent, revocation — rather than naming them literally. + +Output only the 20 lines. No preamble, no commentary, no code fences. diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index 0e02b1f9d..7baea380c 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -27,6 +27,7 @@ import { extractRetryHint, getAgentDbPath, getAgentDir, + getAuthBrokerSnapshotCachePath, getProjectDir, logger, postmortem, @@ -101,7 +102,15 @@ import { } from "./secrets"; import { AgentSession } from "./session/agent-session"; import { resolveAuthBrokerConfig } from "./session/auth-broker-config"; -import { AuthBrokerClient, AuthStorage, RemoteAuthCredentialStore } from "./session/auth-storage"; +import { + AuthBrokerClient, + AuthStorage, + DEFAULT_SNAPSHOT_CACHE_TTL_MS, + RemoteAuthCredentialStore, + readAuthBrokerSnapshotCache, + type SnapshotResponse, + writeAuthBrokerSnapshotCache, +} from "./session/auth-storage"; import { type CustomMessage, convertToLlm, wrapSteeringForModel } from "./session/messages"; import { getRestorableSessionModels, SessionManager } from "./session/session-manager"; import { closeAllConnections } from "./ssh/connection-manager"; @@ -418,6 +427,17 @@ function getDefaultAgentDir(): string { return getAgentDir(); } +function resolveSnapshotTtlMs(): number { + const raw = process.env.OMP_AUTH_BROKER_SNAPSHOT_TTL_MS; + if (raw === undefined) return DEFAULT_SNAPSHOT_CACHE_TTL_MS; + const value = raw.trim(); + if (value === "") return DEFAULT_SNAPSHOT_CACHE_TTL_MS; + const ttlMs = Number(value); + if (Number.isFinite(ttlMs) && ttlMs >= 0) return ttlMs; + logger.warn("Invalid OMP_AUTH_BROKER_SNAPSHOT_TTL_MS; using default", { value: raw }); + return DEFAULT_SNAPSHOT_CACHE_TTL_MS; +} + // Discovery Functions /** @@ -435,9 +455,42 @@ export async function discoverAuthStorage(agentDir: string = getDefaultAgentDir( const brokerConfig = await resolveAuthBrokerConfig(); if (brokerConfig) { const client = new AuthBrokerClient({ url: brokerConfig.url, token: brokerConfig.token }); - const initialResult = await client.fetchSnapshot(); - if (initialResult.status !== 200) throw new Error("Auth broker returned no initial snapshot"); - const store = new RemoteAuthCredentialStore({ client, initialSnapshot: initialResult.snapshot }); + const ttlMs = resolveSnapshotTtlMs(); + const cachePath = getAuthBrokerSnapshotCachePath(); + const persist = + ttlMs > 0 + ? (snapshot: SnapshotResponse): void => { + void writeAuthBrokerSnapshotCache({ + path: cachePath, + token: brokerConfig.token, + url: brokerConfig.url, + snapshot, + }).catch(error => { + logger.debug("auth-broker snapshot cache write failed", { error: String(error) }); + }); + } + : undefined; + + let initialSnapshot: SnapshotResponse | undefined; + if (ttlMs > 0) { + initialSnapshot = + (await readAuthBrokerSnapshotCache({ + path: cachePath, + token: brokerConfig.token, + url: brokerConfig.url, + ttlMs, + }).catch(error => { + logger.debug("auth-broker snapshot cache read failed", { error: String(error) }); + return null; + })) ?? undefined; + } + if (!initialSnapshot) { + const initialResult = await client.fetchSnapshot(); + if (initialResult.status !== 200) throw new Error("Auth broker returned no initial snapshot"); + initialSnapshot = initialResult.snapshot; + persist?.(initialSnapshot); + } + const store = new RemoteAuthCredentialStore({ client, initialSnapshot, onSnapshot: persist }); // Refresh + usage hooks live on RemoteAuthCredentialStore; AuthStorage // discovers them automatically when no explicit option overrides them. const storage = new AuthStorage(store, { @@ -1168,12 +1221,16 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} return preview; }; - // Only top-level sessions own an AsyncJobManager. Subagents reach the - // parent's manager via `AsyncJobManager.instance()` (set below), so creating - // a second instance here just to leave it orphaned wastes a constructor and - // risks accidental disposal of the parent's manager on subagent teardown. + // Only the first top-level session in a process owns an AsyncJobManager. + // Subagents inherit the parent's manager via `AsyncJobManager.instance()` + // (set below), and any additional top-level session spun up in-process + // (e.g. the agent-creation architect in `agent-dashboard.ts`) must share + // the live singleton — otherwise its dispose path would clobber the + // owning session's manager and break the `task`/`bash` async paths + // (issue #1923). The `instance()` guard means later sessions also skip + // constructing an orphaned manager that nothing would ever route to. const asyncJobManager = - backgroundJobsEnabled && !options.parentTaskPrefix + backgroundJobsEnabled && !options.parentTaskPrefix && !AsyncJobManager.instance() ? new AsyncJobManager({ maxRunningJobs: asyncMaxJobs, onJobComplete: async (jobId, result, job) => { @@ -1192,6 +1249,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }) : undefined; + const scopedAsyncJobManager = asyncJobManager ?? (options.parentTaskPrefix ? AsyncJobManager.instance() : undefined); + const agentRegistry = options.agentRegistry ?? AgentRegistry.global(); const resolvedAgentId = options.agentId ?? options.parentTaskPrefix ?? MAIN_AGENT_ID; const resolvedAgentDisplayName = @@ -1293,6 +1352,13 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} authStorage, modelRegistry, getTelemetry: () => agent?.telemetry, + // Subagents inherit the singleton (the parent's manager) so their bash/task + // completions still flow into the spawning conversation's yieldQueue. + // Secondary in-process top-level sessions (no parentTaskPrefix, no + // constructed manager because the singleton was already installed) leave + // this undefined so tools and session job snapshots refuse async work + // instead of silently routing into the owning session (issue #1923). + asyncJobManager: scopedAsyncJobManager, }; // Wire process-wide internal URL singletons owned by their real classes. @@ -2049,6 +2115,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // AsyncJobManager on teardown; subagents inherit the parent's and // **MUST NOT** tear it down. ownedAsyncJobManager: asyncJobManager, + asyncJobManager: scopedAsyncJobManager, scopedModels: options.scopedModels, promptTemplates, slashCommands, @@ -2262,6 +2329,12 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} await session.dispose(); } else { if (hasRegistered) agentRegistry.unregister(resolvedAgentId); + if (asyncJobManager) { + if (AsyncJobManager.instance() === asyncJobManager) { + AsyncJobManager.setInstance(undefined); + } + await asyncJobManager.dispose({ timeoutMs: 3_000 }); + } await disposeKernelSessionsByOwner(evalKernelOwnerId); } } catch (cleanupError) { diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index b7760bb5c..84f78cdce 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -15,6 +15,7 @@ import * as crypto from "node:crypto"; import * as fs from "node:fs"; +import * as os from "node:os"; import * as path from "node:path"; import { scheduler } from "node:timers/promises"; import { isPromise } from "node:util/types"; @@ -94,6 +95,7 @@ import { isUnexpectedSocketCloseMessage, logger, prompt, + relativePathWithinRoot, Snowflake, } from "@oh-my-pi/pi-utils"; import { type AsyncJob, type AsyncJobDeliveryState, AsyncJobManager } from "../async"; @@ -366,6 +368,15 @@ export interface AgentSessionConfig { * **MUST NOT** dispose it on their own teardown. */ ownedAsyncJobManager?: AsyncJobManager; + /** + * AsyncJobManager reachable by this session for scoped job actions. + * + * Top-level owners receive their own manager, subagents receive the inherited + * parent manager, and secondary in-process top-level sessions receive + * `undefined` so job snapshots and ACP drains cannot observe the primary's + * state. + */ + asyncJobManager?: AsyncJobManager; /** Agent identity (registry id like "Main" or "Alice") used for IRC routing. */ agentId?: string; /** Shared agent registry (for forwarding IRC observations to the main session UI). */ @@ -890,6 +901,14 @@ export class AgentSession { * this undefined and **MUST NOT** dispose the global instance on teardown. */ readonly #ownedAsyncJobManager: AsyncJobManager | undefined; + /** + * AsyncJobManager scoped to this session for introspection/cancellation. + * + * This differs from `#ownedAsyncJobManager`: subagents can inherit a parent + * manager for their own owner id, while secondary top-level sessions are left + * undefined to avoid reading the primary's jobs. + */ + readonly #asyncJobManager: AsyncJobManager | undefined; #pendingPythonMessages: PythonExecutionMessage[] = []; #activeEvalExecutions = new Set>(); #evalExecutionDisposing = false; @@ -1080,6 +1099,7 @@ export class AgentSession { this.#evalKernelOwnerId = config.evalKernelOwnerId ?? `agent-session:${Snowflake.next()}`; this.#parentEvalSessionId = config.parentEvalSessionId; this.#ownedAsyncJobManager = config.ownedAsyncJobManager; + this.#asyncJobManager = config.asyncJobManager ?? config.ownedAsyncJobManager; this.#scopedModels = config.scopedModels ?? []; if (config.thinkingLevel === AUTO_THINKING) { // `auto` is session-level: keep the flag and show a provisional concrete @@ -1373,7 +1393,7 @@ export class AgentSession { } getAsyncJobSnapshot(options?: { recentLimit?: number }): AsyncJobSnapshot | null { - const manager = AsyncJobManager.instance(); + const manager = this.#asyncJobManager; if (!manager) return null; const ownerFilter = this.#agentId ? { ownerId: this.#agentId } : undefined; const running = manager.getRunningJobs(ownerFilter).map(job => ({ @@ -1398,11 +1418,20 @@ export class AgentSession { * Cancel async jobs registered by *this* agent only. Used by lifecycle * transitions (newSession, switchSession, handoff, dispose) so a subagent * cleans up its own background work without touching its parent's jobs. - * No-op when no manager is installed or this session has no agent id. + * + * Cancellation runs against this session's scoped manager. Subagents have + * unique agent ids and inherit the parent's manager to clean up their own + * jobs. A secondary in-process top-level session gets no scoped manager, + * because it defaults to `MAIN_AGENT_ID`; reaching through the global + * singleton would tear down the owning primary session's bash/task jobs at + * dispose time (issue #1923). + * + * No-op when no manager is reachable or this session has no agent id. */ #cancelOwnAsyncJobs(): void { if (!this.#agentId) return; - AsyncJobManager.instance()?.cancelAll({ ownerId: this.#agentId }); + const manager = this.#asyncJobManager; + manager?.cancelAll({ ownerId: this.#agentId }); } // ========================================================================= @@ -2128,12 +2157,31 @@ export class AgentSession { if (this.#pendingTtsrInjections.length === 0) return undefined; const rules = this.#pendingTtsrInjections; const content = rules - .map(r => prompt.render(ttsrInterruptTemplate, { name: r.name, path: r.path, content: r.content })) + .map(r => + prompt.render(ttsrInterruptTemplate, { + name: r.name, + path: this.#displayRulePath(r.path), + content: r.content, + }), + ) .join("\n\n"); this.#pendingTtsrInjections = []; return { content, rules }; } + /** + * Render a rule's file path for model-facing TTSR injections without leaking + * the absolute home directory: cwd-relative when the rule lives in the + * project, `~`-relative when it lives under home, else the raw path. + */ + #displayRulePath(rulePath: string): string { + const cwdRel = relativePathWithinRoot(this.sessionManager.getCwd(), rulePath); + if (cwdRel) return cwdRel; + const homeRel = relativePathWithinRoot(os.homedir(), rulePath); + if (homeRel) return `~/${homeRel}`; + return rulePath; + } + #addPendingTtsrInjections(rules: Rule[]): void { const seen = new Set(this.#pendingTtsrInjections.map(rule => rule.name)); for (const rule of rules) { @@ -2186,7 +2234,13 @@ export class AgentSession { if (!rules || rules.length === 0) return undefined; this.#perToolTtsrInjections.delete(ctx.toolCall.id); const reminder = rules - .map(r => prompt.render(ttsrToolReminderTemplate, { name: r.name, path: r.path, content: r.content })) + .map(r => + prompt.render(ttsrToolReminderTemplate, { + name: r.name, + path: this.#displayRulePath(r.path), + content: r.content, + }), + ) .join("\n\n"); // The TTSR manager was already claimed at bucket time; only persistence remains. const ruleNames = rules.map(r => r.name.trim()).filter(n => n.length > 0); @@ -2990,8 +3044,13 @@ export class AgentSession { this.#releasePowerAssertion(); await this.sessionManager.close(); this.#closeAllProviderSessions("dispose"); - const hindsightState = this.setHindsightSessionState(undefined); + // Flush the retain queue BEFORE clearing the session's pointer so + // `HindsightRetainQueue.#doFlush` still sees `session.getHindsightSessionState() === state`. + // Reversed, the spliced batch survives just long enough to fail the + // identity check and get dropped with a `session vanished` warning. + const hindsightState = this.getHindsightSessionState(); await hindsightState?.flushRetainQueue(); + this.setHindsightSessionState(undefined); hindsightState?.dispose(); const mnemopiState = setMnemopiSessionState(this, undefined); mnemopiState?.dispose(); @@ -3069,7 +3128,7 @@ export class AgentSession { } async drainAsyncJobDeliveriesForAcp(options?: { timeoutMs?: number }): Promise { - const manager = AsyncJobManager.instance(); + const manager = this.#asyncJobManager; if (!manager) return false; const ownerFilter = this.#agentId ? { ownerId: this.#agentId } : undefined; const before = manager.getDeliveryState(ownerFilter); diff --git a/packages/coding-agent/src/session/auth-storage.ts b/packages/coding-agent/src/session/auth-storage.ts index 33f0d1607..d3486d7e2 100644 --- a/packages/coding-agent/src/session/auth-storage.ts +++ b/packages/coding-agent/src/session/auth-storage.ts @@ -12,12 +12,16 @@ export type { AuthStorageOptions, OAuthCredential, SerializedAuthStorage, + SnapshotResponse, StoredAuthCredential, } from "@oh-my-pi/pi-ai"; export { AuthBrokerClient, AuthStorage, + DEFAULT_SNAPSHOT_CACHE_TTL_MS, REMOTE_REFRESH_SENTINEL, RemoteAuthCredentialStore, + readAuthBrokerSnapshotCache, SqliteAuthCredentialStore, + writeAuthBrokerSnapshotCache, } from "@oh-my-pi/pi-ai"; diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index da5bada9d..94fb9a63b 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -531,7 +531,7 @@ function createMCPProxyTools(mcpManager: MCPManager): CustomTool[] { }); } -function createSubagentSettings(baseSettings: Settings): Settings { +function createSubagentSettings(baseSettings: Settings, overrides?: Partial>): Settings { const snapshot: Partial> = {}; for (const key of Object.keys(SETTINGS_SCHEMA) as SettingPath[]) { snapshot[key] = baseSettings.get(key); @@ -545,6 +545,7 @@ function createSubagentSettings(baseSettings: Settings): Settings { // the parent task approval is the authorization boundary. Use yolo mode // to preserve unattended subagent execution. User `tools.approval` policies still apply. "tools.approvalMode": "yolo", + ...overrides, }); } @@ -619,7 +620,10 @@ export async function runSubprocess(options: ExecutorOptions): Promise { onUpdate?: AgentToolUpdateCallback; startBackgrounded: boolean; }): ManagedBashJobHandle { - const manager = AsyncJobManager.instance(); + const manager = this.session.asyncJobManager; if (!manager) { throw new ToolError("Background job manager unavailable for this session."); } @@ -716,7 +715,7 @@ export class BashTool implements AgentTool { if (timeoutClampNotice) pendingNotices.push(timeoutClampNotice); if (asyncRequested) { - if (!AsyncJobManager.instance()) { + if (!this.session.asyncJobManager) { throw new ToolError("Async job manager unavailable for this session."); } const job = this.#startManagedBashJob({ @@ -737,7 +736,7 @@ export class BashTool implements AgentTool { }); } - const autoBgManager = AsyncJobManager.instance(); + const autoBgManager = this.session.asyncJobManager; if (this.#autoBackgroundEnabled && !pty && autoBgManager) { const autoBackgroundWaitMs = this.#resolveAutoBackgroundWaitMs(timeoutMs); const startBackgrounded = autoBackgroundWaitMs === 0; diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index 650c8250c..240608176 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -2,6 +2,7 @@ import type { InMemorySnapshotStore } from "@oh-my-pi/hashline"; import type { AgentTelemetryConfig, AgentTool } from "@oh-my-pi/pi-agent-core"; import type { ToolChoice } from "@oh-my-pi/pi-ai"; import { logger } from "@oh-my-pi/pi-utils"; +import type { AsyncJobManager } from "../async/job-manager"; import type { PromptTemplate } from "../config/prompt-templates"; import type { Settings } from "../config/settings"; import { EditTool } from "../edit"; @@ -184,6 +185,21 @@ export interface ToolSession { modelRegistry?: import("../config/model-registry").ModelRegistry; /** Agent output manager for unique agent:// IDs across task invocations */ agentOutputManager?: AgentOutputManager; + /** + * Async job manager scoped to this session. + * + * - Top-level session that constructed one: its own manager. + * - Subagent (`parentTaskPrefix` set): the parent's manager, so background + * bash/task work and `onJobComplete` deliveries flow into the conversation + * that spawned it. + * - Secondary in-process top-level session that found a singleton already + * installed (issue #1923): `undefined`. Tools refuse async work rather + * than silently route completions into the owning session's `yieldQueue`. + * + * Tools MUST use this instead of `AsyncJobManager.instance()` so a secondary + * session never borrows the owning session's manager by accident. + */ + asyncJobManager?: AsyncJobManager; /** MCP manager visible to subagents without relying on the process-global singleton. */ mcpManager?: MCPManager; /** Local protocol root to propagate to nested subagents and eval-created agents. */ diff --git a/packages/coding-agent/src/tools/job.ts b/packages/coding-agent/src/tools/job.ts index ba2daeb25..18defb3f7 100644 --- a/packages/coding-agent/src/tools/job.ts +++ b/packages/coding-agent/src/tools/job.ts @@ -3,7 +3,7 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { type AsyncJob, AsyncJobManager, isBackgroundJobSupportEnabled } from "../async"; +import { type AsyncJob, type AsyncJobManager, isBackgroundJobSupportEnabled } from "../async"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import type { Theme } from "../modes/theme/theme"; import jobDescription from "../prompts/tools/job.md" with { type: "text" }; @@ -90,7 +90,7 @@ export class JobTool implements AgentTool { onUpdate?: AgentToolUpdateCallback, _context?: AgentToolContext, ): Promise> { - const manager = AsyncJobManager.instance(); + const manager = this.session.asyncJobManager; if (!manager) { return { content: [{ type: "text", text: "Async execution is disabled; no background jobs are available." }], @@ -254,7 +254,7 @@ export class JobTool implements AgentTool { ): JobSnapshot[] { const now = Date.now(); return jobs.map(j => { - const current = AsyncJobManager.instance()?.getJob(j.id); + const current = this.session.asyncJobManager?.getJob(j.id); const latest = current ?? j; return { id: latest.id, diff --git a/packages/coding-agent/src/tools/memory-reflect.ts b/packages/coding-agent/src/tools/memory-reflect.ts index 19e5fd2d3..8227a3328 100644 --- a/packages/coding-agent/src/tools/memory-reflect.ts +++ b/packages/coding-agent/src/tools/memory-reflect.ts @@ -1,7 +1,7 @@ import type { AgentTool, AgentToolResult } from "@oh-my-pi/pi-agent-core"; import { logger, untilAborted } from "@oh-my-pi/pi-utils"; import * as z from "zod/v4"; -import { ensureBankMission } from "../hindsight/bank"; +import { ensureBankExists } from "../hindsight/bank"; import reflectDescription from "../prompts/tools/reflect.md" with { type: "text" }; import type { ToolSession } from "."; @@ -67,7 +67,7 @@ export class MemoryReflectTool implements AgentTool } try { - await ensureBankMission(state.client, state.bankId, state.config, state.missionsSet); + await ensureBankExists(state.client, state.bankId, state.config, state.banksSet); const response = await state.client.reflect(state.bankId, params.query, { context: params.context, budget: state.config.recallBudget, diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index d36f04fcc..221b10797 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -259,7 +259,7 @@ interface IndexedContentLines { } const INTERNAL_URL_DISPLAY_RE = /^[a-z][a-z0-9+.-]*:\/\//i; -const OMP_ROOT_URL_RE = /^omp:\/\/\/?$/i; +const OMP_ROOT_URL_RE = /^omp:\/\/(?:\/?|docs\/?)$/i; function normalizeSearchLine(line: string): string { return line.endsWith("\r") ? line.slice(0, -1) : line; diff --git a/packages/coding-agent/src/utils/title-generator.ts b/packages/coding-agent/src/utils/title-generator.ts index 6416205fc..84f4eab1e 100644 --- a/packages/coding-agent/src/utils/title-generator.ts +++ b/packages/coding-agent/src/utils/title-generator.ts @@ -149,7 +149,10 @@ export async function generateSessionTitle( // tiny title model can't reliably decline trivial input, so this happens // deterministically before any model is invoked; the caller retries on the // next user message while the session stays unnamed. - if (isLowSignalTitleInput(firstMessage)) return null; + if (isLowSignalTitleInput(firstMessage)) { + logger.debug("title-generator: skipped low-signal input", { sessionId, reason: "low-signal" }); + return null; + } const tinyModel = settings.get("providers.tinyModel"); if (tinyModel === ONLINE_TINY_TITLE_MODEL_KEY) { @@ -159,7 +162,14 @@ export async function generateSessionTitle( const onlineAbortController = new AbortController(); const localTitle = tinyTitleClient.generate(tinyModel, firstMessage).then( title => title || null, - () => null, + err => { + logger.warn("title-generator: local model error", { + sessionId, + model: tinyModel, + error: err instanceof Error ? err.message : String(err), + }); + return null; + }, ); const startOnline = (): Promise => generateTitleOnline( @@ -188,49 +198,48 @@ export async function generateTitleOnline( ): Promise { const model = getTitleModel(registry, settings, currentModel); if (!model) { - logger.debug("title-generator: no title model found"); + logger.warn("title-generator: no title model found", { sessionId, reason: "no-title-model" }); return null; } const userMessage = formatTitleUserMessage(firstMessage); - - const apiKey = await registry.getApiKey(model, sessionId); - if (!apiKey) { - logger.debug("title-generator: no API key for smol model", { - provider: model.provider, - id: model.id, - }); - return null; - } - // Resolve metadata after getApiKey so the session-sticky credential for this - // request is already recorded; metadataResolver can then return the correct - // account_uuid rather than the snapshot-at-call-site value. - const metadata = metadataResolver?.(model.provider); - - // Title generation is a 3-6 word task, but some reasoning backends ignore - // disableReasoning. Keep the normal cheap budget for non-reasoning models - // while reserving enough output room for reasoning models to still emit - // the forced tool call after any unavoidable thinking tokens. - const maxTokens = model.reasoning ? Math.max(TITLE_MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : TITLE_MAX_TOKENS; - const request = { - model: `${model.provider}/${model.id}`, - systemPrompt: TITLE_SYSTEM_PROMPT, - userMessage, - maxTokens, + const modelName = `${model.provider}/${model.id}`; + const modelContext = { + sessionId, + provider: model.provider, + id: model.id, + model: modelName, }; - logger.debug("title-generator: request", request); + logger.debug("title-generator: start", modelContext); try { + const apiKey = await registry.getApiKey(model, sessionId); + if (!apiKey) { + logger.warn("title-generator: no API key", { ...modelContext, reason: "missing-api-key" }); + return null; + } + // Resolve metadata after getApiKey so the session-sticky credential for this + // request is already recorded; metadataResolver can then return the correct + // account_uuid rather than the snapshot-at-call-site value. + const metadata = metadataResolver?.(model.provider); + + // Title generation is a 3-6 word task, but some reasoning backends ignore + // disableReasoning. Keep the normal cheap budget for non-reasoning models + // while reserving enough output room for reasoning models to still emit + // the forced tool call after any unavoidable thinking tokens. + const maxTokens = model.reasoning ? Math.max(TITLE_MAX_TOKENS, REASONING_SAFE_MAX_TOKENS) : TITLE_MAX_TOKENS; + logger.debug("title-generator: request", { ...modelContext, maxTokens }); + const response = await completeSimple( model, { - systemPrompt: [request.systemPrompt], - messages: [{ role: "user", content: request.userMessage, timestamp: Date.now() }], + systemPrompt: [TITLE_SYSTEM_PROMPT], + messages: [{ role: "user", content: userMessage, timestamp: Date.now() }], tools: [setTitleTool], }, { apiKey, - maxTokens: request.maxTokens, + maxTokens, disableReasoning: true, toolChoice: { type: "tool", name: SET_TITLE_TOOL_NAME }, metadata, @@ -239,8 +248,9 @@ export async function generateTitleOnline( ); if (response.stopReason === "error") { - logger.debug("title-generator: response error", { - model: request.model, + logger.warn("title-generator: response error", { + ...modelContext, + reason: "provider-response-error", stopReason: response.stopReason, errorMessage: response.errorMessage, }); @@ -249,8 +259,18 @@ export async function generateTitleOnline( const title = normalizeGeneratedTitle(extractGeneratedTitle(response.content)); - logger.debug("title-generator: response", { - model: request.model, + if (!title) { + logger.debug("title-generator: no title returned", { + ...modelContext, + reason: "model-returned-none", + usage: response.usage, + stopReason: response.stopReason, + }); + return null; + } + + logger.debug("title-generator: success", { + ...modelContext, title, usage: response.usage, stopReason: response.stopReason, @@ -258,8 +278,9 @@ export async function generateTitleOnline( return title; } catch (err) { - logger.debug("title-generator: error", { - model: request.model, + logger.warn("title-generator: error", { + ...modelContext, + reason: "exception", error: err instanceof Error ? err.message : String(err), }); return null; diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index d7bb707e5..34710c476 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -548,6 +548,7 @@ describe("AgentSession concurrent prompt guard", () => { settings, modelRegistry, agentId: "acp-session-b", + asyncJobManager, }); session = new AgentSession({ agent: agentA, @@ -732,6 +733,70 @@ describe("AgentSession TTSR resume gate", () => { expect(session.isStreaming).toBe(false); }); + it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; + let streamCallCount = 0; + + const sessionManager = SessionManager.inMemory(); + const cwd = sessionManager.getCwd(); + const ruleAbsPath = path.join(cwd, ".omp", "rules", "no-unwrap.md"); + const expectedRel = path.relative(cwd, ruleAbsPath); + const rule: Rule = { + name: "no-unwrap", + path: ruleAbsPath, + content: "Do not use .unwrap()", + condition: ["\\.unwrap\\("], + _source: { provider: "test", providerName: "test", path: ruleAbsPath, level: "project" }, + }; + + const ttsrManager = new TtsrManager({ + enabled: true, + contextMode: "discard", + interruptMode: "always", + repeatMode: "once", + repeatGap: 10, + }); + ttsrManager.addRule(rule); + + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model, systemPrompt: ["Test"], tools: [] }, + streamFn: (_model, _context, options) => { + streamCallCount++; + const stream = new AssistantMessageEventStream(); + if (streamCallCount === 1) { + pushAbortableTtsrStream(stream, options?.signal); + } else { + pushContinuationStream(stream, () => {}); + } + return stream; + }, + }); + + const settings = Settings.isolated(); + const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-rel.db")); + authStorages.push(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + + session = new AgentSession({ agent, sessionManager, settings, modelRegistry, ttsrManager }); + + await session.prompt("Write some Rust code"); + + const injection = sessionManager + .getEntries() + .find(e => e.type === "custom_message" && e.customType === "ttsr-injection"); + expect(injection?.type).toBe("custom_message"); + const content = injection?.type === "custom_message" ? injection.content : undefined; + expect(typeof content).toBe("string"); + const text = content as string; + // The rendered interrupt the model receives references the rule by a + // project-relative path, never the absolute home path. + expect(text).toContain('reason="rule_violation"'); + expect(text).toContain(`path="${expectedRel}"`); + expect(text).not.toContain(ruleAbsPath); + }); + it("prompt() blocks until TTSR deferred continuation completes", async () => { const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; let streamCallCount = 0; diff --git a/packages/coding-agent/test/async-yield-queue.test.ts b/packages/coding-agent/test/async-yield-queue.test.ts index 07984a619..e4ad7a146 100644 --- a/packages/coding-agent/test/async-yield-queue.test.ts +++ b/packages/coding-agent/test/async-yield-queue.test.ts @@ -47,7 +47,7 @@ function asyncDetails(message: AgentMessage): AsyncDetails { return (message as CustomMessage).details ?? { jobs: [] }; } -function createToolSession(): ToolSession { +function createToolSession(asyncJobManager?: AsyncJobManager): ToolSession { return { cwd: process.cwd(), hasUI: false, @@ -57,6 +57,7 @@ function createToolSession(): ToolSession { getSessionFile: () => null, getSessionSpawns: () => null, getAgentId: () => null, + asyncJobManager, } as unknown as ToolSession; } @@ -130,7 +131,7 @@ describe("async result yield queue delivery", () => { await harness.manager.waitForAll(); await waitUntil(() => harness.queue.has("async-result"), "Timed out waiting for staged async result"); - const tool = new JobTool(createToolSession()); + const tool = new JobTool(createToolSession(harness.manager)); const result = await tool.execute("tool-call", { poll: [jobId] }); expect(result.details?.jobs.find(job => job.id === jobId)?.status).toBe("completed"); diff --git a/packages/coding-agent/test/auth-broker-snapshot-cache.test.ts b/packages/coding-agent/test/auth-broker-snapshot-cache.test.ts new file mode 100644 index 000000000..7dba4027c --- /dev/null +++ b/packages/coding-agent/test/auth-broker-snapshot-cache.test.ts @@ -0,0 +1,140 @@ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { type AuthBrokerServerHandle, AuthStorage, SqliteAuthCredentialStore, startAuthBroker } from "@oh-my-pi/pi-ai"; +import { discoverAuthStorage } from "../src/sdk"; +import { + readAuthBrokerSnapshotCache, + type SnapshotResponse, + writeAuthBrokerSnapshotCache, +} from "../src/session/auth-storage"; + +const ENV_KEYS = [ + "OMP_AUTH_BROKER_URL", + "OMP_AUTH_BROKER_TOKEN", + "OMP_AUTH_BROKER_SNAPSHOT_CACHE", + "OMP_AUTH_BROKER_SNAPSHOT_TTL_MS", +] as const; +const PROVIDER = "unit-auth-broker-cache"; +const TOKEN = "coding-agent-cache-token"; + +const savedEnv: Partial> = {}; + +function makeSnapshot(urlTime: number): SnapshotResponse { + return { + generation: 11, + generatedAt: urlTime, + serverNowMs: urlTime, + refresher: { + enabled: false, + intervalMs: 60_000, + skewMs: 300_000, + nextSweepInMs: Number.MAX_SAFE_INTEGER, + }, + credentials: [ + { + id: 1, + provider: PROVIDER, + credential: { type: "api_key", key: "cached-api-key" }, + identityKey: null, + rotatesInMs: null, + }, + ], + }; +} + +async function waitUntil(predicate: () => boolean | Promise, timeoutMs = 2_000): Promise { + const deadline = Date.now() + timeoutMs; + while (Date.now() < deadline) { + if (await predicate()) return; + await Bun.sleep(10); + } + if (!(await predicate())) throw new Error("waitUntil timeout"); +} + +describe("discoverAuthStorage auth-broker snapshot cache", () => { + let tempDir = ""; + + beforeEach(async () => { + for (const key of ENV_KEYS) savedEnv[key] = process.env[key]; + tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "coding-agent-auth-broker-cache-")); + }); + + afterEach(async () => { + for (const key of ENV_KEYS) { + if (savedEnv[key] === undefined) delete process.env[key]; + else process.env[key] = savedEnv[key]; + } + await fs.rm(tempDir, { recursive: true, force: true }); + }); + + test("boots from a fresh encrypted cache when the broker is down", async () => { + const cachePath = path.join(tempDir, "snapshot.enc"); + const downUrl = "http://127.0.0.1:1"; + process.env.OMP_AUTH_BROKER_URL = downUrl; + process.env.OMP_AUTH_BROKER_TOKEN = TOKEN; + process.env.OMP_AUTH_BROKER_SNAPSHOT_CACHE = cachePath; + process.env.OMP_AUTH_BROKER_SNAPSHOT_TTL_MS = "3600000"; + await writeAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: downUrl, + snapshot: makeSnapshot(Date.now()), + }); + + const storage = await discoverAuthStorage(tempDir); + try { + expect(await storage.getApiKey(PROVIDER)).toBe("cached-api-key"); + } finally { + storage.close(); + } + }); + + test("seeds the encrypted cache after an initial broker fetch", async () => { + const cachePath = path.join(tempDir, "snapshot.enc"); + const brokerStore = await SqliteAuthCredentialStore.open(path.join(tempDir, "broker.db")); + brokerStore.saveApiKey(PROVIDER, "broker-api-key"); + const brokerStorage = new AuthStorage(brokerStore); + await brokerStorage.reload(); + let handle: AuthBrokerServerHandle | undefined; + let storage: AuthStorage | undefined; + try { + handle = startAuthBroker({ + storage: brokerStorage, + bind: "127.0.0.1:0", + bearerTokens: [TOKEN], + disableRefresher: true, + }); + process.env.OMP_AUTH_BROKER_URL = handle.url; + process.env.OMP_AUTH_BROKER_TOKEN = TOKEN; + process.env.OMP_AUTH_BROKER_SNAPSHOT_CACHE = cachePath; + process.env.OMP_AUTH_BROKER_SNAPSHOT_TTL_MS = "3600000"; + + storage = await discoverAuthStorage(tempDir); + expect(await storage.getApiKey(PROVIDER)).toBe("broker-api-key"); + await waitUntil(async () => { + const cached = await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: handle!.url, + ttlMs: 3_600_000, + }); + return cached?.credentials.some(entry => entry.provider === PROVIDER) ?? false; + }); + const cached = await readAuthBrokerSnapshotCache({ + path: cachePath, + token: TOKEN, + url: handle.url, + ttlMs: 3_600_000, + }); + const entry = cached?.credentials.find(candidate => candidate.provider === PROVIDER); + expect(entry?.credential).toEqual({ type: "api_key", key: "broker-api-key" }); + } finally { + storage?.close(); + await handle?.close(); + brokerStorage.close(); + brokerStore.close(); + } + }); +}); diff --git a/packages/coding-agent/test/discovery/agent-fields.test.ts b/packages/coding-agent/test/discovery/agent-fields.test.ts index 8ff6b3b7f..56a3e8974 100644 --- a/packages/coding-agent/test/discovery/agent-fields.test.ts +++ b/packages/coding-agent/test/discovery/agent-fields.test.ts @@ -109,4 +109,27 @@ describe("parseAgentFields", () => { expect(fields).toBeDefined(); expect(fields?.autoloadSkills).toBeUndefined(); }); + + test("parses readSummarize from boolean frontmatter", () => { + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: false })?.readSummarize).toBe( + false, + ); + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: true })?.readSummarize).toBe(true); + }); + + test("parses readSummarize from string frontmatter", () => { + expect(parseAgentFields({ name: "explore", description: "desc", readSummarize: "false" })?.readSummarize).toBe( + false, + ); + }); + + test("ignores invalid readSummarize values", () => { + expect( + parseAgentFields({ name: "explore", description: "desc", readSummarize: "nope" })?.readSummarize, + ).toBeUndefined(); + }); + + test("returns undefined readSummarize when field absent", () => { + expect(parseAgentFields({ name: "explore", description: "desc" })?.readSummarize).toBeUndefined(); + }); }); diff --git a/packages/coding-agent/test/discovery/github-skills.test.ts b/packages/coding-agent/test/discovery/github-skills.test.ts new file mode 100644 index 000000000..019d16129 --- /dev/null +++ b/packages/coding-agent/test/discovery/github-skills.test.ts @@ -0,0 +1,70 @@ +/** + * Regression for https://github.com/can1357/oh-my-pi/issues/1906 + * + * The `github` discovery provider previously registered only context-files and + * instructions, leaving `.github/skills//SKILL.md` — the layout GitHub + * documents for agent skills — silently unscanned. This test pins the wiring: + * loading the `skills` capability with the github provider scoped to a cwd + * containing `.github/skills//SKILL.md` must surface the skill. + * + * @see https://docs.github.com/en/copilot/how-tos/copilot-on-github/customize-copilot/customize-cloud-agent/add-skills + */ +import { afterEach, beforeEach, describe, expect, test } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { loadCapability } from "@oh-my-pi/pi-coding-agent/capability"; +import { clearCache } from "@oh-my-pi/pi-coding-agent/capability/fs"; +import type { Skill } from "@oh-my-pi/pi-coding-agent/capability/skill"; +import "@oh-my-pi/pi-coding-agent/capability/skill"; +import "@oh-my-pi/pi-coding-agent/discovery/github"; + +function writeSkill(root: string, name: string, description: string | null): void { + const skillDir = path.join(root, name); + fs.mkdirSync(skillDir, { recursive: true }); + const frontmatter = + description === null ? `---\nname: ${name}\n---\n` : `---\nname: ${name}\ndescription: ${description}\n---\n`; + fs.writeFileSync(path.join(skillDir, "SKILL.md"), `${frontmatter}\n# ${name}\n\nSkill body.\n`); +} + +describe("github discovery — skills", () => { + let tempDir!: string; + + beforeEach(() => { + clearCache(); + tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-github-skills-")); + }); + + afterEach(() => { + clearCache(); + fs.rmSync(tempDir, { recursive: true, force: true }); + }); + + test("discovers .github/skills//SKILL.md via the github provider", async () => { + writeSkill(path.join(tempDir, ".github", "skills"), "demo-skill", "Demo skill for Copilot"); + + const result = await loadCapability("skills", { cwd: tempDir, providers: ["github"] }); + + const found = result.all.find(skill => skill.name === "demo-skill"); + expect(found).toBeDefined(); + expect(found?.path).toBe(path.join(tempDir, ".github", "skills", "demo-skill", "SKILL.md")); + expect(found?.level).toBe("project"); + expect(found?._source.provider).toBe("github"); + expect(result.warnings).toEqual([]); + }); + + test("skips skills missing a description (matches GitHub agent-skills standard)", async () => { + writeSkill(path.join(tempDir, ".github", "skills"), "no-desc", null); + + const result = await loadCapability("skills", { cwd: tempDir, providers: ["github"] }); + + expect(result.all.find(skill => skill.name === "no-desc")).toBeUndefined(); + }); + + test("returns no skills when .github/skills/ is absent", async () => { + const result = await loadCapability("skills", { cwd: tempDir, providers: ["github"] }); + + expect(result.all).toEqual([]); + expect(result.warnings).toEqual([]); + }); +}); diff --git a/packages/coding-agent/test/hindsight-backend.test.ts b/packages/coding-agent/test/hindsight-backend.test.ts index 550306447..e479a39a1 100644 --- a/packages/coding-agent/test/hindsight-backend.test.ts +++ b/packages/coding-agent/test/hindsight-backend.test.ts @@ -19,6 +19,7 @@ interface FakeSessionDeps { sessionId: string | null; cwd?: string; entries?: Array<{ role: "user" | "assistant"; text: string }>; + settings?: Settings; } function makeFakeSession(deps: FakeSessionDeps) { @@ -27,7 +28,7 @@ function makeFakeSession(deps: FakeSessionDeps) { let hindsightState: HindsightSessionState | undefined; const session = { sessionId: deps.sessionId, - settings: Settings.isolated(), + settings: deps.settings ?? Settings.isolated(), sessionManager: { getEntries: () => entries.map((e, i) => ({ @@ -70,6 +71,7 @@ function makeFakeSession(deps: FakeSessionDeps) { emit(event: Parameters[0]) { for (const l of [...listeners]) l(event); }, + listenerCount: () => listeners.size, }; return session; } @@ -207,7 +209,7 @@ describe("hindsightBackend.start", () => { expect(subState?.aliasOf).toBe(parentState); expect(subState?.bankId).toBe(parentState?.bankId); expect(subState?.client).toBe(parentState?.client); - expect(subState?.missionsSet).toBe(parentState?.missionsSet); + expect(subState?.banksSet).toBe(parentState?.banksSet); // Aliases must not subscribe to session events — the parent owns auto-recall/auto-retain. expect(subState?.unsubscribe).toBeUndefined(); // hasRecalledForFirstTurn=true suppresses beforeAgentStartPrompt auto-recall on the sub. @@ -532,3 +534,378 @@ describe("hindsightBackend.clear", () => { expect(deleteSpy).not.toHaveBeenCalled(); }); }); + +describe("hindsightBackend live bank routing", () => { + beforeEach(() => { + resetSettingsForTest(); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + // Regression for issue #1902: changing `hindsight.bankId` during a live + // session used to leave the active `HindsightSessionState` pinned to the + // bank that was selected when the session started, so subsequent retains + // kept landing in the stale bank ("omp") instead of the new one + // ("Minigames"). The bank-routing settings must re-resolve on `set`. + it("rebuilds the primary state when hindsight.bankId changes mid-session", async () => { + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.scoping": "global", + }); + // Seed bankId via `set` (not `isolated` overrides), otherwise the + // follow-up `set` writes to `#global` while `get` keeps returning the + // `#overrides` value — exactly the precedence the live settings UI + // does NOT have, since real config writes land in `#global`. + settings.set("hindsight.bankId", "omp"); + const session = makeFakeSession({ sessionId: "s-rebuild", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + + const initial = session.getHindsightSessionState(); + expect(initial?.bankId).toBe("omp"); + + settings.set("hindsight.bankId", "Minigames"); + // Hook is sync but the rebuild is async; yield once so the handler runs. + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next).toBeDefined(); + expect(next?.bankId).toBe("Minigames"); + // Must be a brand-new state — the old one was disposed. + expect(next).not.toBe(initial); + }); + + // Same regression, exercising the `hindsight.scoping` axis: switching + // scope mode also reshapes the bank id / tag filters and must rebuild. + it("rebuilds the primary state when hindsight.scoping changes mid-session", async () => { + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.scoping", "global"); + const session = makeFakeSession({ sessionId: "s-scoping", cwd: "/work/proj", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + + const initial = session.getHindsightSessionState(); + expect(initial?.bankId).toBe("omp"); + expect(initial?.retainTags).toBeUndefined(); + + settings.set("hindsight.scoping", "per-project"); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next).toBeDefined(); + expect(next?.bankId).toBe("omp-proj"); + expect(next).not.toBe(initial); + }); + + // Same setting written with the same value MUST NOT rebuild — a rebuild + // would reset `lastRetainedTurn` / `hasRecalledForFirstTurn` and force a + // fresh mental-model bootstrap for no observable reason. + it("does not rebuild when the bank-routing setting is rewritten with the same value", async () => { + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.bankId", "omp"); + const session = makeFakeSession({ sessionId: "s-noop", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + + const initial = session.getHindsightSessionState(); + settings.set("hindsight.bankId", "omp"); // unchanged + await Bun.sleep(0); + + expect(session.getHindsightSessionState()).toBe(initial); + }); + + // Same regression flipped: resetting `hindsight.bankId` back to blank / + // default after a non-empty value MUST also rebuild and route subsequent + // retains to the default bank. The Codex-flagged follow-up was that the + // fix had to be bidirectional — set→value AND value→reset. We defend the + // reset direction end-to-end by enqueuing a retain after the reset and + // asserting the batch call hits the recomputed bank, not the previous one. + it("rebuilds when hindsight.bankId is reset to blank after a non-empty value", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.scoping", "per-project"); + settings.set("hindsight.bankId", "Minigames"); + const session = makeFakeSession({ sessionId: "s-reset", cwd: "/work/_NEW_XenGameKit", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + const initial = session.getHindsightSessionState(); + expect(initial?.bankId).toBe("Minigames-_NEW_XenGameKit"); + + // Operator clears the bankId via the TUI — `settings.set(path, "")` is + // the same call shape `#setSettingValue` uses for an empty text input. + settings.set("hindsight.bankId", ""); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next).toBeDefined(); + expect(next).not.toBe(initial); + // With scoping=per-project the base falls back to the default ("omp"), + // so the reset bank id picks up the project suffix from cwd. + expect(next?.bankId).toBe("omp-_NEW_XenGameKit"); + + next!.enqueueRetain("post-reset fact", "reset routing"); + await next!.flushRetainQueue(); + + expect(retainBatchSpy).toHaveBeenCalledTimes(1); + expect(retainBatchSpy.mock.calls[0][0]).toBe("omp-_NEW_XenGameKit"); + }); + + // Companion case: when `hindsight.scoping` is `global`, clearing the + // non-empty bankId should restore the bare `omp` default — the operator's + // stated expectation in the live repro from #1902. + it("routes future retains to the bare omp bank when bankId is cleared in global scoping", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + settings.set("hindsight.scoping", "global"); + settings.set("hindsight.bankId", "Minigames-_NEW_XenGameKit"); + const session = makeFakeSession({ sessionId: "s-reset-global", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + expect(session.getHindsightSessionState()?.bankId).toBe("Minigames-_NEW_XenGameKit"); + + settings.set("hindsight.bankId", ""); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next?.bankId).toBe("omp"); + + next!.enqueueRetain("post-reset global fact"); + await next!.flushRetainQueue(); + + expect(retainBatchSpy).toHaveBeenCalledTimes(1); + expect(retainBatchSpy.mock.calls[0][0]).toBe("omp"); + }); + + it("coalesces synchronous routing hooks so rebuilt states do not leak agent listeners", async () => { + const retainSpy = vi.spyOn(HindsightApi.prototype, "retain").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.retainEveryNTurns": 1, + }); + settings.set("hindsight.bankId", "omp"); + settings.set("hindsight.scoping", "global"); + const entries = [ + { role: "user" as const, text: "remember this routing coalesce fact" }, + { role: "assistant" as const, text: "acknowledged routing coalesce fact" }, + ]; + const session = makeFakeSession({ sessionId: "s-coalesce", cwd: "/work/proj", entries, settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + expect(session.listenerCount()).toBe(1); + + // Mirrors `Settings.#fireAllHooks()` during cwd reload: all three + // Hindsight routing hooks can fire synchronously before the first async + // queue flush continuation resumes. They must collapse into one rebuild. + settings.set("hindsight.bankIdPrefix", "live"); + settings.set("hindsight.bankId", "Minigames"); + settings.set("hindsight.scoping", "per-project"); + await Bun.sleep(0); + + const next = session.getHindsightSessionState(); + expect(next?.bankId).toBe("live-Minigames-proj"); + expect(session.listenerCount()).toBe(1); + + session.emit({ type: "agent_end", messages: [] }); + await Bun.sleep(0); + + expect(retainSpy).toHaveBeenCalledTimes(1); + expect(retainSpy.mock.calls[0][0]).toBe("live-Minigames-proj"); + }); + + // Regression for issue #1902 fix #2: mental-model auto-seed used to POST + // `createMentalModel` against a bank the server never saw, because the + // old `ensureBankMission` skipped creation entirely when `bankMission` + // was blank. The bank MUST be PUT (created) before any mental-model POST. + it("creates the bank before mental-model bootstrap even when bankMission is blank", async () => { + const callOrder: string[] = []; + const createBankSpy = vi.spyOn(HindsightApi.prototype, "createBank").mockImplementation(async () => { + callOrder.push("createBank"); + return {} as never; + }); + const listMentalSpy = vi.spyOn(HindsightApi.prototype, "listMentalModels").mockImplementation(async () => { + callOrder.push("listMentalModels"); + return { items: [] } as never; + }); + const createMentalModelSpy = vi + .spyOn(HindsightApi.prototype, "createMentalModel") + .mockImplementation(async () => { + callOrder.push("createMentalModel"); + return {} as never; + }); + + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.mentalModelsEnabled": true, + "hindsight.mentalModelAutoSeed": true, + "hindsight.bankMission": "", + }); + const session = makeFakeSession({ sessionId: "s-bank-first", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + await session.getHindsightSessionState()?.mentalModelsLoadPromise; + + expect(createBankSpy).toHaveBeenCalled(); + // First call must be `createBank`. Otherwise the mental-model POST + // lands against a never-created bank and the server FK-fails it. + expect(callOrder[0]).toBe("createBank"); + // Mental-model POSTs are allowed but they MUST come after the bank + // is on the server. + if (createMentalModelSpy.mock.calls.length > 0) { + const bankIdx = callOrder.indexOf("createBank"); + const mmIdx = callOrder.indexOf("createMentalModel"); + expect(bankIdx).toBeLessThan(mmIdx); + } + void listMentalSpy; + }); +}); + +describe("hindsightBackend retain queue flush on session teardown", () => { + beforeEach(() => { + resetSettingsForTest(); + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + // Regression for issue #1902 fix #3: `AgentSession.dispose` used to clear + // `#hindsightSessionState` BEFORE flushing the retain queue, so + // `HindsightRetainQueue.#doFlush` saw `session.getHindsightSessionState() + // !== state` and dropped the spliced batch. The fix flips the order: + // flush MUST complete while the session pointer still references the same + // state. We defend the contract end-to-end by enqueuing a tool-initiated + // retain, then calling `flushRetainQueue` in the order + // `AgentSession.dispose` uses (flush → clear → state.dispose). + it("flushes the retain queue to the server before the session pointer clears", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + "hindsight.bankId": "omp", + }); + const session = makeFakeSession({ sessionId: "s-dispose-flush", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + const state = session.getHindsightSessionState(); + expect(state).toBeDefined(); + + state!.enqueueRetain("durable fact", "test context"); + + // AgentSession.dispose order: flush first, THEN clear, THEN dispose. + // Reversed (clear → flush), `#doFlush`'s identity check would fail and + // the batch would be dropped with a `session vanished` warning. + await state!.flushRetainQueue(); + session.setHindsightSessionState(undefined); + state!.dispose(); + + expect(retainBatchSpy).toHaveBeenCalledTimes(1); + const [bankId, items] = retainBatchSpy.mock.calls[0]; + expect(bankId).toBe("omp"); + expect(items).toHaveLength(1); + expect(items[0].content).toBe("durable fact"); + }); + + // Companion contract test: documents the failure mode the dispose-order + // fix prevents. If the session pointer is cleared first, the queue's + // identity guard drops the spliced batch (and `retainBatch` is NOT + // called). This is exactly what was happening before the fix. + it("drops the spliced batch when the session pointer is cleared before flush (anti-regression)", async () => { + const retainBatchSpy = vi.spyOn(HindsightApi.prototype, "retainBatch").mockResolvedValue({} as never); + vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); + + const settings = Settings.isolated({ + "memory.backend": "hindsight", + "hindsight.apiUrl": "http://localhost:8888", + }); + const session = makeFakeSession({ sessionId: "s-buggy-order", settings }); + + await hindsightBackend.start({ + session: session as never, + settings, + modelRegistry: {} as never, + agentDir: "/tmp", + taskDepth: 0, + }); + const state = session.getHindsightSessionState(); + state!.enqueueRetain("dropped fact"); + + // Buggy order — clear THEN flush. The queue's identity check fails. + session.setHindsightSessionState(undefined); + await state!.flushRetainQueue(); + + expect(retainBatchSpy).not.toHaveBeenCalled(); + }); +}); diff --git a/packages/coding-agent/test/hindsight-bank.test.ts b/packages/coding-agent/test/hindsight-bank.test.ts index bbda61602..0ce81b792 100644 --- a/packages/coding-agent/test/hindsight-bank.test.ts +++ b/packages/coding-agent/test/hindsight-bank.test.ts @@ -1,5 +1,5 @@ import { afterEach, beforeEach, describe, expect, it, type Mock, vi } from "bun:test"; -import { computeBankScope, deriveBankId, ensureBankMission } from "@oh-my-pi/pi-coding-agent/hindsight/bank"; +import { computeBankScope, deriveBankId, ensureBankExists } from "@oh-my-pi/pi-coding-agent/hindsight/bank"; import { HindsightApi } from "@oh-my-pi/pi-coding-agent/hindsight/client"; import type { HindsightConfig } from "@oh-my-pi/pi-coding-agent/hindsight/config"; @@ -117,7 +117,7 @@ describe("deriveBankId (legacy wrapper)", () => { }); }); -describe("ensureBankMission", () => { +describe("ensureBankExists", () => { let client: HindsightApi; let createSpy: Mock | undefined; @@ -129,14 +129,14 @@ describe("ensureBankMission", () => { createSpy?.mockRestore(); }); - it("calls createBank exactly once per bank id", async () => { + it("calls createBank exactly once per bank id and forwards the mission body", async () => { createSpy = vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); const seen = new Set(); const config = baseConfig({ bankMission: "remember everything", retainMission: "extract facts" }); - await ensureBankMission(client, "bank-a", config, seen); - await ensureBankMission(client, "bank-a", config, seen); - await ensureBankMission(client, "bank-b", config, seen); + await ensureBankExists(client, "bank-a", config, seen); + await ensureBankExists(client, "bank-a", config, seen); + await ensureBankExists(client, "bank-b", config, seen); expect(createSpy).toHaveBeenCalledTimes(2); expect(createSpy).toHaveBeenCalledWith( @@ -148,13 +148,22 @@ describe("ensureBankMission", () => { expect(seen.has("bank-b")).toBe(true); }); - it("is a no-op when no mission is configured", async () => { + // Regression: mental-model auto-seed used to POST `createMentalModel` against + // a never-created bank when `bankMission` was blank, because the old + // `ensureBankMission` skipped creation entirely without a mission. + it("still PUTs the bank when no mission is configured (so the bank gets created)", async () => { createSpy = vi.spyOn(HindsightApi.prototype, "createBank").mockResolvedValue({} as never); const seen = new Set(); - await ensureBankMission(client, "bank", baseConfig({ bankMission: "" }), seen); - await ensureBankMission(client, "bank", baseConfig({ bankMission: " " }), seen); - expect(createSpy).not.toHaveBeenCalled(); - expect(seen.size).toBe(0); + + await ensureBankExists(client, "bank", baseConfig({ bankMission: "" }), seen); + await ensureBankExists(client, "bank", baseConfig({ bankMission: " " }), seen); + + expect(createSpy).toHaveBeenCalledTimes(1); + expect(createSpy).toHaveBeenCalledWith( + "bank", + expect.objectContaining({ reflectMission: undefined, retainMission: undefined }), + ); + expect(seen.has("bank")).toBe(true); }); it("swallows API failures and does not mark the bank as initialised", async () => { @@ -162,7 +171,7 @@ describe("ensureBankMission", () => { const seen = new Set(); const config = baseConfig({ bankMission: "do the thing" }); - await expect(ensureBankMission(client, "bank-x", config, seen)).resolves.toBeUndefined(); + await expect(ensureBankExists(client, "bank-x", config, seen)).resolves.toBeUndefined(); expect(seen.has("bank-x")).toBe(false); }); }); diff --git a/packages/coding-agent/test/internal-urls/omp-protocol.test.ts b/packages/coding-agent/test/internal-urls/omp-protocol.test.ts new file mode 100644 index 000000000..d779bf393 --- /dev/null +++ b/packages/coding-agent/test/internal-urls/omp-protocol.test.ts @@ -0,0 +1,20 @@ +import { describe, expect, it } from "bun:test"; +import { InternalUrlRouter } from "@oh-my-pi/pi-coding-agent/internal-urls"; + +describe("OmpProtocolHandler", () => { + it("treats omp://docs as the documentation root", async () => { + const resource = await InternalUrlRouter.instance().resolve("omp://docs"); + + expect(resource.content).toContain("# Documentation"); + expect(resource.content).toContain("tools/read.md"); + }); + + it("resolves docs-prefixed documentation paths", async () => { + const router = InternalUrlRouter.instance(); + const direct = await router.resolve("omp://tools/read.md"); + const prefixed = await router.resolve("omp://docs/tools/read.md"); + + expect(prefixed.content).toBe(direct.content); + expect(prefixed.content).toContain("# read"); + }); +}); diff --git a/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts new file mode 100644 index 000000000..f137d9801 --- /dev/null +++ b/packages/coding-agent/test/mcp-manager-oauth-refresh.test.ts @@ -0,0 +1,130 @@ +/** + * Regression tests for MCP OAuth refresh failure handling (issue #1908). + * + * Before the fix, a refresh that came back with `invalid_grant` was logged and + * the stale access token was re-attached as `Authorization: Bearer …` on every + * subsequent MCP request — producing a permanent 401 / reauth loop until the + * user hand-cleared the row in `agent.db`. The fix routes definitive failures + * (`invalid_grant`, `invalid_token`, `revoked`, plain 401/403 not classified as + * transient) through `AuthStorage.remove(credentialId)` and suppresses the + * Bearer injection, so the next request surfaces a clean auth error instead. + */ +import { Database } from "bun:sqlite"; +import { afterEach, beforeEach, describe, expect, test, vi } from "bun:test"; +import { AuthStorage, SqliteAuthCredentialStore } from "@oh-my-pi/pi-ai"; +import { MCPManager } from "../src/mcp/manager"; +import * as oauthFlow from "../src/mcp/oauth-flow"; +import type { MCPServerConfig } from "../src/mcp/types"; + +const CREDENTIAL_ID = "mcp_oauth_test_1908"; +const TOKEN_URL = "https://example.com/oauth/token"; +const STALE_ACCESS = "stale-access-token"; +const STALE_REFRESH = "stale-refresh-token"; + +/** Build a `Headers` snapshot from a prepared MCP config. */ +function getAuthorizationHeader(config: MCPServerConfig): string | undefined { + if (config.type !== "http" && config.type !== "sse") return undefined; + return config.headers?.Authorization; +} + +describe("MCPManager OAuth refresh failure", () => { + let manager: MCPManager; + let authStorage: AuthStorage; + let serverConfig: MCPServerConfig; + + beforeEach(async () => { + const store = new SqliteAuthCredentialStore(new Database(":memory:")); + authStorage = new AuthStorage(store); + await authStorage.reload(); + + // Seed an expired credential so `#resolveAuthConfig` decides to refresh + // (a non-expired credential takes the no-refresh branch and never reaches + // the bug). + await authStorage.set(CREDENTIAL_ID, { + type: "oauth", + access: STALE_ACCESS, + refresh: STALE_REFRESH, + expires: Date.now() - 60_000, + }); + + manager = new MCPManager(process.cwd()); + manager.setAuthStorage(authStorage); + + serverConfig = { + type: "http", + url: "https://logfire.example.com/mcp", + auth: { + type: "oauth", + credentialId: CREDENTIAL_ID, + tokenUrl: TOKEN_URL, + }, + }; + }); + + afterEach(() => { + vi.restoreAllMocks(); + }); + + test("clears the credential and skips Bearer injection on invalid_grant", async () => { + const refreshSpy = vi + .spyOn(oauthFlow, "refreshMCPOAuthToken") + .mockRejectedValue( + new Error( + 'MCP OAuth refresh failed: 400 {"error":"invalid_grant","error_description":"Refresh token has been revoked"}', + ), + ); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(refreshSpy).toHaveBeenCalledTimes(1); + // The poisoned Bearer must not be re-injected — that is the loop the user + // reported (#1908). + expect(getAuthorizationHeader(prepared)).toBeUndefined(); + // The credential row is gone so neither this nor a future session keeps + // shipping the dead refresh token. + expect(authStorage.get(CREDENTIAL_ID)).toBeUndefined(); + }); + + test("clears the credential when the token endpoint replies HTTP 401", async () => { + vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockRejectedValue( + new Error("MCP OAuth refresh failed: 401 Unauthorized"), + ); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(getAuthorizationHeader(prepared)).toBeUndefined(); + expect(authStorage.get(CREDENTIAL_ID)).toBeUndefined(); + }); + + test("keeps the credential and falls back to the existing token on transient failure", async () => { + // Network blip during refresh — the access token may still be live, so + // we preserve the prior behavior of one best-effort attempt with what we + // already have rather than tearing down the credential. + vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockRejectedValue( + new Error("MCP OAuth refresh failed: fetch failed ECONNREFUSED 127.0.0.1:443"), + ); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(getAuthorizationHeader(prepared)).toBe(`Bearer ${STALE_ACCESS}`); + const remaining = authStorage.get(CREDENTIAL_ID); + expect(remaining?.type).toBe("oauth"); + }); + + test("persists rotated credential on successful refresh", async () => { + // Sanity: the happy path still rotates the row and attaches the fresh + // Bearer. Guards against accidentally short-circuiting refresh while + // fixing the failure path. + vi.spyOn(oauthFlow, "refreshMCPOAuthToken").mockResolvedValue({ + access: "fresh-access", + refresh: "fresh-refresh", + expires: Date.now() + 3_600_000, + }); + + const prepared = await manager.prepareConfig(serverConfig); + + expect(getAuthorizationHeader(prepared)).toBe("Bearer fresh-access"); + const remaining = authStorage.get(CREDENTIAL_ID); + expect(remaining).toMatchObject({ type: "oauth", access: "fresh-access", refresh: "fresh-refresh" }); + }); +}); diff --git a/packages/coding-agent/test/memory-tools.test.ts b/packages/coding-agent/test/memory-tools.test.ts index bd9da3a22..9875bf69d 100644 --- a/packages/coding-agent/test/memory-tools.test.ts +++ b/packages/coding-agent/test/memory-tools.test.ts @@ -100,7 +100,7 @@ function registerState(client: HindsightApi, settings?: Settings, opts: Register getHindsightSessionState: () => registeredState, ...opts.sessionOverrides, } as never, - missionsSet: new Set(), + banksSet: new Set(), lastRetainedTurn: 0, hasRecalledForFirstTurn: false, }); diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index a6a9e5858..c7a1817d8 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -1405,6 +1405,51 @@ describe("ModelRegistry", () => { )?.name; expect(restoredName).not.toBe("Custom Name"); }); + + test("modelOverrides can set omitMaxOutputTokens on a built-in model", () => { + writeRawModelsJson({ + openai: { + modelOverrides: { + "gpt-5.4": { + omitMaxOutputTokens: true, + }, + }, + }, + }); + + const registry = new ModelRegistry(authStorage, modelsJsonPath); + const model = registry.find("openai", "gpt-5.4"); + expect(model?.omitMaxOutputTokens).toBe(true); + // maxTokens is still populated locally — only the wire emission is suppressed. + expect(model?.maxTokens).toBeGreaterThan(0); + }); + + test("custom model definitions accept omitMaxOutputTokens", () => { + writeRawModelsJson({ + ollama: { + baseUrl: "http://localhost:11434/v1", + api: "openai-responses", + auth: "none", + models: [ + { + id: "glm-5.1:cloud", + name: "GLM 5.1 Cloud (Ollama)", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 202752, + maxTokens: 202752, + omitMaxOutputTokens: true, + }, + ], + }, + }); + + const registry = new ModelRegistry(authStorage, modelsJsonPath); + const model = registry.find("ollama", "glm-5.1:cloud"); + expect(model?.omitMaxOutputTokens).toBe(true); + expect(model?.maxTokens).toBe(202752); + }); }); describe("github-copilot oauth endpoint alignment", () => { diff --git a/packages/coding-agent/test/modes/components/transcript-container.test.ts b/packages/coding-agent/test/modes/components/transcript-container.test.ts index 5ff34729a..eb7c4435c 100644 --- a/packages/coding-agent/test/modes/components/transcript-container.test.ts +++ b/packages/coding-agent/test/modes/components/transcript-container.test.ts @@ -53,6 +53,22 @@ describe("TranscriptContainer", () => { expect(container.render(40)).toEqual(["a2", "b2"]); }); + it("reports the live block start for native scrollback pinning (ED3-risk)", () => { + riskFlag.eagerEraseScrollbackRisk = true; + const container = new TranscriptContainer(); + const a = new MutableBlock(["a1", "a2"]); + const b = new MutableBlock(["b1"]); + container.addChild(a); + container.addChild(b); + + expect(container.render(40)).toEqual(["a1", "a2", "b1"]); + expect(container.getNativeScrollbackLiveRegionStart()).toBe(2); + + b.set(["b1", "b2"]); + expect(container.render(40)).toEqual(["a1", "a2", "b1", "b2"]); + expect(container.getNativeScrollbackLiveRegionStart()).toBe(2); + }); + it("seals the prior block at its final content when finalize+append coalesce (ED3-risk)", () => { riskFlag.eagerEraseScrollbackRisk = true; const container = new TranscriptContainer(); diff --git a/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts b/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts new file mode 100644 index 000000000..273306856 --- /dev/null +++ b/packages/coding-agent/test/modes/components/tree-selector-empty-state-1909.test.ts @@ -0,0 +1,107 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import { TreeSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tree-selector"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { SessionEntry, SessionTreeNode } from "@oh-my-pi/pi-coding-agent/session/session-manager"; + +beforeAll(async () => { + await initTheme(false, undefined, undefined, "dark", "light"); +}); + +function freshSessionTree(): SessionTreeNode[] { + // Mirror what `sdk.ts` writes on session start: a `model_change` plus a + // `thinking_level_change`. Both are settings/bookkeeping entries that the + // tree selector's default filter hides — a fresh session contains nothing + // else, so the selector sees `flatNodes.length === 2`, `filteredNodes.length === 0`. + const modelChange: SessionEntry = { + type: "model_change", + id: "e1", + parentId: null, + timestamp: new Date().toISOString(), + model: "anthropic/claude-sonnet-4-20250514", + }; + const thinkingChange: SessionEntry = { + type: "thinking_level_change", + id: "e2", + parentId: "e1", + timestamp: new Date().toISOString(), + thinkingLevel: "medium", + }; + return [ + { + entry: modelChange, + children: [{ entry: thinkingChange, children: [] }], + }, + ]; +} + +function userMessageTree(): SessionTreeNode[] { + const entry: SessionEntry = { + type: "message", + id: "e1", + parentId: null, + timestamp: new Date().toISOString(), + message: { role: "user", content: "hello there", timestamp: 1 }, + }; + return [{ entry, children: [] }]; +} + +function renderSelector(selector: TreeSelectorComponent): string { + const lines = (selector as unknown as { render: (w: number) => string[] }).render(120); + return Bun.stripANSI(lines.join("\n")); +} + +describe("issue #1909: tree-selector empty-state messaging", () => { + it("explains that the filter — not missing data — is hiding entries on a fresh session", () => { + const selector = new TreeSelectorComponent( + freshSessionTree(), + "e2", + 60, + () => {}, + () => {}, + ); + const text = renderSelector(selector); + + // Filter-hiding hint and recovery key must both be present so the user knows + // the panel isn't broken and can widen the view without leaving the screen. + expect(text).toContain("hidden by the current filter"); + expect(text).toContain("[default]"); + expect(text.toLowerCase()).toContain("alt+a"); + // Total count must reflect the real flatNodes count, not 0/0 (otherwise the + // "filter hides things" framing is unconvincing). + expect(text).toContain("(0/2)"); + }); + + it("explains a zero-result search as a search problem, not a filter problem", () => { + const selector = new TreeSelectorComponent( + userMessageTree(), + "e1", + 60, + () => {}, + () => {}, + ); + // Type a character that won't match anything in the tree. + selector.handleInput("z"); + const text = renderSelector(selector); + + expect(text).toContain('No entries match search "z"'); + expect(text.toLowerCase()).toContain("backspace"); + // Must NOT misattribute the empty result to the filter mode. + expect(text).not.toContain("hidden by the current filter"); + }); + + it("falls back to the bare 'No entries found' line when the tree is genuinely empty", () => { + const selector = new TreeSelectorComponent( + [], + null, + 60, + () => {}, + () => {}, + ); + const text = renderSelector(selector); + + expect(text).toContain("No entries found"); + expect(text).toContain("(0/0)"); + // Don't tell the user to widen the filter when there's nothing to widen to. + expect(text).not.toContain("hidden by the current filter"); + }); +}); diff --git a/packages/coding-agent/test/plugin-extensions-discovery.test.ts b/packages/coding-agent/test/plugin-extensions-discovery.test.ts index ab34ec339..d429cda0a 100644 --- a/packages/coding-agent/test/plugin-extensions-discovery.test.ts +++ b/packages/coding-agent/test/plugin-extensions-discovery.test.ts @@ -139,6 +139,356 @@ describe("plugin extension discovery", () => { expect(extension?.tools.has("legacy-pi-ext")).toBe(true); }); + it("loads installed legacy Pi plugin extensions that use package imports", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "package-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src", "feature"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "package-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "package-import-plugin", + version: "1.0.0", + imports: { + "#src/*": "./src/*", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#src/feature/command";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "src", "feature", "command.ts"), + [ + 'import { isToolCallEventType as legacyExtensions } from "@earendil-works/pi-coding-agent/extensibility/extensions";', + `import { isToolCallEventType as modernExtensions } from ${JSON.stringify(currentPiExtensionsPath)};`, + "", + 'if (legacyExtensions !== modernExtensions) throw new Error("legacy extension import did not remap");', + 'export const commandName = "package-import-ext";', + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("package-import-ext")).toBe(true); + }); + + it("honors package import conditional object order", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "conditional-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "node"), { recursive: true }); + fs.mkdirSync(path.join(pluginDir, "import"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "conditional-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "conditional-import-plugin", + version: "1.0.0", + imports: { + "#src/*": { + node: "./node/*", + import: "./import/*", + }, + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.mkdirSync(path.dirname(extensionPath), { recursive: true }); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#src/command";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "node", "command.ts"), + 'export const commandName = "node-conditional-ext";', + ); + fs.writeFileSync( + path.join(pluginDir, "import", "command.ts"), + 'export const commandName = "import-conditional-ext";', + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("node-conditional-ext")).toBe(true); + expect(extension?.commands.has("import-conditional-ext")).toBe(false); + }); + + it("leaves package import aliases that point at non-source files for Bun's native loaders", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "json-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "json-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "json-import-plugin", + version: "1.0.0", + imports: { + "#schema": "./src/schema.json", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync(path.join(pluginDir, "src", "schema.json"), JSON.stringify({ commandName: "json-schema-ext" })); + fs.writeFileSync( + extensionPath, + [ + 'import schema from "#schema" with { type: "json" };', + "", + "export default function(pi) {", + "\tpi.registerCommand(schema.commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("json-schema-ext")).toBe(true); + }); + + it("preserves exact null package import exclusions ahead of wildcard fallbacks", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "null-exact-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "null-exact-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "null-exact-import-plugin", + version: "1.0.0", + imports: { + "#src/internal": null, + "#src/*": "./src/*", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync(path.join(pluginDir, "src", "internal.ts"), 'export const commandName = "null-exact-ext";'); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#src/internal";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + const pluginError = result.errors.find(err => err.path === extensionPath); + + expect(pluginError?.error).toContain("#src/internal"); + expect(extension).toBeUndefined(); + }); + + it("preserves active null conditional package import exclusions", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "null-conditional-import-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "null-conditional-import-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "null-conditional-import-plugin", + version: "1.0.0", + imports: { + "#blocked": { + node: null, + default: "./src/blocked.ts", + }, + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync(path.join(pluginDir, "src", "blocked.ts"), 'export const commandName = "null-conditional-ext";'); + fs.writeFileSync( + extensionPath, + [ + 'import { commandName } from "#blocked";', + "", + "export default function(pi) {", + "\tpi.registerCommand(commandName, { handler: async () => {} });", + "}", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + const pluginError = result.errors.find(err => err.path === extensionPath); + + expect(pluginError?.error).toContain("#blocked"); + expect(extension).toBeUndefined(); + }); + + it("rewrites side-effect imports of package-import aliases and legacy Pi scopes", async () => { + const pluginsDir = getPluginsDir(); + const pluginDir = path.join(pluginsDir, "node_modules", "side-effect-plugin"); + const extensionPath = path.join(pluginDir, "src", "index.ts"); + fs.rmSync(path.join(pluginsDir, "node_modules"), { recursive: true, force: true }); + fs.mkdirSync(path.join(pluginDir, "src"), { recursive: true }); + fs.writeFileSync( + path.join(pluginsDir, "package.json"), + JSON.stringify({ + name: "omp-plugins", + private: true, + dependencies: { + "side-effect-plugin": "1.0.0", + }, + }), + ); + fs.writeFileSync( + path.join(pluginDir, "package.json"), + JSON.stringify({ + name: "side-effect-plugin", + version: "1.0.0", + imports: { + "#src/*": "./src/*", + }, + pi: { + extensions: ["./src/index.ts"], + }, + }), + ); + fs.writeFileSync( + extensionPath, + [ + // Side-effect imports — no `from`, no dynamic `import()`. The + // regex matchers must walk and rewrite both shapes so the legacy + // `@earendil-works` import inside `register.ts` resolves to the + // host `@oh-my-pi` package. + 'import "#src/register";', + 'import "./marker";', + "", + "declare global { var __sideEffectMarker: { ok: boolean; runs: number } | undefined; }", + "", + "export default function(pi) {", + '\tif (!globalThis.__sideEffectMarker?.ok) throw new Error("register side-effect did not run");', + '\tpi.registerCommand("side-effect-ext", { handler: async () => {} });', + "}", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "src", "register.ts"), + [ + 'import { isToolCallEventType as legacyExtensions } from "@earendil-works/pi-coding-agent/extensibility/extensions";', + `import { isToolCallEventType as modernExtensions } from ${JSON.stringify(currentPiExtensionsPath)};`, + "", + 'if (legacyExtensions !== modernExtensions) throw new Error("legacy side-effect import did not remap");', + "(globalThis as { __sideEffectMarker?: { ok: boolean; runs: number } }).__sideEffectMarker = { ok: true, runs: 1 };", + ].join("\n"), + ); + fs.writeFileSync( + path.join(pluginDir, "src", "marker.ts"), + [ + "const slot = (globalThis as { __sideEffectMarker?: { ok: boolean; runs: number } }).__sideEffectMarker;", + 'if (!slot) throw new Error("relative side-effect import did not run before sibling");', + "slot.runs += 1;", + ].join("\n"), + ); + + const result = await discoverAndLoadExtensions([], projectDir.path()); + const extension = result.extensions.find(ext => ext.path === extensionPath); + + expect(result.errors).toHaveLength(0); + expect(extension).toBeDefined(); + expect(extension?.commands.has("side-effect-ext")).toBe(true); + expect((globalThis as { __sideEffectMarker?: { ok: boolean; runs: number } }).__sideEffectMarker).toEqual({ + ok: true, + runs: 2, + }); + delete (globalThis as { __sideEffectMarker?: unknown }).__sideEffectMarker; + }); + it("loads installed plugin extensions whose manifest entry points at a directory with index.ts", async () => { const pluginsDir = getPluginsDir(); const pluginDir = path.join(pluginsDir, "node_modules", "dir-entry-plugin"); diff --git a/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts new file mode 100644 index 000000000..0d4a8029d --- /dev/null +++ b/packages/coding-agent/test/sdk-async-job-manager-singleton.test.ts @@ -0,0 +1,172 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { AsyncJobManager } from "@oh-my-pi/pi-coding-agent/async/job-manager"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk"; +import { Snowflake } from "@oh-my-pi/pi-utils"; + +describe("AsyncJobManager singleton across concurrent top-level sessions", () => { + const tempDirs: string[] = []; + + afterEach(async () => { + for (const tempDir of tempDirs.splice(0)) { + fs.rmSync(tempDir, { recursive: true, force: true }); + } + AsyncJobManager.resetForTests(); + }); + + async function spawnTopLevelSession(extraSettings?: Record) { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-sdk-async-singleton-${Snowflake.next()}-`)); + tempDirs.push(tempDir); + const cwd = path.join(tempDir, `project-${Snowflake.next()}`); + const agentDir = path.join(tempDir, "agent"); + fs.mkdirSync(cwd, { recursive: true }); + const { session } = await createAgentSession({ + cwd, + agentDir, + settings: Settings.isolated({ "bash.autoBackground.enabled": true, ...(extraSettings ?? {}) }), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + }); + return session; + } + + it("keeps the primary session's manager installed after a secondary session disposes", async () => { + const primary = await spawnTopLevelSession(); + try { + const primaryManager = AsyncJobManager.instance(); + expect(primaryManager).toBeDefined(); + + const secondary = await spawnTopLevelSession(); + try { + // While the secondary is alive the global instance MUST still point at + // the primary's manager so background tools keep delivering completions + // to the primary session that owns them. + expect(AsyncJobManager.instance()).toBe(primaryManager); + } finally { + await secondary.dispose(); + } + + // After the secondary disposes, the primary's manager MUST still be the + // reachable singleton — otherwise the `task` async path errors with + // "Async execution is enabled but no async job manager is available". + expect(AsyncJobManager.instance()).toBe(primaryManager); + } finally { + await primary.dispose(); + } + + // Once the owning primary session disposes the singleton clears, matching + // the documented single-owner invariant. + expect(AsyncJobManager.instance()).toBeUndefined(); + }); + + it("does not cancel the primary session's running jobs when a secondary session disposes", async () => { + const primary = await spawnTopLevelSession(); + try { + const primaryManager = AsyncJobManager.instance(); + expect(primaryManager).toBeDefined(); + + // Register a long-running job on the primary's manager under the + // MAIN_AGENT_ID owner — the same owner the secondary would inherit by + // default. The secondary's dispose-time `cancelOwnAsyncJobs` must NOT + // cancel this job (issue #1923). + const release = Promise.withResolvers(); + const jobId = primaryManager!.register( + "bash", + "sleep", + async ({ signal }) => { + const aborted = Promise.withResolvers(); + signal.addEventListener("abort", () => aborted.resolve(), { once: true }); + await Promise.race([release.promise, aborted.promise]); + return signal.aborted ? "aborted" : "completed"; + }, + { ownerId: "Main" }, + ); + expect(primary.getAsyncJobSnapshot()?.running.some(job => job.id === jobId)).toBe(true); + + const secondary = await spawnTopLevelSession(); + try { + expect(secondary.getAsyncJobSnapshot()).toBeNull(); + } finally { + await secondary.dispose(); + } + + const job = primaryManager!.getJob(jobId); + expect(job?.status).toBe("running"); + + release.resolve("done"); + await primaryManager!.waitForAll(); + } finally { + await primary.dispose(); + } + }); + + it("refuses async bash from a secondary session instead of routing it to the primary's manager", async () => { + const primary = await spawnTopLevelSession({ "async.enabled": true }); + try { + const primaryManager = AsyncJobManager.instance(); + expect(primaryManager).toBeDefined(); + const primaryJobCountBefore = primaryManager!.getAllJobs().length; + + const secondary = await spawnTopLevelSession({ "async.enabled": true }); + try { + const bashTool = secondary.getToolByName("bash"); + expect(bashTool).toBeDefined(); + await expect(bashTool!.execute("call-1", { command: "echo hi", async: true })).rejects.toThrow( + /Async job manager unavailable/, + ); + } finally { + await secondary.dispose(); + } + + // The secondary's failed async attempt must not have leaked a job into + // the primary's manager. + expect(primaryManager!.getAllJobs().length).toBe(primaryJobCountBefore); + } finally { + await primary.dispose(); + } + }); + + it("clears a manager installed before a top-level session startup failure takes ownership", async () => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-sdk-async-startup-failure-${Snowflake.next()}-`)); + tempDirs.push(tempDir); + const cwd = path.join(tempDir, `project-${Snowflake.next()}`); + const agentDir = path.join(tempDir, "agent"); + fs.mkdirSync(cwd, { recursive: true }); + + await expect( + createAgentSession({ + cwd, + agentDir, + settings: Settings.isolated({ "bash.autoBackground.enabled": true }), + disableExtensionDiscovery: true, + skills: [], + contextFiles: [], + promptTemplates: [], + slashCommands: [], + enableMCP: false, + enableLsp: false, + systemPrompt: () => { + throw new Error("forced startup failure"); + }, + }), + ).rejects.toThrow("forced startup failure"); + + expect(AsyncJobManager.instance()).toBeUndefined(); + + const replacement = await spawnTopLevelSession(); + try { + expect(AsyncJobManager.instance()).toBeDefined(); + expect(replacement.getAsyncJobSnapshot()).not.toBeNull(); + } finally { + await replacement.dispose(); + } + }); +}); diff --git a/packages/coding-agent/test/setup-wizard-sign-in.test.ts b/packages/coding-agent/test/setup-wizard-sign-in.test.ts new file mode 100644 index 000000000..f4c0c4864 --- /dev/null +++ b/packages/coding-agent/test/setup-wizard-sign-in.test.ts @@ -0,0 +1,77 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import type { AuthStorage } from "@oh-my-pi/pi-ai"; +import type { OAuthLoginCallbacks, OAuthProviderId } from "@oh-my-pi/pi-ai/utils/oauth/types"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import { SignInTab } from "../src/modes/setup-wizard/scenes/sign-in"; +import type { SetupSceneHost } from "../src/modes/setup-wizard/scenes/types"; + +beforeAll(async () => { + await initTheme(); +}); + +describe("SignInTab", () => { + it("keeps the OSC8 login link and manual-code prompt above clipped wizard rows", async () => { + const url = `https://example.com/oauth/authorize?client_id=omp&redirect_uri=http%3A%2F%2Flocalhost%3A45454%2Fcallback&state=${"a".repeat(96)}`; + const loginGate = Promise.withResolvers(); + const openedUrls: string[] = []; + + const authStorage = { + has: (_providerId: string) => false, + hasAuth: (_providerId: string) => false, + async login(_provider: OAuthProviderId, ctrl: OAuthLoginCallbacks): Promise { + ctrl.onAuth({ url }); + const prompt = ctrl.onManualCodeInput?.(); + await loginGate.promise; + await prompt; + }, + } as unknown as AuthStorage; + + const host = { + ctx: { + openInBrowser(openedUrl: string): void { + openedUrls.push(openedUrl); + }, + session: { + modelRegistry: { + authStorage, + async refresh(): Promise {}, + }, + }, + }, + requestRender(): void {}, + finish(): void {}, + setFocus(): void {}, + restoreFocus(): void {}, + } as unknown as SetupSceneHost; + + const tab = new SignInTab(host); + try { + for (const char of "anthropic") { + tab.handleInput(char); + } + tab.handleInput("\n"); + + const rendered = tab.render(36); + const compact = rendered.map(line => Bun.stripANSI(line).trim()).join(""); + expect(compact).toContain(url); + expect(compact).not.toContain("…"); + expect(rendered.join("\n")).toContain(`\x1b]8;;${url}\x07Open login URL\x1b]8;;\x07`); + expect(openedUrls).toEqual([url]); + + // On a ~24-row terminal the wizard body ends up ~8 rows; the OSC8 + // link, a plain URL row, and the focused input must survive that clip. + const clippedBody = rendered.slice(0, 8).map(line => Bun.stripANSI(line).trim()); + const plainUrlIndex = clippedBody.findIndex(line => line.startsWith("https://example.com/oauth/authorize?")); + const inputIndex = clippedBody.findIndex(line => line.startsWith(">")); + expect(clippedBody.some(line => line === "Browser login: Open login URL")).toBe(true); + expect(plainUrlIndex).toBeGreaterThanOrEqual(0); + expect(clippedBody).toContain("Paste the authorization code (or full redirect URL):"); + expect(inputIndex).toBeGreaterThanOrEqual(0); + expect(plainUrlIndex).toBeLessThan(inputIndex); + } finally { + tab.dispose(); + loginGate.resolve(); + await loginGate.promise; + } + }); +}); diff --git a/packages/coding-agent/test/title-generator.test.ts b/packages/coding-agent/test/title-generator.test.ts index 3447b3f03..dc1a6b4a4 100644 --- a/packages/coding-agent/test/title-generator.test.ts +++ b/packages/coding-agent/test/title-generator.test.ts @@ -1,6 +1,7 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import * as ai from "@oh-my-pi/pi-ai"; import { type Api, getBundledModel, type Model } from "@oh-my-pi/pi-ai"; +import { logger } from "@oh-my-pi/pi-utils"; import { generateSessionTitle } from "../src/utils/title-generator"; function getModelOrThrow(id: string): Model { @@ -116,6 +117,65 @@ describe("title generator", () => { expect(completeSimpleMock).toHaveBeenCalledTimes(1); }); + it("logs and returns null when title credentials are missing", async () => { + const model = getModelOrThrow("claude-sonnet-4-5"); + const completeSimpleMock = vi.spyOn(ai, "completeSimple"); + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + const title = await generateSessionTitle( + "Investigate the resolver", + { + getAvailable: () => [model], + getApiKey: async () => undefined, + } as never, + createSettings(model), + "session-1", + ); + + expect(title).toBeNull(); + expect(completeSimpleMock).not.toHaveBeenCalled(); + expect(warnSpy).toHaveBeenCalledWith( + "title-generator: no API key", + expect.objectContaining({ + sessionId: "session-1", + provider: model.provider, + id: model.id, + reason: "missing-api-key", + }), + ); + }); + + it("logs and returns null when title credential lookup throws", async () => { + const model = getModelOrThrow("claude-sonnet-4-5"); + const completeSimpleMock = vi.spyOn(ai, "completeSimple"); + const warnSpy = vi.spyOn(logger, "warn").mockImplementation(() => {}); + + const title = await generateSessionTitle( + "Investigate the resolver", + { + getAvailable: () => [model], + getApiKey: async () => { + throw new Error("credential lookup failed"); + }, + } as never, + createSettings(model), + "session-2", + ); + + expect(title).toBeNull(); + expect(completeSimpleMock).not.toHaveBeenCalled(); + expect(warnSpy).toHaveBeenCalledWith( + "title-generator: error", + expect.objectContaining({ + sessionId: "session-2", + provider: model.provider, + id: model.id, + reason: "exception", + error: "credential lookup failed", + }), + ); + }); + it("uses a reasoning-safe output budget for reasoning models", async () => { const model = getModelOrThrow("claude-sonnet-4-5"); const completeSimpleMock = vi.spyOn(ai, "completeSimple").mockResolvedValue({ diff --git a/packages/coding-agent/test/tools/search-internal-urls.test.ts b/packages/coding-agent/test/tools/search-internal-urls.test.ts index a3383faee..52edda3b5 100644 --- a/packages/coding-agent/test/tools/search-internal-urls.test.ts +++ b/packages/coding-agent/test/tools/search-internal-urls.test.ts @@ -183,6 +183,20 @@ describe("SearchTool internal URL resolution", () => { expect(text).toContain("Search file contents with a regex across files"); }); + it("expands omp://docs to grep embedded documentation files", async () => { + const session = createSession(); + const tool = new SearchTool(session); + + const result = await tool.execute("test-call", { + pattern: "Read files, directories, archives", + paths: ["omp://docs"], + }); + + const text = getResultText(result); + expect(text).toContain("# omp://tools/read.md"); + expect(text).toContain("Read files, directories, archives"); + }); + it("throws when internal URL has no sourcePath", async () => { const session = createSession(); const tool = new SearchTool(session); diff --git a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts index 71fed5634..d1de88092 100644 --- a/packages/coding-agent/test/tools/task-agent-capabilities.test.ts +++ b/packages/coding-agent/test/tools/task-agent-capabilities.test.ts @@ -36,6 +36,16 @@ describe("task agent capability descriptions", () => { } }); + it("disables read summarization for explore and librarian, leaves other agents summarizing", () => { + const agents = loadBundledAgents(); + + expect(agentByName(agents, "explore").readSummarize).toBe(false); + expect(agentByName(agents, "librarian").readSummarize).toBe(false); + for (const name of ["task", "quick_task", "plan", "reviewer", "oracle", "designer"]) { + expect(agentByName(agents, name).readSummarize).toBeUndefined(); + } + }); + it("marks read-only agents in the task description and keeps full agents unmarked", async () => { vi.spyOn(discoveryModule, "discoverAgents").mockResolvedValue({ agents: [ diff --git a/packages/coding-agent/test/tools/task-async-fallback.test.ts b/packages/coding-agent/test/tools/task-async-fallback.test.ts new file mode 100644 index 000000000..9bf7e2ee3 --- /dev/null +++ b/packages/coding-agent/test/tools/task-async-fallback.test.ts @@ -0,0 +1,64 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { AsyncJobManager } from "../../src/async"; +import { Settings } from "../../src/config/settings"; +import { TaskTool } from "../../src/task"; +import * as discoveryModule from "../../src/task/discovery"; +import type { TaskParams } from "../../src/task/types"; +import type { ToolSession } from "../../src/tools"; + +function createSession(overrides: Partial> = {}): ToolSession { + return { + cwd: "/tmp", + hasUI: false, + settings: Settings.isolated(overrides), + getSessionFile: () => null, + getSessionSpawns: () => "*", + } as unknown as ToolSession; +} + +function getFirstText(result: { content: Array<{ type: string; text?: string }> }): string { + const content = result.content.find(part => part.type === "text"); + return content?.type === "text" ? (content.text ?? "") : ""; +} + +describe("task.async-fallback", () => { + afterEach(() => { + vi.restoreAllMocks(); + AsyncJobManager.resetForTests(); + }); + + it("falls back to sync execution when async is enabled but no manager is registered", async () => { + // Two-stage spy: the initial discovery during `TaskTool.create` advertises + // `task` so the tool builds; the executor's later call (inside + // `#executeSync`) advertises *nothing*, forcing the unique "Unknown agent" + // message — which is only reachable from the sync codepath. Hitting it + // proves we fell back instead of returning the old hard error. + const discoverSpy = vi.spyOn(discoveryModule, "discoverAgents"); + discoverSpy.mockResolvedValueOnce({ + agents: [ + { + name: "task", + description: "General-purpose task agent", + systemPrompt: "You are a task agent.", + source: "bundled", + }, + ], + projectAgentsDir: null, + }); + discoverSpy.mockResolvedValue({ agents: [], projectAgentsDir: null }); + + AsyncJobManager.resetForTests(); + expect(AsyncJobManager.instance()).toBeUndefined(); + + const tool = await TaskTool.create(createSession({ "async.enabled": true })); + + const result = await tool.execute("tool-1", { + agent: "task", + tasks: [{ id: "One", description: "label", assignment: "Do the thing." }], + } as TaskParams); + + const text = getFirstText(result); + expect(text).toContain('Unknown agent "task"'); + expect(text).not.toContain("no async job manager is available"); + }); +}); diff --git a/packages/hashline/package.json b/packages/hashline/package.json index 31dcc4e9d..dde5a1987 100644 --- a/packages/hashline/package.json +++ b/packages/hashline/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/hashline", - "version": "15.9.1", + "version": "15.9.2", "description": "Hashline: a compact, line-anchored patch language and applier. Pluggable FS/IO so it works over disk, in-memory, or any custom backend.", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/mnemopi/package.json b/packages/mnemopi/package.json index 4de037377..af532fc40 100644 --- a/packages/mnemopi/package.json +++ b/packages/mnemopi/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-mnemopi", - "version": "15.9.1", + "version": "15.9.2", "description": "Local SQLite memory engine for Oh My Pi agents", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 3890932c1..060e22860 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -136,7 +136,7 @@ export declare class Shell { * `packages/natives/native/index.js` (which derives the name from * `package.json#version`). */ -export declare function __piNativesV15_9_1(): void +export declare function __piNativesV15_9_2(): void /** * Apply conservative pre-execution rewrites to a bash command. diff --git a/packages/natives/native/index.js b/packages/natives/native/index.js index fdc0d51b6..4bfa3a826 100644 --- a/packages/natives/native/index.js +++ b/packages/natives/native/index.js @@ -23,7 +23,7 @@ export const PtySession = nativeBindings.PtySession; export const Shell = nativeBindings.Shell; // functions -export const __piNativesV15_9_1 = nativeBindings.__piNativesV15_9_1; +export const __piNativesV15_9_2 = nativeBindings.__piNativesV15_9_2; export const applyBashFixups = nativeBindings.applyBashFixups; export const astEdit = nativeBindings.astEdit; export const astGrep = nativeBindings.astGrep; diff --git a/packages/natives/package.json b/packages/natives/package.json index 33d473f79..b331763c8 100644 --- a/packages/natives/package.json +++ b/packages/natives/package.json @@ -1,6 +1,6 @@ { "name": "@oh-my-pi/pi-natives", - "version": "15.9.1", + "version": "15.9.2", "description": "Native Rust bindings for grep, clipboard, image processing, syntax highlighting, PTY, and shell operations via N-API", "type": "module", "homepage": "https://omp.sh", diff --git a/packages/stats/package.json b/packages/stats/package.json index 0547ef087..e6e296803 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/omp-stats", - "version": "15.9.1", + "version": "15.9.2", "description": "Local observability dashboard for pi AI usage statistics", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/swarm-extension/package.json b/packages/swarm-extension/package.json index fa42f9758..25868ff20 100644 --- a/packages/swarm-extension/package.json +++ b/packages/swarm-extension/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/swarm-extension", - "version": "15.9.1", + "version": "15.9.2", "description": "Swarm orchestration extension for omp", "homepage": "https://omp.sh", "author": "Derek Rynd", diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index 5fb546c22..44f7af3cd 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -2,7 +2,19 @@ ## [Unreleased] +## [15.9.2] - 2026-06-05 + +### Changed + +- Changed foreground-stream rendering on ED3-risk terminals (Ghostty/kitty/Alacritty/VTE/iTerm2 on POSIX) to defer native-scrollback commits for unpinned transient frames: while a turn streams, generic frames repaint only the viewport and suppress `\r\n` scroll growth, so transient output (spinner ticks, partial lines, status rows) never pollutes terminal history. Components that report a `NativeScrollbackLiveRegion` still commit newly sealed prefix rows while keeping the active suffix dirty for checkpoint replay. Native scrollback is reconciled in a single ED3 (`CSI 3 J`) + re-emit at the next checkpoint (prompt submit) or on an explicit user-input/IME opt-in; an erase is never emitted mid-stream under a possibly-scrolled reader. Non-ED3-risk terminals keep their eager live rebuild. ([#1895](https://github.com/can1357/oh-my-pi/pull/1895)) + +### Fixed + +- Fixed ED3-risk foreground streaming dropping sealed transcript rows above the live block until the next prompt-submit checkpoint, which made scrollback beyond the viewport appear duplicated or out of order. The renderer restores native-scrollback live-region pinning so newly sealed rows are appended once while active live rows remain deferred. +- Fixed inline images (added in 15.9) rendering as a wall of empty PUA box glyphs and producing laggy scrolling on Kitty-protocol terminals that do not implement Unicode placeholders — most notably WezTerm (per upstream wezterm/wezterm#986, placeholder support is still unchecked) and the tmux/screen `getFallbackImageProtocol` path that forces Kitty mode even on non-supporting outer terminals (Terminal.app, etc.). `unicodePlaceholders` now defaults on only for `kitty` and `ghostty`; everything else falls back to direct `a=p,i=…,p=…` placement, which those paths already render correctly. `PI_NO_KITTY_PLACEHOLDERS=1` is still honored as a hard opt-out, and a new `PI_KITTY_PLACEHOLDERS=1` opts in on otherwise-unsupported terminals (e.g. a wezterm nightly that has merged placeholder support) ([#1877](https://github.com/can1357/oh-my-pi/issues/1877)). + ## [15.9.1] - 2026-06-04 + ### Fixed - Fixed the OSC 11 appearance poll re-querying every 2s forever on terminals that support Mode 2031 but never change theme, whose repeated OSC 11/DA1 writes cleared the user's active text selection (breaking copy every 2 seconds). The poll now stops as soon as DECRQM confirms Mode 2031 support, since push notifications make polling redundant. diff --git a/packages/tui/package.json b/packages/tui/package.json index bcbb60469..a28a4c50b 100644 --- a/packages/tui/package.json +++ b/packages/tui/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-tui", - "version": "15.9.1", + "version": "15.9.2", "description": "Terminal User Interface library with differential rendering for efficient text-based applications", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/tui/src/kitty-graphics.ts b/packages/tui/src/kitty-graphics.ts index b531be732..8b65bf5d8 100644 --- a/packages/tui/src/kitty-graphics.ts +++ b/packages/tui/src/kitty-graphics.ts @@ -17,7 +17,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; -import { $env, $flag, logger } from "@oh-my-pi/pi-utils"; +import { $env, logger } from "@oh-my-pi/pi-utils"; /** Kitty Unicode placeholder base character (U+10EEEE, Plane 16 PUA). */ export const KITTY_PLACEHOLDER = "\u{10eeee}"; @@ -76,8 +76,36 @@ function transmissionOverride(): KittyTransmissionMedium | "auto" { return "auto"; } +/** + * Whether the detected terminal renders Kitty Unicode placeholders (`U=1` + + * U+10EEEE with row/column diacritics). + * + * Only `kitty` (the protocol's origin) and `ghostty` ship a working + * implementation; WezTerm advertises Kitty graphics but treats placeholder + * cells as literal PUA glyphs (see wezterm/wezterm#986, "placeholder support" + * still unchecked), and the tmux/screen fallback can land on any outer + * terminal. Enabling placeholders on those paths emits a `columns × rows` + * grid of U+10EEEE per image per frame; the cells render as boxed fallback + * glyphs and re-emit on every repaint, which is exactly the + * "stuck/laggy scrolling + ASCII artifact" symptom reported in #1877. + * + * `PI_NO_KITTY_PLACEHOLDERS=1` forces off (e.g. for tmux passthrough to a + * non-supporting outer terminal); `PI_KITTY_PLACEHOLDERS=1` forces on (e.g. + * for a wezterm nightly that has merged placeholder support). + */ +export function detectKittyUnicodePlaceholdersSupport(terminalId: string, env: NodeJS.ProcessEnv = Bun.env): boolean { + const offRaw = env.PI_NO_KITTY_PLACEHOLDERS?.trim().toLowerCase(); + if (offRaw === "1" || offRaw === "true" || offRaw === "on" || offRaw === "yes" || offRaw === "y") return false; + const force = env.PI_KITTY_PLACEHOLDERS?.trim().toLowerCase(); + if (force === "1" || force === "true" || force === "on" || force === "yes" || force === "y") return true; + if (force === "0" || force === "false" || force === "off" || force === "no" || force === "n") return false; + return terminalId === "kitty" || terminalId === "ghostty"; +} + let features: KittyGraphicsFeatures = { - unicodePlaceholders: !$flag("PI_NO_KITTY_PLACEHOLDERS"), + // Off until `terminal-capabilities` seeds it from the detected terminal id — + // the default-on path corrupts wezterm and tmux-passthrough sessions. + unicodePlaceholders: false, // Start direct; a successful probe (or explicit `temp-file` override) promotes. transmissionMedium: transmissionOverride() === "temp-file" ? "temp-file" : "direct", }; diff --git a/packages/tui/src/terminal-capabilities.ts b/packages/tui/src/terminal-capabilities.ts index bc7ef4d19..31fce69e8 100644 --- a/packages/tui/src/terminal-capabilities.ts +++ b/packages/tui/src/terminal-capabilities.ts @@ -1,12 +1,14 @@ import { encodeSixel } from "@oh-my-pi/pi-natives"; import { $env, isBunTestRuntime } from "@oh-my-pi/pi-utils"; import { + detectKittyUnicodePlaceholdersSupport, encodeKittyTempFileTransmit, getKittyGraphics, isPngBase64, KITTY_PLACEHOLDER, kittyPlaceholdersFit, renderKittyPlaceholderLines, + setKittyGraphics, } from "./kitty-graphics"; export enum ImageProtocol { @@ -321,6 +323,12 @@ export const TERMINAL = (() => { return resolved; })(); +// Seed Kitty Unicode placeholder support from the resolved terminal id. Only +// kitty/ghostty are known to honor `U=1` placement; other Kitty-protocol paths +// (wezterm, tmux/screen fallback) treat the placeholder cells as literal PUA +// glyphs, which is the "ASCII artifact + laggy scrolling" reported in #1877. +setKittyGraphics({ unicodePlaceholders: detectKittyUnicodePlaceholdersSupport(TERMINAL.id, Bun.env) }); + type MutableTerminalInfo = { imageProtocol: ImageProtocol | null; deccara: boolean; diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index b94625eda..100ab0a0b 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -117,6 +117,21 @@ export interface Component { invalidate(): void; } +/** + * Optional component seam for native-scrollback pinning. A component that + * renders a stable prefix followed by a live/transient suffix reports the local + * line index where that suffix begins after each render. TUI treats that suffix + * — and every root child rendered below it — as not yet safe to commit to native + * scrollback on ED3-risk terminals whose viewport position is unobservable. + */ +export interface NativeScrollbackLiveRegion { + getNativeScrollbackLiveRegionStart(): number | undefined; +} + +function getNativeScrollbackLiveRegionStart(component: Component): number | undefined { + return (component as Component & Partial).getNativeScrollbackLiveRegionStart?.(); +} + /** * Interface for components that can receive focus and display a cursor. * When focused, the component should emit CURSOR_MARKER at the cursor position @@ -317,6 +332,9 @@ export class Container implements Component { * - `historyRebuild`: a geometry change (terminal resize) left native history * wrapped at the old size — clear viewport and scrollback so it rewraps at the * new geometry. Also flushes deferred content-only rewrites. + * - `liveRegionPinned`: ED3-risk/unknown foreground stream with a reported live + * suffix — optionally append newly sealed rows, then repaint the live tail + * without letting transient rows enter native history. * - `viewportRepaint`: rewrite the visible viewport in place. If `appendFrom` * is set, emit those tail rows as scrollback growth first so streaming * output reaches terminal history before the corrected viewport is drawn. @@ -334,6 +352,7 @@ type RenderIntent = | { kind: "sessionReplace" } | { kind: "historyRebuild" } | { kind: "overlayRebuild" } + | { kind: "liveRegionPinned"; appendFrom: number; appendTo: number } | { kind: "viewportRepaint"; appendFrom?: number } | { kind: "deferredShrink"; paddedLength: number } | { kind: "deferredMutation" } @@ -383,7 +402,18 @@ export class TUI extends Container { // Set after a clear+full replay so the next insert-above-suffix frame does // not scroll replayed live chrome (status/editor) into fresh history. #suppressNextSuffixScroll = false; + #nativeScrollbackLiveRegionStart: number | undefined; #nativeScrollbackDirty = false; + // Highest `#maxLinesRendered` reached during a foreground tool turn while + // intermediate frames were prevented from committing to terminal scrollback. + // Used after the tool finishes to push the settled content into scrollback + // via a non-destructive full paint (no ED 3). Reset to 0 once rows are + // committed (via any `#emitFullPaint`, `#emitDiff`, or `#emitAppendTail` + // path). + #streamingHighWater = 0; + // Tracks whether the previous frame was inside a foreground tool streaming + // turn. Used to reset `#streamingHighWater` on fresh streaming starts. + #previousStreamingActive = false; #fullRedrawCount = 0; // Caps how many inline images render as live graphics; older ones fall back // to text via a purge + full redraw. Cap is configured by the host app. @@ -423,6 +453,25 @@ export class TUI extends Container { this.#showHardwareCursor = showHardwareCursor === undefined ? this.#showHardwareCursor : showHardwareCursor; } + override render(width: number): string[] { + width = Math.max(1, width); + this.#nativeScrollbackLiveRegionStart = undefined; + const lines: string[] = []; + for (const child of this.children) { + const offset = lines.length; + const childLines = child.render(width); + const liveRegionStart = getNativeScrollbackLiveRegionStart(child); + if (liveRegionStart !== undefined) { + const boundedStart = Number.isFinite(liveRegionStart) + ? Math.max(0, Math.min(childLines.length, Math.trunc(liveRegionStart))) + : childLines.length; + this.#nativeScrollbackLiveRegionStart = offset + boundedStart; + } + lines.push(...childLines); + } + return lines; + } + #syncTerminalCursorMode(component: Component | null): void { if (isFocusable(component)) { component.setUseTerminalCursor?.(this.#showHardwareCursor); @@ -1383,11 +1432,12 @@ export class TUI extends Container { (resizeEventOccurred && this.#previousHeight > 0); const eagerEraseScrollbackRisk = process.platform !== "win32" && TERMINAL.eagerEraseScrollbackRisk; const eagerRebuildAllowed = this.#eagerNativeScrollbackRebuild && !eagerEraseScrollbackRisk; - const allowUnknownViewportMutation = this.#allowUnknownViewportMutationOnNextRender || eagerRebuildAllowed; + const explicitViewportMutation = this.#allowUnknownViewportMutationOnNextRender; + const allowUnknownViewportMutation = explicitViewportMutation || eagerRebuildAllowed; this.#allowUnknownViewportMutationOnNextRender = false; // 3. Classify intent. - const intent = this.#planRender( + let intent = this.#planRender( lines, widthChanged, heightChanged, @@ -1396,7 +1446,52 @@ export class TUI extends Container { visibleOverlayComponents.length > 0, overlayVisibilityReduced, allowUnknownViewportMutation, + this.#nativeScrollbackLiveRegionStart, ); + // 3b. Defer scrollback commits during foreground streaming, but only on + // ED3-risk terminals whose committed scrollback cannot be rewritten without + // yanking a scrolled reader. There the eager rebuild is gated off and the + // diff emitter would otherwise `\r\n`-scroll every transient frame (spinner + // ticks, partial output) into native history. Non-ED3-risk terminals keep + // their eager live rebuild, which already commits cleanly. Explicit + // reconciles — the prompt-submit checkpoint (`clearScrollbackOnNextRender`) + // and user-input/IME opt-ins (`explicitViewportMutation`) — are never + // deferred: ED3 is safe there because the keystroke pins the host to the + // bottom. + const streamingWasActive = this.#eagerNativeScrollbackRebuild; + if (streamingWasActive && !this.#previousStreamingActive) { + this.#streamingHighWater = 0; + } + this.#previousStreamingActive = streamingWasActive; + if (streamingWasActive && eagerEraseScrollbackRisk) { + const streamingActive = + this.#eagerNativeScrollbackRebuild && !this.#eagerNativeScrollbackRebuildDisablePending; + const explicitReconcile = explicitViewportMutation || this.#clearScrollbackOnNextRender; + if (!streamingActive) { + // Streaming just ended. Keep native scrollback dirty so the next + // checkpoint reconciles the settled transcript; never erase here. + this.#streamingHighWater = 0; + this.#markNativeScrollbackDirty(); + } else if ( + !explicitReconcile && + (intent.kind === "sessionReplace" || + intent.kind === "historyRebuild" || + intent.kind === "overlayRebuild" || + (intent.kind === "diff" && intent.appendedLines)) + ) { + // Cap the frame to the viewport and keep scrollback dirty: transient + // rows never enter history, and the checkpoint reconciles later. + this.#markNativeScrollbackDirty(); + this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); + this.#scrollbackHighWater = 0; + lines = lines.slice(-height); + intent = { kind: "viewportRepaint" }; + } else { + // Explicit reconcile or a non-committing frame (noop): let the + // planned intent stand, but keep tracking the streaming peak. + this.#streamingHighWater = Math.max(this.#streamingHighWater, lines.length); + } + } if (this.#eagerNativeScrollbackRebuildDisablePending) { this.#eagerNativeScrollbackRebuildDisablePending = false; this.#eagerNativeScrollbackRebuild = false; @@ -1448,6 +1543,18 @@ export class TUI extends Container { }); this.#emitViewportRepaint(lines, width, height, cursorPos); return; + case "liveRegionPinned": + this.#emitLiveRegionPinnedRepaint( + lines, + width, + height, + cursorPos, + intent.appendFrom, + intent.appendTo, + prevViewportTop, + prevHardwareCursorRow, + ); + return; case "viewportRepaint": if (intent.appendFrom !== undefined) { this.#emitAppendTail(lines, intent.appendFrom, height, width, prevViewportTop, prevHardwareCursorRow); @@ -1500,6 +1607,7 @@ export class TUI extends Container { hasVisibleOverlay: boolean, overlayVisibilityReduced: boolean, allowUnknownViewportMutation: boolean, + liveRegionStart: number | undefined, ): RenderIntent { // Initial paint after start(): scrollback must keep its prior shell // content, but the viewport must be cleared so stale rows do not bleed @@ -1532,6 +1640,29 @@ export class TUI extends Container { return { kind: "viewportRepaint" }; } + const liveRegionPinnedIntent = this.#planLiveRegionPinnedRender( + newLines, + height, + liveRegionStart, + eagerEraseScrollbackRisk, + allowUnknownViewportMutation, + widthChanged || heightChanged, + ); + if (liveRegionPinnedIntent) return liveRegionPinnedIntent; + + // After foreground tool streaming: when content finally shrinks from the + // streaming peak, rebuild with ED 3 to commit the settled state cleanly. + // The check uses `#streamingHighWater` (the real peak) rather than + // `#previousLines.length` because unpinned ED3-risk streaming frames may + // commit only a viewport slice while native history is deferred. + if (this.#streamingHighWater > height && newLines.length < this.#streamingHighWater && newLines.length > height) { + this.#streamingHighWater = 0; + return { kind: "historyRebuild" }; + } + if (this.#streamingHighWater > 0 && newLines.length <= height) { + this.#streamingHighWater = 0; + } + if ( this.#nativeScrollbackDirty && !isMultiplexerSession() && @@ -2036,6 +2167,40 @@ export class TUI extends Container { ); } + #planLiveRegionPinnedRender( + newLines: string[], + height: number, + liveRegionStart: number | undefined, + eagerEraseScrollbackRisk: boolean, + allowUnknownViewportMutation: boolean, + geometryChanged: boolean, + ): RenderIntent | undefined { + // A width/height change reflows the whole terminal: the relative cursor + // positioning this emitter relies on is computed from the pre-resize + // geometry and would land on the wrong rows. Defer to the geometry branch + // (a full reflow rebuild), which is the established behavior for resizes. + if ( + liveRegionStart === undefined || + liveRegionStart >= newLines.length || + !this.#eagerNativeScrollbackRebuild || + !eagerEraseScrollbackRisk || + allowUnknownViewportMutation || + geometryChanged || + isMultiplexerSession() + ) { + return undefined; + } + if (newLines.length <= height && this.#scrollbackHighWater === 0) return undefined; + if (this.#readNativeViewportAtBottom() !== undefined) return undefined; + + this.#markNativeScrollbackDirty(); + const viewportTop = Math.max(0, newLines.length - height); + const sealedEnd = Math.max(0, Math.min(liveRegionStart, newLines.length)); + const appendTo = Math.min(sealedEnd, viewportTop); + const appendFrom = Math.min(this.#scrollbackHighWater, appendTo); + return { kind: "liveRegionPinned", appendFrom, appendTo }; + } + #padDeferredShrinkLines(lines: string[], paddedLength: number): string[] { if (lines.length >= paddedLength) return lines; return [...lines, ...new Array(paddedLength - lines.length).fill("")]; @@ -2208,6 +2373,74 @@ export class TUI extends Container { this.#commit(lines, width, height, viewportTop, toRow); } + /** + * Foreground-stream live-region paint for ED3-risk terminals with an + * unobservable viewport. Commits the newly-sealed chunk to native scrollback + * (so finished blocks stay scrollable) and repaints the live tail in place, + * leaving the transient live region out of saved lines. + * + * Uses only the no-scroll-snap vocabulary of {@link #emitDiff}: relative + * cursor moves, per-line `\x1b[2K`, and `\r\n` to push the sealed chunk into + * history. It deliberately avoids a full-screen erase (`\x1b[2J`) and absolute + * cursor home (`\x1b[H`): on Ghostty those snap a reader scrolled into history + * back to the bottom on every frame. + */ + #emitLiveRegionPinnedRepaint( + lines: string[], + width: number, + height: number, + cursorPos: { row: number; col: number } | null, + appendFrom: number, + appendTo: number, + prevViewportTop: number, + prevHardwareCursorRow: number, + ): void { + this.#fullRedrawCount += 1; + const viewportTop = Math.max(0, lines.length - height); + const boundedAppendTo = Math.max(0, Math.min(appendTo, viewportTop, lines.length)); + const boundedAppendFrom = Math.max(0, Math.min(appendFrom, boundedAppendTo)); + + // Position at the top visible row with a relative move. Terminals clamp the + // hardware cursor to the viewport on resize, so clamp our tracking to match + // before computing the delta (mirrors #emitDiff). + const clampedCursor = Math.min(prevHardwareCursorRow, prevViewportTop + height - 1); + const currentScreenRow = Math.max(0, Math.min(height - 1, clampedCursor - prevViewportTop)); + let buffer = this.#paintBeginSequence; + if (currentScreenRow > 0) buffer += `\x1b[${currentScreenRow}A`; + buffer += "\r"; + + // Write the sealed chunk followed by the full viewport from the top row. + // The first (boundedAppendTo - boundedAppendFrom) rows scroll into native + // history; the trailing `height` rows fill the viewport. Each row clears + // itself with `\x1b[2K` instead of relying on a screen-wide erase. + let wroteLine = false; + for (let i = boundedAppendFrom; i < boundedAppendTo; i++) { + if (wroteLine) buffer += "\r\n"; + buffer += `\x1b[2K${this.#fitLineToWidth(lines[i] ?? "", width)}`; + wroteLine = true; + } + for (let screenRow = 0; screenRow < height; screenRow++) { + if (wroteLine) buffer += "\r\n"; + buffer += `\x1b[2K${this.#fitLineToWidth(lines[viewportTop + screenRow] ?? "", width)}`; + wroteLine = true; + } + + const viewportBottomRow = viewportTop + height - 1; + const contentBottomRow = Math.min(viewportBottomRow, Math.max(viewportTop, lines.length - 1)); + const parkUp = viewportBottomRow - contentBottomRow; + if (parkUp > 0) buffer += `\x1b[${parkUp}A`; + const { seq, toRow } = this.#cursorControlSequence(cursorPos, lines.length, contentBottomRow); + buffer += seq; + buffer += this.#paintEndSequence; + this.terminal.write(buffer); + + this.#maxLinesRendered = lines.length; + if (boundedAppendTo > this.#scrollbackHighWater) { + this.#scrollbackHighWater = boundedAppendTo; + } + this.#commit(lines, width, height, viewportTop, toRow); + } + /** * Push the appended tail into terminal scrollback by `\r\n`-ing past the * previous viewport bottom. Used as a prefix to {@link #emitViewportRepaint} @@ -2443,9 +2676,11 @@ export class TUI extends Container { const detail = intent.kind === "diff" ? `${intent.kind}(first=${intent.firstChanged}, last=${intent.lastChanged}, appended=${intent.appendedLines})` - : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined - ? `${intent.kind}(appendFrom=${intent.appendFrom})` - : intent.kind; + : intent.kind === "liveRegionPinned" + ? `${intent.kind}(append=${intent.appendFrom}..${intent.appendTo})` + : intent.kind === "viewportRepaint" && intent.appendFrom !== undefined + ? `${intent.kind}(appendFrom=${intent.appendFrom})` + : intent.kind; const msg = `[${new Date().toISOString()}] render: ${detail} (prev=${this.#previousLines.length}, new=${newLength}, height=${height})\n`; fs.appendFileSync(getDebugLogPath(), msg); } diff --git a/packages/tui/test/kitty-graphics.test.ts b/packages/tui/test/kitty-graphics.test.ts index 03870907c..b2e81fb6c 100644 --- a/packages/tui/test/kitty-graphics.test.ts +++ b/packages/tui/test/kitty-graphics.test.ts @@ -2,6 +2,7 @@ import { afterEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import { visibleWidth } from "@oh-my-pi/pi-natives"; import { + detectKittyUnicodePlaceholdersSupport, encodeKittyPlaceholderGrid, encodeKittyTempFileProbe, encodeKittyTempFileTransmit, @@ -156,3 +157,42 @@ describe("kitty graphics feature state", () => { } }); }); + +describe("detectKittyUnicodePlaceholdersSupport", () => { + function env(extra: Record = {}): NodeJS.ProcessEnv { + return extra as NodeJS.ProcessEnv; + } + + it("enables for kitty and ghostty by default (the only terminals that render U=1 placement)", () => { + expect(detectKittyUnicodePlaceholdersSupport("kitty", env())).toBe(true); + expect(detectKittyUnicodePlaceholdersSupport("ghostty", env())).toBe(true); + }); + + it("disables for wezterm and other Kitty-protocol paths that treat placeholders as literal PUA glyphs (#1877)", () => { + expect(detectKittyUnicodePlaceholdersSupport("wezterm", env())).toBe(false); + // Tmux/screen fallback: base terminal id with Kitty protocol forced on by + // `getFallbackImageProtocol`. The outer terminal need not understand U=1. + expect(detectKittyUnicodePlaceholdersSupport("base", env())).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("iterm2", env())).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("alacritty", env())).toBe(false); + }); + + it("honors PI_NO_KITTY_PLACEHOLDERS=1 as a hard off override on supporting terminals", () => { + expect(detectKittyUnicodePlaceholdersSupport("kitty", env({ PI_NO_KITTY_PLACEHOLDERS: "1" }))).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("ghostty", env({ PI_NO_KITTY_PLACEHOLDERS: "true" }))).toBe(false); + }); + + it("honors PI_KITTY_PLACEHOLDERS=1 as opt-in on otherwise-unsupported terminals", () => { + expect(detectKittyUnicodePlaceholdersSupport("wezterm", env({ PI_KITTY_PLACEHOLDERS: "1" }))).toBe(true); + }); + + it("PI_NO_KITTY_PLACEHOLDERS beats PI_KITTY_PLACEHOLDERS when both are set", () => { + const both = env({ PI_NO_KITTY_PLACEHOLDERS: "1", PI_KITTY_PLACEHOLDERS: "1" }); + expect(detectKittyUnicodePlaceholdersSupport("kitty", both)).toBe(false); + }); + + it("PI_KITTY_PLACEHOLDERS=0 forces off on a default-on terminal", () => { + expect(detectKittyUnicodePlaceholdersSupport("kitty", env({ PI_KITTY_PLACEHOLDERS: "0" }))).toBe(false); + expect(detectKittyUnicodePlaceholdersSupport("ghostty", env({ PI_KITTY_PLACEHOLDERS: "off" }))).toBe(false); + }); +}); diff --git a/packages/tui/test/streaming-scrollback-defer.test.ts b/packages/tui/test/streaming-scrollback-defer.test.ts new file mode 100644 index 000000000..3e6233e4f --- /dev/null +++ b/packages/tui/test/streaming-scrollback-defer.test.ts @@ -0,0 +1,216 @@ +import { describe, expect, it } from "bun:test"; +import { type Component, type NativeScrollbackLiveRegion, TERMINAL, TUI } from "@oh-my-pi/pi-tui"; +import { VirtualTerminal } from "./virtual-terminal"; + +class LineList implements Component { + #lines: string[]; + + constructor(lines: string[]) { + this.#lines = [...lines]; + } + + invalidate(): void {} + + render(width: number): string[] { + return this.#lines.map(line => line.slice(0, width)); + } + + setLines(lines: string[]): void { + this.#lines = [...lines]; + } +} + +class LiveLineList extends LineList implements NativeScrollbackLiveRegion { + getNativeScrollbackLiveRegionStart(): number | undefined { + return 0; + } +} + +async function settle(term: VirtualTerminal): Promise { + await Bun.sleep(20); + await term.flush(); +} + +function capture(term: VirtualTerminal): string[] { + const writes: string[] = []; + const realWrite = term.write.bind(term); + (term as unknown as { write: (s: string) => void }).write = (data: string) => { + writes.push(data); + realWrite(data); + }; + return writes; +} + +function overrideProbe(term: VirtualTerminal, answer: boolean | undefined): void { + (term as unknown as { isNativeViewportAtBottom: () => boolean | undefined }).isNativeViewportAtBottom = () => answer; +} + +type MutableTerminalInfo = { + eagerEraseScrollbackRisk: boolean; +}; + +const mutableTerminalInfo = TERMINAL as unknown as MutableTerminalInfo; + +async function withTerminalRisk(risk: boolean, run: () => T | Promise): Promise { + const saved = TERMINAL.eagerEraseScrollbackRisk; + mutableTerminalInfo.eagerEraseScrollbackRisk = risk; + try { + return await run(); + } finally { + mutableTerminalInfo.eagerEraseScrollbackRisk = saved; + } +} + +const ERASE_SCROLLBACK = /\x1b\[3J/g; + +function eraseScrollbackCount(writes: string[]): number { + return writes.join("").match(ERASE_SCROLLBACK)?.length ?? 0; +} + +function rows(prefix: string, count: number): string[] { + return Array.from({ length: count }, (_, i) => `${prefix}${i}`); +} + +describe("streaming scrollback defer", () => { + it("keeps sealed prefix scrollable while deferring live-region rows on ED3-risk terminals", async () => { + if (process.platform === "win32") return; + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(20, 4); + overrideProbe(term, undefined); + const tui = new TUI(term); + const sealed = new LineList(rows("prior-", 12)); + const live = new LiveLineList([]); + + try { + tui.addChild(sealed); + tui.addChild(live); + tui.start(); + await settle(term); + + const writes = capture(term); + tui.setEagerNativeScrollbackRebuild(true); + + live.setLines(rows("think-", 6)); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect(term.getScrollBuffer().map(line => line.trimEnd())).toEqual([ + ...rows("prior-", 12), + ...rows("think-", 6).slice(-4), + ]); + + live.setLines(rows("think-", 8)); + tui.requestRender(); + await settle(term); + + const buffer = term.getScrollBuffer().map(line => line.trimEnd()); + expect(eraseScrollbackCount(writes)).toBe(0); + expect(buffer.filter(line => line.startsWith("prior-"))).toEqual(rows("prior-", 12)); + expect(buffer.slice(-4)).toEqual(rows("think-", 8).slice(-4)); + } finally { + tui.stop(); + } + }); + }); + + it("defers scrollback growth during eager streaming on ED3-risk and reconciles at the checkpoint", async () => { + if (process.platform === "win32") return; + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(40, 10); + overrideProbe(term, undefined); + const tui = new TUI(term); + const component = new LineList([...rows("init-", 10), "prompt"]); + + try { + tui.addChild(component); + tui.start(); + await settle(term); + + const writes = capture(term); + const scrollbackBefore = term.getScrollBuffer().length; + + tui.setEagerNativeScrollbackRebuild(true); + + // Grow content past the viewport — capped, no rows enter native + // scrollback during streaming, and no ED3 erase fires. + component.setLines([...rows("stream-", 10), ...rows("more-", 30), "prompt"]); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect(term.getScrollBuffer().length).toBe(scrollbackBefore); + expect( + term + .getViewport() + .map(line => line.trim()) + .at(-1), + ).toBe("prompt"); + + // Grow even more — still capped, still no ED3. + component.setLines([...rows("stream-", 10), ...rows("more-", 50), "prompt"]); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect(term.getScrollBuffer().length).toBe(scrollbackBefore); + + // The prompt-submit checkpoint reconciles the deferred transcript with + // a single ED3 + re-emit — even while eager is still active, because an + // explicit reconcile is never deferred. + expect(tui.refreshNativeScrollbackIfDirty({ allowUnknownViewport: true })).toBe(true); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(1); + + const scrollbackAfter = term.getScrollBuffer(); + expect(scrollbackAfter.length).toBeGreaterThan(scrollbackBefore); + expect(scrollbackAfter.join("\n")).toContain("stream-"); + expect(scrollbackAfter.join("\n")).toContain("more-"); + } finally { + tui.stop(); + } + }); + }); + + it("does not emit ED3 during streaming on ED3-risk terminals", async () => { + if (process.platform === "win32") return; + await withTerminalRisk(true, async () => { + const term = new VirtualTerminal(40, 10); + overrideProbe(term, undefined); + const tui = new TUI(term); + const component = new LineList([...rows("init-", 10), "prompt"]); + + try { + tui.addChild(component); + tui.start(); + await settle(term); + + const writes = capture(term); + + tui.setEagerNativeScrollbackRebuild(true); + + component.setLines([...rows("grow-", 30), "prompt"]); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + + // Disable on ED3-risk — no historyRebuild + tui.setEagerNativeScrollbackRebuild(false); + tui.requestRender(); + await settle(term); + + expect(eraseScrollbackCount(writes)).toBe(0); + expect( + term + .getViewport() + .map(line => line.trim()) + .at(-1), + ).toBe("prompt"); + } finally { + tui.stop(); + } + }); + }); +}); diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 98ba884df..002905c92 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -5,6 +5,13 @@ ### Added - Added profile-aware directory helpers and isolated profile state roots, while keeping the install ID shared across profiles. + +## [15.9.2] - 2026-06-05 + +### Added + +- Added `getAuthBrokerSnapshotCachePath()` with `OMP_AUTH_BROKER_SNAPSHOT_CACHE` override support for isolating the encrypted broker snapshot cache. + ## [15.9.1] - 2026-06-04 ### Fixed diff --git a/packages/utils/package.json b/packages/utils/package.json index 231cd020e..5fd061372 100644 --- a/packages/utils/package.json +++ b/packages/utils/package.json @@ -1,7 +1,7 @@ { "type": "module", "name": "@oh-my-pi/pi-utils", - "version": "15.9.1", + "version": "15.9.2", "description": "Shared utilities for pi packages", "homepage": "https://omp.sh", "author": "Can Boluk", diff --git a/packages/utils/src/dirs.ts b/packages/utils/src/dirs.ts index ee07bfa83..24f798256 100644 --- a/packages/utils/src/dirs.ts +++ b/packages/utils/src/dirs.ts @@ -532,6 +532,17 @@ export function getGithubCacheDbPath(): string { return dirs.rootSubdir(path.join("cache", "github-cache.db"), "cache"); } +/** + * Get the encrypted auth-broker snapshot cache path (~/.omp/cache/auth-broker-snapshot.enc). + * Honors the `OMP_AUTH_BROKER_SNAPSHOT_CACHE` env var when set so tests and + * operators can isolate or relocate the cache file. + */ +export function getAuthBrokerSnapshotCachePath(): string { + const override = process.env.OMP_AUTH_BROKER_SNAPSHOT_CACHE; + if (override) return override; + return dirs.rootSubdir(path.join("cache", "auth-broker-snapshot.enc"), "cache"); +} + /** Get the local FastEmbed model cache directory (~/.omp/cache/fastembed). */ export function getFastembedCacheDir(): string { return dirs.rootSubdir(path.join("cache", "fastembed"), "cache");