diff --git a/Dockerfile b/Dockerfile index f98f81e4c..ec15e9fcc 100644 --- a/Dockerfile +++ b/Dockerfile @@ -184,9 +184,11 @@ RUN bun install --frozen-lockfile --ignore-scripts # hoisted node_modules that `bun install` just produced. COPY . /pi/ -# Regenerate the docs index that `--ignore-scripts` skipped above. The root -# package.json's `prepare` script normally handles this on a vanilla install. -RUN bun --cwd=packages/coding-agent run gen:docs +# Regenerate the docs index and tool views that `--ignore-scripts` skipped +# above. The root package.json's `prepare` script normally handles these on a +# vanilla install. +RUN bun --cwd=packages/coding-agent run gen:docs \ + && bun --cwd=/pi/packages/coding-agent run gen:tool-views ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/omp"] CMD ["--help"] diff --git a/crates/pi-shell/src/process.rs b/crates/pi-shell/src/process.rs index ede0fd3f8..67a78f187 100644 --- a/crates/pi-shell/src/process.rs +++ b/crates/pi-shell/src/process.rs @@ -1571,6 +1571,11 @@ impl TerminationTargets { /// Record a pid. Duplicates are ignored. If the pid is alive, opens /// a stable [`Process`] reference so the descendant tree can be /// killed even if the original pid is reused later. + /// + /// Prefer [`add_process`](Self::add_process) when the caller already holds a + /// [`Process`] captured at spawn time: opening by pid here loses the + /// original identity if the pid was recycled between the child exiting + /// and this call. pub fn add_pid(&mut self, pid: i32) { if self.seen_pids.insert(pid) && let Some(process) = Process::from_pid(pid) @@ -1579,6 +1584,17 @@ impl TerminationTargets { } } + /// Record a pre-pinned [`Process`] handle. Duplicates (by pid) are ignored. + /// + /// This is the correct entry point when the caller captured the handle at + /// spawn time — the handle already pins OS-level identity, so no `from_pid` + /// re-open (and its PID-reuse race) is needed at cancellation time. + pub fn add_process(&mut self, process: Process) { + if self.seen_pids.insert(process.pid()) { + self.processes.push(process); + } + } + /// True when no targets have been recorded. #[must_use] pub const fn is_empty(&self) -> bool { @@ -1599,10 +1615,19 @@ impl TerminationTargets { } /// A single external child reported by the shell's spawn-observer hook. -#[derive(Debug, Clone, Copy)] +/// +/// `process` is captured *at spawn time* so its OS-level identity is pinned +/// before the pid can be recycled. On Windows an open process handle keeps +/// the pid reserved for the lifetime of the reference; on Linux the pidfd +/// pins identity; on macOS the recorded `(pid, start_time)` triple detects +/// impersonation. Storing only the raw pid and re-opening at cancellation +/// time — as previous versions did — leaked kills onto unrelated processes +/// that happened to acquire the recycled pid between the child exiting and +/// the run being cancelled (issue #4605). +#[derive(Clone)] struct SpawnedProcess { - pid: i32, - pgid: Option, + process: Option, + pgid: Option, } /// Per-run record of the OS processes a single shell command launched, @@ -1613,12 +1638,43 @@ struct SpawnedProcess { /// host process: a run that cancelled would signal *any* descendant spawned /// after its baseline, including another run's children. Ownership is now /// explicit — only processes this run actually spawned are ever signalled. +#[derive(Default)] +struct RegistryState { + spawned: Vec, + /// The next `spawned.len()` at which `record` runs a sweep. Bounds sweep + /// frequency when the live set stabilizes above the initial threshold: + /// without this watermark, every subsequent `record` would find + /// `len >= PRUNE_THRESHOLD` true and sweep on every spawn (O(n²) in a + /// large-fan-out run like `for i in {1..1000}; do sleep 60 & done`). With + /// it, the next sweep only fires once the vec has grown by another + /// `PRUNE_THRESHOLD` entries since the previous sweep — restoring true + /// amortized O(1) per spawn regardless of how many entries survive each + /// sweep. + next_sweep_at: usize, +} + #[derive(Default)] pub struct SpawnRegistry { - spawned: Mutex>, + state: Mutex, } impl SpawnRegistry { + /// Amortized-cost threshold for opportunistic pruning of exited entries. + /// + /// A shell run that spawns many short-lived external commands (e.g. a bash + /// loop invoking a binary per iteration) would otherwise retain one owned + /// process handle per spawn — a pidfd on Linux, a `HANDLE` on Windows — for + /// the lifetime of the run, exhausting per-process FD/handle limits. + /// + /// Each sweep costs `O(N)` (one non-blocking status probe per entry, plus + /// a Toolhelp descendant walk on Windows for exited roots). The next sweep + /// is scheduled `PRUNE_THRESHOLD` further records away — via the + /// `next_sweep_at` watermark — so a run that keeps many concurrent + /// long-lived children (`for i in {1..1000}; do sleep 60 & done`) does not + /// sweep on every spawn just because the vec is already above threshold. + /// Amortized cost per spawn stays `O(1)` regardless of the live-set size. + const PRUNE_THRESHOLD: usize = 64; + /// Create an empty registry. #[must_use] pub fn new() -> Self { @@ -1626,24 +1682,64 @@ impl SpawnRegistry { } /// Record a freshly spawned child. Called from the spawn-observer hook. - pub fn record(&self, pid: i32, pgid: Option) { - self.spawned.lock().push(SpawnedProcess { pid, pgid }); + /// + /// The `Process` handle MUST be opened by the caller *immediately* after + /// the child's pid becomes visible, so identity is pinned before any race + /// with pid recycling can start. When the pin fails (child already exited + /// before we could `Process::from_pid`) the entry becomes a no-op at + /// termination time — there is nothing left to signal. + /// + /// Exited entries are swept opportunistically once the recorded vec + /// crosses the next-sweep watermark, so long-running loops of short + /// external commands cannot exhaust the process' FD/handle limit by + /// retaining one owned handle per historical spawn. + pub fn record(&self, pgid: Option, process: Option) { + let mut state = self.state.lock(); + state.spawned.push(SpawnedProcess { process, pgid }); + if state.spawned.len() >= state.next_sweep_at.max(Self::PRUNE_THRESHOLD) { + prune_exited(&mut state.spawned); + // Schedule the next sweep `PRUNE_THRESHOLD` further records away. + // Comparing against the post-sweep live-set size (not the pre-sweep + // length) bounds the sweep frequency when many entries survive: + // each sweep costs O(N) but now runs at most once per + // `PRUNE_THRESHOLD` records, so amortized per-record cost is O(1) + // even if the live set stays large. + state.next_sweep_at = state.spawned.len() + Self::PRUNE_THRESHOLD; + } } /// Build the kill set from the processes recorded so far. Re-read on every /// signal wave so a child spawned during a grace window — between the /// cancel firing and the next wave — is still reaped. /// - /// A recorded pid contributes only while alive (`add_pid` opens a stable - /// handle, skipping the dead); a recorded pgid contributes only while the - /// group still has members, so once the run's whole tree exits the targets - /// are empty and the wave loop can stop early. + /// A recorded process contributes only while alive; a recorded pgid + /// contributes only while the group still has members, so once the run's + /// whole tree exits the targets are empty and the wave loop can stop early. + /// + /// Pruning also runs here so a cancellation cycle sees a compact target + /// set even when the record-time threshold hasn't fired yet. #[must_use] pub fn build_targets(&self) -> TerminationTargets { let mut targets = TerminationTargets::new(); - let spawned = self.spawned.lock().clone(); + let spawned = { + let mut state = self.state.lock(); + prune_exited(&mut state.spawned); + // Reset the watermark to the current live-set size + threshold; + // leaving a stale pre-sweep value would misgate the next + // record-time sweep. + state.next_sweep_at = state.spawned.len() + Self::PRUNE_THRESHOLD; + state.spawned.clone() + }; for entry in spawned { - targets.add_pid(entry.pid); + if let Some(process) = entry.process { + targets.add_process(process); + } + // If the observer failed to pin a handle at spawn time (the child + // exited before `Process::from_pid` could open it), the child is + // already gone — signalling anything for that pid would either + // no-op or, worse, race a recycled pid onto an unrelated process. + // Drop the entry entirely rather than reintroduce the pid-reuse + // window this whole change exists to close (#4605). if let Some(pgid) = entry.pgid && pgid > 0 && process_group_alive(pgid) @@ -1655,6 +1751,40 @@ impl SpawnRegistry { } } +/// Drop registry entries whose pinned process, process group, and — on +/// Windows — descendant tree are all gone. With nothing still-live the entry +/// contributes nothing to the next termination wave and only pins an owned OS +/// handle for no reason. +/// +/// The platform split matters because Windows has no process groups. On Unix +/// a child reparented onto init keeps its pgid, so a live pgid still catches +/// grandchildren whose immediate parent exited. On Windows there is no +/// reparenting and no pgid, so we probe the descendant tree directly through +/// the still-open pinned handle — dropping that handle would release the pid +/// slot, letting a recycled pid make future Toolhelp walks unsafe (issue +/// #4605) and orphaning any leftover child from the next cancellation wave. +fn prune_exited(spawned: &mut Vec) { + spawned.retain(|entry| { + if let Some(process) = &entry.process { + if process.status() == ProcessStatus::Running { + return true; + } + // Windows-only: root exited but the pinned handle still keeps its + // pid reserved, so `live_descendants` walks the *original* subtree + // via Toolhelp. If any child is still running we must keep the + // entry — closing the handle would both release the pid (racing + // pid reuse) and strand the surviving child. + #[cfg(target_os = "windows")] + if !process.live_descendants().is_empty() { + return true; + } + } + entry + .pgid + .is_some_and(|pgid| pgid > 0 && process_group_alive(pgid)) + }); +} + /// True when process group `pgid` still has at least one member. `kill(2)` /// with signal 0 performs permission/existence checks without delivering a /// signal; `EPERM` means the group exists but is not ours to signal, which @@ -1752,4 +1882,206 @@ mod tests { broken `proc_listchildpids`", ); } + + /// Regression test for issue #4605: `SpawnRegistry` MUST pin a stable + /// [`Process`] reference at spawn time rather than defer re-opening the + /// pid until termination. + /// + /// Before the fix, `SpawnRegistry` stored only the raw pid; `build_targets` + /// called `Process::from_pid` at cancellation time. On Windows pids recycle + /// aggressively, so a bash-spawned `pwsh.exe` that had already exited could + /// see its pid reassigned to an unrelated PowerShell session (e.g. the + /// user's other Cursor terminal). `Process::from_pid` at cancel time would + /// happily open that unrelated process, and `signal_tree` would then + /// enumerate — and `TerminateProcess` — the entire foreign subtree. + /// + /// This test cannot literally trigger Windows pid recycling from a + /// cross-platform Rust test, but it can prove the observable defense: a + /// recorded process reference survives the original pid's death (so no + /// "look it up again" step exists to be raced), and the registry never + /// consults `Process::from_pid` when a handle was pinned at record time. + #[cfg(unix)] + #[test] + fn spawn_registry_pins_identity_at_record_time() { + use std::{process::Command, thread, time::Duration}; + + // Phase 1: while the child is alive, the pinned handle carries identity + // forward into `build_targets` without any `Process::from_pid` re-open + // step existing to be raced against pid reuse. + let mut long = Command::new("sleep") + .arg("30") + .spawn() + .expect("spawn sleep"); + let long_pid = i32::try_from(long.id()).expect("child pid fits in i32"); + + let registry = SpawnRegistry::new(); + let pinned = Process::from_pid(long_pid).expect("pin child at record time"); + registry.record(None, Some(pinned)); + + let live_targets = registry.build_targets(); + assert!( + !live_targets.is_empty(), + "a still-live pinned child must appear in the target set — otherwise the \ + cancellation cleanup would silently miss it" + ); + let live_pids: Vec = live_targets.processes.iter().map(Process::pid).collect(); + assert_eq!( + live_pids, + vec![long_pid], + "target set must come from the pinned handle recorded at spawn time, not a \ + re-lookup by pid (which would race pid reuse — issue #4605)" + ); + + let _ = long.kill(); + let _ = long.wait(); + + // Phase 2: once the child exits, the registry MUST drop the entry + // rather than reintroduce a `Process::from_pid` re-open at kill time. + // Poll until pruning sees the pidfd as Exited (kernel-visible within + // milliseconds in practice). + let mut empty_after_exit = false; + for _ in 0..40 { + if registry.build_targets().is_empty() { + empty_after_exit = true; + break; + } + thread::sleep(Duration::from_millis(25)); + } + assert!( + empty_after_exit, + "once the pinned child exits the registry must drop it — re-opening by pid at \ + termination time is exactly the pid-reuse race #4605 closes" + ); + } + + /// `TerminationTargets::add_process` must accept a pre-pinned handle + /// without going through `Process::from_pid`. This is the API contract + /// `SpawnRegistry` relies on to avoid the PID-reuse race. + #[cfg(unix)] + #[test] + fn add_process_bypasses_from_pid_lookup() { + let self_pid = i32::try_from(std::process::id()).expect("self pid fits in i32"); + let pinned = Process::from_pid(self_pid).expect("pin self"); + + let mut targets = TerminationTargets::new(); + targets.add_process(pinned.clone()); + assert!(!targets.is_empty(), "add_process must record the pinned handle"); + + // Adding the same pid again through either entry point must dedupe: + // otherwise every wave in `terminate_run` would re-signal the same + // tree N times. + targets.add_process(pinned); + targets.add_pid(self_pid); + assert_eq!(targets.processes.len(), 1, "duplicate pids must be deduped"); + } + + /// Regression test for the review on PR #4606: a long-running shell + /// command that spawns many short-lived external processes must not + /// retain one owned handle per historical spawn — that would exhaust + /// per-process FD/handle limits (pidfd on Linux, `HANDLE` on Windows). + /// The registry MUST prune dead entries once the recorded vec crosses + /// the sweep threshold. + #[cfg(unix)] + #[test] + fn spawn_registry_prunes_exited_entries() { + use std::{thread, time::Duration}; + + let registry = SpawnRegistry::new(); + + // Fabricate many recorded-then-exited children by pinning ourselves, + // pushing the entry, then immediately treating it as "dead" from the + // registry's perspective. To simulate the exit without actually + // killing the harness, use `Process::from_pid(1)` for a pid that + // (on Linux) is init and never exits — but wrap the recording in a + // pattern that guarantees `status()` returns Exited for the pruner: + // spawn a tiny child, pin it, wait for exit, then record. + for _ in 0..(SpawnRegistry::PRUNE_THRESHOLD * 2) { + let mut child = std::process::Command::new("true") + .spawn() + .expect("spawn true"); + let pid = i32::try_from(child.id()).expect("child pid fits in i32"); + let pinned = Process::from_pid(pid); + let _ = child.wait(); + // Give the kernel a moment to mark the pidfd readable so `status()` + // reports Exited when the pruner probes. + for _ in 0..20 { + if pinned + .as_ref() + .is_some_and(|process| process.status() == ProcessStatus::Exited) + { + break; + } + thread::sleep(Duration::from_millis(5)); + } + registry.record(None, pinned); + } + + let retained = registry.state.lock().spawned.len(); + assert!( + retained < SpawnRegistry::PRUNE_THRESHOLD, + "pruning must bound retained entries below the sweep threshold once the pinned \ + processes have exited; got {retained} retained (threshold {})", + SpawnRegistry::PRUNE_THRESHOLD + ); + + // build_targets sees no live handles → empty target set, matching the + // contract that fully-exited registries stop the wave loop early. + let targets = registry.build_targets(); + assert!( + targets.is_empty(), + "registry of only-dead entries must produce an empty target set" + ); + } + + /// Regression test for the third review on PR #4606: once the recorded + /// vec crosses `PRUNE_THRESHOLD`, subsequent `record` calls must NOT + /// sweep on every spawn. Without the `next_sweep_at` watermark, a large + /// fan-out run whose live children exceed the threshold turned every + /// spawn into an O(N) status probe of the whole retained set. + /// + /// The check reasons about the observable side effect: after N records + /// past threshold with entries that CANNOT be pruned (all still live), + /// the retained size grows monotonically by exactly N — no sweep runs + /// have modified the vec in between. The direct signal of "did a sweep + /// happen" is a stable pinned handle count across records. + #[cfg(unix)] + #[test] + fn spawn_registry_watermark_bounds_sweep_frequency() { + let self_pid = i32::try_from(std::process::id()).expect("self pid fits in i32"); + let registry = SpawnRegistry::new(); + + // Fill past threshold with entries that are permanently alive + // (pinning ourselves) so the pruner has nothing to remove. + let fill = SpawnRegistry::PRUNE_THRESHOLD + 10; + for _ in 0..fill { + registry.record(None, Process::from_pid(self_pid)); + } + let after_fill = registry.state.lock().spawned.len(); + assert_eq!( + after_fill, fill, + "live-only entries must not be pruned during warm-up" + ); + let watermark_after_fill = registry.state.lock().next_sweep_at; + + // Every additional record with a live entry must land in the vec + // verbatim and — critically — NOT re-enter `prune_exited` until the + // vec crosses the freshly scheduled watermark. If the guard were + // still `len >= PRUNE_THRESHOLD` (pre-fix), a sweep would fire on + // every one of these records. + let extra = 20; + for _ in 0..extra { + registry.record(None, Process::from_pid(self_pid)); + } + let after_extra = registry.state.lock().spawned.len(); + assert_eq!( + after_extra, + after_fill + extra, + "records with live entries must accumulate without triggering per-spawn sweeps" + ); + assert_eq!( + registry.state.lock().next_sweep_at, + watermark_after_fill, + "watermark must not advance while the vec stays below it — otherwise a sweep ran" + ); + } } diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs index f00982a92..80634bf91 100644 --- a/crates/pi-shell/src/shell.rs +++ b/crates/pi-shell/src/shell.rs @@ -1372,7 +1372,14 @@ async fn read_output_bytes( impl SpawnObserver for process::SpawnRegistry { fn on_spawn(&self, pid: i32, pgid: Option) { - self.record(pid, pgid); + // Pin a stable process reference *now*, before the pid can be recycled. + // On Windows an open handle keeps the pid slot reserved for the lifetime + // of the handle; on Linux the pidfd carries identity; on macOS the + // recorded start-time triple detects impersonation. Deferring the open + // to `build_targets` (as the old code did) let a recycled pid resolve + // to an unrelated process — issue #4605. + let process = process::Process::from_pid(pid); + self.record(pgid, process); } } diff --git a/docs/compaction.md b/docs/compaction.md index 2112875eb..186d53054 100644 --- a/docs/compaction.md +++ b/docs/compaction.md @@ -240,9 +240,9 @@ Prompt selection: Remote summarization modes: -- If `compaction.remoteEndpoint` is set and remote compaction is enabled, local summary generation POSTs: - - `{ systemPrompt, prompt }` -- Expects JSON containing at least `{ summary }`. +- If `compaction.remoteEndpoint` is set and remote compaction is enabled, local summary generation POSTs one of two wire formats: + - custom omp summarizer endpoints receive `{ systemPrompt, prompt }` and must return JSON containing at least `{ summary }`. + - OpenAI-compatible endpoints whose path ends in `/chat/completions` receive `{ model, messages, stream: false }`, where `messages` contains one system prompt and one user prompt. The summary is read from `choices[0].message.content`, which lets self-hosted servers such as llama.cpp and vLLM act as remote compactors without a separate summarizer shim. - For OpenAI/OpenAI Codex models, compaction first tries the provider-native `/responses/compact` endpoint when remote compaction is enabled. It preserves provider replacement history in `preserveData.openaiRemoteCompaction` and falls back to local summarization if that native request fails. ### Handoff generation diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index 1de3328af..0d1b46a12 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,15 @@ ## [Unreleased] +### Added + +- Added per-tool abort metadata so stream-wide aborts can label matching tool-call placeholders separately from unaffected sibling calls ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). + +### Fixed + +- Fixed handoff generation retrying with `toolChoice: "auto"` when custom OpenAI-compatible providers reject `toolChoice: "none"` with an auto-only 400. ([#4715](https://github.com/can1357/oh-my-pi/issues/4715)) +- Fixed generic remote compaction against OpenAI-compatible `/chat/completions` endpoints (for example llama.cpp `openai-completions`) by sending chat messages instead of the custom `{ systemPrompt, prompt }` summarizer payload. ([#4630](https://github.com/can1357/oh-my-pi/issues/4630)) + ## [16.3.7] - 2026-07-05 ### Fixed diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index 4c8f7b64f..c8b1b82e5 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -112,6 +112,26 @@ function hardToolChoiceBlocks(choice: ToolChoice | undefined, requiredTool: stri * tool's own window elapses. A cheap synchronous queue check; latency-bounded * at one tick. */ +/** + * Abort reason for a turn-wide interruption where only some tool calls caused + * the abort and sibling placeholders need neutral messages. + */ +export interface ToolScopedAbortReason { + readonly kind: "tool-scoped-abort"; + readonly message: string; + readonly toolCallMessages: Record; + readonly defaultToolCallMessage: string; +} + +/** Creates an abort reason that labels matching tool calls separately from siblings. */ +export function createToolScopedAbortReason( + message: string, + toolCallMessages: Record, + defaultToolCallMessage: string, +): ToolScopedAbortReason { + return { kind: "tool-scoped-abort", message, toolCallMessages, defaultToolCallMessage }; +} + const STEERING_INTERRUPT_POLL_MS = 250; class HarmonyLeakInterruption extends Error { @@ -173,6 +193,7 @@ function snapshotAssistantMessage(message: AssistantMessage): AssistantMessage { cost: { ...message.usage.cost }, }, disabledFeatures: message.disabledFeatures ? [...message.disabledFeatures] : undefined, + toolCallAbortMessages: message.toolCallAbortMessages ? { ...message.toolCallAbortMessages } : undefined, }; } @@ -918,9 +939,17 @@ async function runLoopBody( (c): c is ToolCallContent => c.type === "toolCall" && (c as CursorExecResolvedCarrier)[kCursorExecResolved] !== true, ); + // Provider-built aborted messages (stream error events) carry no + // per-tool labels; derive them from a tool-scoped abort signal so + // only the matching call is blamed and siblings stay neutral. + const scopedAbort = toolScopedAbortReason(signal); + const toolCallAbortMessages = + message.toolCallAbortMessages ?? + (scopedAbort ? buildToolCallAbortMessages(message, scopedAbort) : undefined); const toolResults: ToolResultMessage[] = []; for (const toolCall of toolCalls) { - const result = createAbortedToolResult(toolCall, stream, message.stopReason, message.errorMessage); + const errorMessage = toolCallAbortMessages?.[toolCall.id] ?? message.errorMessage; + const result = createAbortedToolResult(toolCall, stream, message.stopReason, errorMessage); currentContext.messages.push(result); newMessages.push(result); toolResults.push(result); @@ -1610,6 +1639,34 @@ function emitDiscardedHarmonyPartial( }); } +function isStringRecord(value: unknown): value is Record { + if (!value || typeof value !== "object" || Array.isArray(value)) return false; + return Object.values(value).every(child => typeof child === "string"); +} + +function toolScopedAbortReason(signal: AbortSignal | undefined): ToolScopedAbortReason | undefined { + const reason = signal?.reason; + if (!reason || typeof reason !== "object") return undefined; + if (Reflect.get(reason, "kind") !== "tool-scoped-abort") return undefined; + if (typeof Reflect.get(reason, "message") !== "string") return undefined; + if (typeof Reflect.get(reason, "defaultToolCallMessage") !== "string") return undefined; + return isStringRecord(Reflect.get(reason, "toolCallMessages")) ? reason : undefined; +} + +function buildToolCallAbortMessages( + message: AssistantMessage, + reason: ToolScopedAbortReason, +): Record | undefined { + let hasToolCall = false; + const messages: Record = {}; + for (const block of message.content) { + if (block.type !== "toolCall") continue; + hasToolCall = true; + messages[block.id] = reason.toolCallMessages[block.id] ?? reason.defaultToolCallMessage; + } + return hasToolCall ? messages : undefined; +} + /** Resolve the human-readable reason an abort carried. A caller that aborts via * `AbortController.abort(reason)` with a string or a non-`AbortError` `Error` * (e.g. the coding agent's user-interrupt label) gets that text surfaced on the @@ -1617,6 +1674,8 @@ function emitDiscardedHarmonyPartial( * `signal.reason` is the default `AbortError` `DOMException`) falls back to the * generic sentinel that downstream renderers treat as "no specific reason". */ export function abortReasonText(signal: AbortSignal | undefined): string { + const scopedReason = toolScopedAbortReason(signal); + if (scopedReason) return scopedReason.message; const reason = signal?.reason; if (typeof reason === "string" && reason.trim().length > 0) return reason; if (reason instanceof Error && reason.name !== "AbortError" && reason.message.trim().length > 0) { @@ -1665,6 +1724,11 @@ function emitAbortedAssistantMessage( // labeled user interrupt still surfaces through `errorMessage`, but partial // tool arguments are unsafe to keep and can carry incomplete provider IDs. const retained = retainCompletedToolCalls(base, completedToolCallIds); + const scopedAbort = toolScopedAbortReason(requestSignal); + const toolCallAbortMessages = scopedAbort ? buildToolCallAbortMessages(retained, scopedAbort) : undefined; + if (toolCallAbortMessages) { + retained.toolCallAbortMessages = toolCallAbortMessages; + } const abortedMessage = snapshotAssistantMessage(retained); if (addedPartial) { context.messages[context.messages.length - 1] = abortedMessage; diff --git a/packages/agent/src/compaction/compaction.ts b/packages/agent/src/compaction/compaction.ts index 9f502bac5..36b8feadd 100644 --- a/packages/agent/src/compaction/compaction.ts +++ b/packages/agent/src/compaction/compaction.ts @@ -698,6 +698,12 @@ function createSummarizationError(prefix: string, response: AssistantMessage): E return response.errorStatus === undefined ? new Error(text) : new ProviderHttpError(text, response.errorStatus); } +function shouldRetryHandoffWithAutoToolChoice(response: AssistantMessage): boolean { + if (response.errorStatus !== 400) return false; + const message = response.errorMessage ?? ""; + return /\btool_choice\b/i.test(message) && /\bauto\b/i.test(message) && /\bsupported\b/i.test(message); +} + /** * Generate a summary of the conversation using the LLM. * If previousSummary is provided, uses the update prompt to merge. @@ -813,14 +819,17 @@ export async function generateSummary( ]; if (options?.remoteEndpoint) { - const remote = await requestRemoteCompaction( - options.remoteEndpoint, - { - systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, - prompt: promptText, - }, - signal, - { fetch: options.fetch }, + const endpoint = options.remoteEndpoint; + const remote = await withAuth( + apiKey, + key => + requestRemoteCompaction( + endpoint, + { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, prompt: promptText }, + signal, + { fetch: options.fetch, model, apiKey: key }, + ), + { signal, missingKeyMessage: "Remote compaction credentials unavailable" }, ); return remote.summary; } @@ -917,24 +926,33 @@ export interface HandoffFromContextOptions { * `streamOptions` that mirror the live turn's cache routing. That keeps the * cache-preserving context construction in the host (which owns the transform * pipeline) while this function centralizes the handoff request contract: - * `toolChoice: "none"`, clamped reasoning effort, oneshot telemetry, text-only - * extraction, and provider-error mapping. + * cache-first `toolChoice: "none"`, clamped reasoning effort, one retry for + * auto-only `tool_choice` providers, oneshot telemetry, text-only extraction, + * and provider-error mapping. */ export async function generateHandoffFromContext( context: Context, model: Model, options: HandoffFromContextOptions, ): Promise { - const response = await instrumentedCompleteSimple( - model, - context, - { - ...options.streamOptions, - reasoning: resolveCompactionEffort(model, options.thinkingLevel), - toolChoice: "none", - }, - { telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl }, - ); + const requestOptions = { + ...options.streamOptions, + reasoning: resolveCompactionEffort(model, options.thinkingLevel), + toolChoice: "none" as const, + }; + let response = await instrumentedCompleteSimple(model, context, requestOptions, { + telemetry: options.telemetry, + oneshotKind: "handoff", + completeImpl: options.completeImpl, + }); + if (response.stopReason === "error" && shouldRetryHandoffWithAutoToolChoice(response)) { + response = await instrumentedCompleteSimple( + model, + context, + { ...requestOptions, toolChoice: "auto" }, + { telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl }, + ); + } if (response.stopReason === "error") { throw createSummarizationError("Handoff generation failed", response); @@ -1001,14 +1019,17 @@ async function generateShortSummary( promptText += SHORT_SUMMARY_PROMPT; if (options?.remoteEndpoint) { - const remote = await requestRemoteCompaction( - options.remoteEndpoint, - { - systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, - prompt: promptText, - }, - signal, - { fetch: options?.fetch }, + const endpoint = options.remoteEndpoint; + const remote = await withAuth( + apiKey, + key => + requestRemoteCompaction( + endpoint, + { systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, prompt: promptText }, + signal, + { fetch: options?.fetch, model, apiKey: key }, + ), + { signal, missingKeyMessage: "Remote compaction credentials unavailable" }, ); return remote.summary; } diff --git a/packages/agent/src/compaction/openai.ts b/packages/agent/src/compaction/openai.ts index 5a65d7507..2695e4466 100644 --- a/packages/agent/src/compaction/openai.ts +++ b/packages/agent/src/compaction/openai.ts @@ -547,16 +547,55 @@ export async function requestOpenAiRemoteCompaction( return { provider: model.provider, replacementHistory, compactionItem }; } +/** + * Generic remote-compaction POST. Two wire shapes are auto-selected by + * endpoint suffix so a single `compaction.remoteEndpoint` setting can point at + * either a purpose-built omp summarizer (`{systemPrompt, prompt}` → `{summary}`) + * or any OpenAI-compatible chat-completions server (`/chat/completions`, + * `/v1/chat/completions`, …) as reported for llama.cpp / vLLM / etc. in + * issue #4630: without this, the omp payload was rejected with + * HTTP 400 `"'messages' is required"`, compaction silently fell back to + * local summarization, and context grew unbounded. + * + * When `context.model` is provided the chat-completions body is tagged with + * that model's wire id (llama.cpp requires the field) and `context.apiKey` is + * forwarded as `Authorization: Bearer`. Callers wrap this in `withAuth` so + * 401s force-refresh through the standard credential rotation policy. + */ export async function requestRemoteCompaction( endpoint: string, request: RemoteCompactionRequest, signal?: AbortSignal, - opts?: { fetch?: FetchImpl; timeoutMs?: number }, + opts?: { fetch?: FetchImpl; timeoutMs?: number; model?: Model; apiKey?: string }, ): Promise { + let endpointPath = endpoint; + try { + endpointPath = new URL(endpoint).pathname; + } catch { + // Keep the raw endpoint for relative/custom fetch implementations. + } + const isChatCompletions = /\/chat\/completions\/?$/.test(endpointPath); + const headers: Record = { "content-type": "application/json" }; + if (isChatCompletions) { + if (opts?.apiKey) headers.Authorization = `Bearer ${opts.apiKey}`; + if (opts?.model?.headers) Object.assign(headers, opts.model.headers); + } + + const body: Record = isChatCompletions + ? { + model: opts?.model ? resolveOpenAiCompactModel(opts.model) : undefined, + messages: [ + { role: "system", content: request.systemPrompt }, + { role: "user", content: request.prompt }, + ], + stream: false, + } + : { systemPrompt: request.systemPrompt, prompt: request.prompt }; + const response = await (opts?.fetch ?? fetch)(endpoint, { method: "POST", - headers: { "content-type": "application/json" }, - body: JSON.stringify(request), + headers, + body: JSON.stringify(body), signal: withRequestTimeout(signal, opts?.timeoutMs ?? REMOTE_COMPACTION_TIMEOUT_MS), }); @@ -577,6 +616,31 @@ export async function requestRemoteCompaction( ); } + if (isChatCompletions) { + type ChatCompletionsResponse = { + choices?: Array<{ + message?: { + content?: string | Array<{ type?: string; text?: string }> | null; + }; + }>; + }; + const data = (await response.json()) as ChatCompletionsResponse | undefined; + const choice = data?.choices?.[0]?.message?.content; + let summary: string | undefined; + if (typeof choice === "string") { + summary = choice; + } else if (Array.isArray(choice)) { + summary = choice + .filter((part): part is { type?: string; text: string } => typeof part?.text === "string") + .map(part => part.text) + .join(""); + } + if (typeof summary !== "string" || summary.length === 0) { + throw new Error("Remote compaction response missing choices[0].message.content"); + } + return { summary }; + } + const data = (await response.json()) as RemoteCompactionResponse | undefined; if (!data || typeof data.summary !== "string") { throw new Error("Remote compaction response missing summary"); diff --git a/packages/agent/test/handoff.test.ts b/packages/agent/test/handoff.test.ts index a3bdc293c..cf89f4344 100644 --- a/packages/agent/test/handoff.test.ts +++ b/packages/agent/test/handoff.test.ts @@ -9,7 +9,7 @@ import { import { ThinkingLevel } from "@oh-my-pi/pi-agent-core/thinking"; import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai"; import * as ai from "@oh-my-pi/pi-ai"; -import { Effort } from "@oh-my-pi/pi-ai"; +import { Effort, z } from "@oh-my-pi/pi-ai"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; function createAssistantMessage(content: AssistantMessage["content"]): AssistantMessage { @@ -32,6 +32,28 @@ function createAssistantMessage(content: AssistantMessage["content"]): Assistant }; } +function createAssistantError(errorStatus: number, errorMessage: string): AssistantMessage { + return { + ...createAssistantMessage([]), + stopReason: "error", + errorStatus, + errorMessage, + }; +} + +const handoffToolSchema = z.object({ note: z.string().optional() }); + +function createHandoffTool(): AgentTool { + return { + name: "handoff_probe", + label: "Handoff Probe", + description: "Confirms handoff requests keep live tools available.", + parameters: handoffToolSchema, + intent: "omit", + execute: async () => ({ content: [{ type: "text", text: "ok" }], details: {} }), + }; +} + function getTestModel(): Model { const model = getBundledModel("anthropic", "claude-sonnet-4-5"); if (!model) { @@ -155,4 +177,99 @@ describe("handoff helpers", () => { reasoning: Effort.Medium, }); }); + + test("generateHandoffFromContext retries auto-only tool_choice rejection with live tools", async () => { + const completeSimpleSpy = vi + .spyOn(ai, "completeSimple") + .mockResolvedValueOnce( + createAssistantError( + 400, + "400 Bad Request: Only a tool_choice of 'auto' is supported for this model; param=tool_choice", + ), + ) + .mockResolvedValueOnce(createAssistantMessage([{ type: "text", text: "## Goal\nRecovered on retry" }])); + const model = getTestModel(); + const tools = [createHandoffTool()]; + const context = { + systemPrompt: ["Live system prompt"], + tools, + messages: [{ role: "user" as const, content: "prepare handoff", timestamp: 1 }], + }; + + const document = await generateHandoffFromContext(context, model, { + streamOptions: { + apiKey: "test-key", + sessionId: "sess-auto-only:side:42", + promptCacheKey: "sess-auto-only", + }, + thinkingLevel: ThinkingLevel.Medium, + }); + + expect(document).toBe("## Goal\nRecovered on retry"); + expect(completeSimpleSpy).toHaveBeenCalledTimes(2); + const firstCall = completeSimpleSpy.mock.calls[0]; + const secondCall = completeSimpleSpy.mock.calls[1]; + if (!firstCall) throw new Error("Expected initial completeSimple call"); + if (!secondCall) throw new Error("Expected retry completeSimple call"); + const [firstModel, firstContext, firstOptions] = firstCall; + const [secondModel, secondContext, secondOptions] = secondCall; + expect(firstModel).toBe(model); + expect(secondModel).toBe(model); + expect(firstContext).toBe(context); + expect(secondContext).toBe(context); + expect(firstContext.tools).toBe(tools); + expect(secondContext.tools).toBe(tools); + expect(firstOptions).toMatchObject({ + apiKey: "test-key", + sessionId: "sess-auto-only:side:42", + promptCacheKey: "sess-auto-only", + toolChoice: "none", + reasoning: Effort.Medium, + }); + expect(secondOptions).toMatchObject({ + apiKey: "test-key", + sessionId: "sess-auto-only:side:42", + promptCacheKey: "sess-auto-only", + toolChoice: "auto", + reasoning: Effort.Medium, + }); + }); + + test("generateHandoffFromContext surfaces unrelated provider 400 without retrying", async () => { + const completeSimpleSpy = vi + .spyOn(ai, "completeSimple") + .mockResolvedValueOnce(createAssistantError(400, "400 Bad Request: unsupported max_tokens; param=max_tokens")); + const model = getTestModel(); + const tools = [createHandoffTool()]; + const context = { + systemPrompt: ["Live system prompt"], + tools, + messages: [{ role: "user" as const, content: "prepare handoff", timestamp: 1 }], + }; + + const error = await generateHandoffFromContext(context, model, { + streamOptions: { + apiKey: "test-key", + sessionId: "sess-unrelated-400:side:42", + promptCacheKey: "sess-unrelated-400", + }, + thinkingLevel: ThinkingLevel.Medium, + }).catch((caught: unknown) => caught); + + if (!(error instanceof Error)) throw new Error("Expected handoff generation to reject"); + expect(error.message).toContain("unsupported max_tokens"); + expect(completeSimpleSpy).toHaveBeenCalledTimes(1); + const call = completeSimpleSpy.mock.calls[0]; + if (!call) throw new Error("Expected completeSimple call"); + const [, calledContext, options] = call; + expect(calledContext).toBe(context); + expect(calledContext.tools).toBe(tools); + expect(options).toMatchObject({ + apiKey: "test-key", + sessionId: "sess-unrelated-400:side:42", + promptCacheKey: "sess-unrelated-400", + toolChoice: "none", + reasoning: Effort.Medium, + }); + }); }); diff --git a/packages/agent/test/remote-compaction.test.ts b/packages/agent/test/remote-compaction.test.ts index d88104b18..c2cc50dd1 100644 --- a/packages/agent/test/remote-compaction.test.ts +++ b/packages/agent/test/remote-compaction.test.ts @@ -13,6 +13,7 @@ import { getCompactionV2PreserveData, requestCompactionV2Streaming, requestOpenAiRemoteCompaction, + requestRemoteCompaction, shouldUseCompactionV2Streaming, shouldUseOpenAiRemoteCompaction, } from "@oh-my-pi/pi-agent-core/compaction/openai"; @@ -540,6 +541,76 @@ describe("requestOpenAiRemoteCompaction timeout", () => { }); }); +describe("requestRemoteCompaction wire formats", () => { + test("uses OpenAI chat completions format for /chat/completions endpoints", async () => { + const model = buildModel({ + id: "catalog-selection-id", + name: "Qwopus 3.6 35B-A3B Coder", + requestModelId: "provider-wire-id", + remoteCompaction: { model: "provider-compact-wire-id" }, + api: "openai-completions", + provider: "local-llama", + baseUrl: "http://127.0.0.1:8001/v1", + headers: { "x-local-llama": "1" }, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 131072, + maxTokens: 4096, + }); + let sentBody: unknown; + const fetchMock: FetchImpl = async (_input, init) => { + if (typeof init?.body !== "string") throw new Error("missing remote compaction request body"); + sentBody = JSON.parse(init.body) as unknown; + const headers = new Headers(init.headers); + expect(headers.get("authorization")).toBe("Bearer local-key"); + expect(headers.get("x-local-llama")).toBe("1"); + return new Response(JSON.stringify({ choices: [{ message: { content: "remote summary" } }] }), { + headers: { "content-type": "application/json" }, + }); + }; + + const result = await requestRemoteCompaction( + "http://127.0.0.1:8001/v1/chat/completions", + { systemPrompt: "summarize", prompt: "hello" }, + undefined, + { fetch: fetchMock, model, apiKey: "local-key" }, + ); + + expect(result).toEqual({ summary: "remote summary" }); + expect(sentBody).toEqual({ + model: "provider-compact-wire-id", + messages: [ + { role: "system", content: "summarize" }, + { role: "user", content: "hello" }, + ], + stream: false, + }); + }); + + test("keeps the generic omp summarizer format for other endpoints", async () => { + let sentBody: unknown; + const fetchMock: FetchImpl = async (_input, init) => { + if (typeof init?.body !== "string") throw new Error("missing remote compaction request body"); + sentBody = JSON.parse(init.body) as unknown; + expect(new Headers(init.headers).get("authorization")).toBeNull(); + return new Response(JSON.stringify({ summary: "generic summary", shortSummary: "generic" }), { + headers: { "content-type": "application/json" }, + }); + }; + + const result = await requestRemoteCompaction( + "https://compaction.example.test/summarize", + { systemPrompt: "summarize", prompt: "hello" }, + undefined, + { fetch: fetchMock, apiKey: "unused-for-generic" }, + ); + + expect(result).toEqual({ summary: "generic summary", shortSummary: "generic" }); + expect(sentBody).toEqual({ systemPrompt: "summarize", prompt: "hello" }); + }); +}); + describe("compact() remote compaction failure handling", () => { afterEach(() => { vi.restoreAllMocks(); @@ -771,6 +842,53 @@ describe("compact() remote compaction failure handling", () => { expect(completeSpy).not.toHaveBeenCalled(); }); + test("uses configured chat completions endpoints for openai-completions remote compaction", async () => { + const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local fallback")); + const preparation = makePreparation(); + preparation.settings = { + ...preparation.settings, + remoteEndpoint: "http://127.0.0.1:8001/v1/chat/completions", + remoteStreamingV2Enabled: false, + }; + const model = buildModel({ + id: "catalog-selection-id", + name: "Qwopus 3.6 35B-A3B Coder", + requestModelId: "provider-wire-id", + api: "openai-completions", + provider: "local-llama", + baseUrl: "http://127.0.0.1:8001/v1", + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 131072, + maxTokens: 4096, + }); + const requestBodies: unknown[] = []; + const fetchMock: FetchImpl = async (_input, init) => { + if (typeof init?.body !== "string") throw new Error("missing remote compaction request body"); + requestBodies.push(JSON.parse(init.body) as unknown); + expect(new Headers(init.headers).get("authorization")).toBe("Bearer local-key"); + const summary = requestBodies.length === 1 ? "remote history summary" : "remote short summary"; + return new Response(JSON.stringify({ choices: [{ message: { content: summary } }] }), { + headers: { "content-type": "application/json" }, + }); + }; + + const result = await compact(preparation, model, "local-key", undefined, undefined, { + fetch: fetchMock, + }); + + expect(result.summary).toContain("remote history summary"); + expect(result.shortSummary).toBe("remote short summary"); + expect(completeSpy).not.toHaveBeenCalled(); + expect(requestBodies).toHaveLength(2); + expect(requestBodies[0]).toMatchObject({ + model: "provider-wire-id", + messages: [{ role: "system" }, { role: "user", content: expect.stringContaining("long history") }], + stream: false, + }); + }); + test("remote compact server failure without abort still falls back to local summarization", async () => { const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local summary")); const fetchMock: FetchImpl = async () => diff --git a/packages/ai/CHANGELOG.md b/packages/ai/CHANGELOG.md index 9c99508fd..a72f0fc1e 100644 --- a/packages/ai/CHANGELOG.md +++ b/packages/ai/CHANGELOG.md @@ -2,6 +2,22 @@ ## [Unreleased] +### Added + +- Added `AssistantMessage.toolCallAbortMessages` for per-tool placeholder labels on aborted assistant turns ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). + +### Fixed + +- Fixed Anthropic replay 400s (`tool_use ids were found without tool_result blocks immediately after`) when a persisted assistant turn carries content after a completed tool call — such as a mid-turn `server-side-fallback` handoff (fallback block plus continued text/tool calls after the primary model's `tool_use`) or trailing text from cross-provider replays — by stable-partitioning assistant content so all `tool_use` blocks trail the non-`tool_use` chain. ([#4781](https://github.com/can1357/oh-my-pi/issues/4781), [#544](https://github.com/can1357/oh-my-pi/issues/544)) +- Fixed access-token-only OAuth credentials attempting token refresh with an empty refresh token after expiry. +- Fixed gateway usage-limit retries falling through to cross-provider model fallback before trying a sibling credential from the same provider. +- Fixed Codex usage-limit rotation treating Plus and K-12 accounts as separate quota groups for shared 5-hour/7-day windows. +- Fixed OpenAI Responses streams that end with `response.done` being misclassified as premature stream closures. +- Fixed OpenCode Go `/login` credentials being shadowed by an existing `OPENCODE_API_KEY` env fallback after switching accounts. ([#4688](https://github.com/can1357/oh-my-pi/issues/4688)) +- Fixed OpenAI Codex WebSocket continuations to treat proxy stale-anchor codes such as `codex_previous_response_stale` as an expired `previous_response_id` chain — same recovery class as the OpenAI-standard `previous_response_not_found` — so the turn is retried with full context instead of surfacing the error to the user ([#4624](https://github.com/can1357/oh-my-pi/issues/4624)). +- Fixed Azure Foundry Anthropic utility requests to omit the structured-output beta whenever strict tools are disabled, preventing `structured_outputs not supported in your workspace` failures for Sonnet 5 compaction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). +- Fixed OAuth `launchUrl` advertisement for flows whose redirect never returns to the local callback server: custom-scheme redirects (e.g. GitLab Duo's `vscode://` URI, which `new URL` parses without complaint) and fixed non-loopback hosts no longer receive a `http://localhost:/launch` copy target that misrepresents the callback endpoint and resolves nowhere for remote users. + ## [16.3.11] - 2026-07-06 ### Fixed diff --git a/packages/ai/src/auth-gateway/server.ts b/packages/ai/src/auth-gateway/server.ts index 3caf4808e..c7eff02f4 100644 --- a/packages/ai/src/auth-gateway/server.ts +++ b/packages/ai/src/auth-gateway/server.ts @@ -233,6 +233,7 @@ async function refreshGatewayApiKeyAfterAuthError( retryAfterMs, baseUrl: model.baseUrl, modelId: model.id, + apiKey: oldKey, signal, }); logger.debug("auth-gateway retrying provider request after usage-limit block", { diff --git a/packages/ai/src/auth-retry.ts b/packages/ai/src/auth-retry.ts index dc35d6a1c..22a8f8e55 100644 --- a/packages/ai/src/auth-retry.ts +++ b/packages/ai/src/auth-retry.ts @@ -1,6 +1,7 @@ import type { OAuthAccess } from "./auth-storage"; import * as AIError from "./error"; import { isAuthRetryableError } from "./error/auth-classify"; +import { isUsageLimit } from "./error/flags"; /** * Context passed to an {@link ApiKeyResolver} on each resolution attempt. @@ -23,6 +24,8 @@ export interface ApiKeyResolveContext { lastChance: boolean; /** The auth error that triggered this re-resolution, or `undefined` on the initial resolve. */ error: unknown; + /** Bearer used by the failed attempt, when the caller can expose it. */ + previousKey?: string; /** Caller cancel signal, threaded into any credential refresh / rotation work. */ signal?: AbortSignal; } @@ -87,9 +90,11 @@ export async function resolveRetryKey( lastChance: boolean, error: unknown, signal?: AbortSignal, + previousKey?: string, ): Promise { try { - return (await resolver({ lastChance, error, signal })) || undefined; + const rotateSibling = lastChance || (!lastChance && isUsageLimit(error)); + return (await resolver({ lastChance: rotateSibling, error, signal, previousKey })) || undefined; } catch { return undefined; } @@ -136,7 +141,7 @@ export async function withAuth( } for (let i = 0; i < AUTH_RETRY_STEPS.length; i++) { - const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal); + const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal, lastKey); if (nextKey === undefined || nextKey === lastKey) continue; lastKey = nextKey; try { diff --git a/packages/ai/src/auth-storage.ts b/packages/ai/src/auth-storage.ts index a790ce5d5..3076276fd 100644 --- a/packages/ai/src/auth-storage.ts +++ b/packages/ai/src/auth-storage.ts @@ -66,6 +66,7 @@ const USAGE_RANKING_METRIC_EPSILON = 1e-9; export type ApiKeyCredential = { type: "api_key"; key: string; + source?: "login"; }; export type OAuthCredential = { @@ -1390,12 +1391,14 @@ export class AuthStorage { const credentialId = this.#getStoredCredentials(provider)[credentialIndex]?.id; if (credentialId === undefined) return blockedUntil; - const persistedGlobalBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, ""); - if ( - persistedGlobalBlockedUntil !== undefined && - (blockedUntil === undefined || persistedGlobalBlockedUntil > blockedUntil) - ) { - blockedUntil = persistedGlobalBlockedUntil; + if (!blockScope || provider !== "openai-codex") { + const persistedGlobalBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, ""); + if ( + persistedGlobalBlockedUntil !== undefined && + (blockedUntil === undefined || persistedGlobalBlockedUntil > blockedUntil) + ) { + blockedUntil = persistedGlobalBlockedUntil; + } } if (blockScope) { const persistedScopedBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, blockScope); @@ -1554,13 +1557,14 @@ export class AuthStorage { provider: string, type: T, sessionId?: string, + filter?: (credential: AuthCredential) => boolean, ): { credential: Extract; index: number } | undefined { const credentials = this.#getCredentialsForProvider(provider) .map((credential, index) => ({ credential, index })) - .filter( - (entry): entry is { credential: Extract; index: number } => - entry.credential.type === type, - ); + .filter((entry): entry is { credential: Extract; index: number } => { + if (entry.credential.type !== type) return false; + return filter?.(entry.credential) ?? true; + }); if (credentials.length === 0) return undefined; if (credentials.length === 1) return credentials[0]; @@ -1851,8 +1855,8 @@ export class AuthStorage { /** * Classify where a provider's auth comes from, following the same precedence * as {@link AuthStorage.getApiKey}: runtime override → config override → - * stored OAuth → env var → stored api_key → fallback resolver. Returns - * undefined when no auth is configured. + * stored OAuth → login-stored api_key → env var → stored api_key → + * fallback resolver. Returns undefined when no auth is configured. * * Compact, structured counterpart to {@link describeCredentialSource}. */ @@ -1861,6 +1865,9 @@ export class AuthStorage { if (this.#configOverrides.has(provider)) return { kind: "config" }; const stored = this.#getCredentialsForProvider(provider); if (stored.some(credential => credential.type === "oauth")) return { kind: "oauth" }; + if (stored.some(credential => credential.type === "api_key" && credential.source === "login")) { + return { kind: "api_key" }; + } if (getEnvApiKey(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) }; if (stored.some(credential => credential.type === "api_key")) return { kind: "api_key" }; if (this.#fallbackResolver?.(provider)) return { kind: "fallback" }; @@ -2005,7 +2012,7 @@ export class AuthStorage { if (!result) { return; } - const newCredential: ApiKeyCredential = { type: "api_key", key: result }; + const newCredential: ApiKeyCredential = { type: "api_key", key: result, source: "login" }; const stored = this.#store.upsertAuthCredentialRemote ? await this.#store.upsertAuthCredentialRemote(provider, newCredential) : this.#store.upsertAuthCredentialForProvider(provider, newCredential); @@ -3057,9 +3064,20 @@ export class AuthStorage { async markUsageLimitReached( provider: string, sessionId: string | undefined, - options?: { retryAfterMs?: number; baseUrl?: string; modelId?: string; signal?: AbortSignal }, + options?: { retryAfterMs?: number; baseUrl?: string; modelId?: string; apiKey?: string; signal?: AbortSignal }, ): Promise { - const sessionCredential = this.#getSessionCredential(provider, sessionId); + let sessionCredential: { type: AuthCredential["type"]; index: number } | undefined; + if (options?.apiKey) { + const stored = this.#getStoredCredentials(provider); + for (let index = 0; index < stored.length; index++) { + const entry = stored[index]; + if (entry && (await this.#credentialMatchesApiKey(entry.credential, options.apiKey))) { + sessionCredential = { type: entry.credential.type, index }; + break; + } + } + } + sessionCredential ??= this.#getSessionCredential(provider, sessionId); if (!sessionCredential) return { switched: false }; const providerKey = this.#getProviderTypeKey(provider, sessionCredential.type); @@ -3387,12 +3405,21 @@ export class AuthStorage { const checkUsage = strategy !== undefined && (credentials.length > 1 || requiresProModel); const sessionCredential = this.#getSessionCredential(provider, sessionId); const sessionPreferredIndex = sessionCredential?.type === "oauth" ? sessionCredential.index : undefined; + const sessionPreferredCredential = + sessionPreferredIndex !== undefined + ? credentials.find(entry => entry.index === sessionPreferredIndex)?.credential + : undefined; + const sessionPreferredCanRefreshOrUse = + sessionPreferredCredential !== undefined && + (sessionPreferredCredential.refresh.trim().length > 0 || + Date.now() + OAUTH_REFRESH_SKEW_MS < sessionPreferredCredential.expires); // Skip ranking only when the session already has a working preferred credential — re-ranking // mid-session causes account switches that cold-start the server-side prompt cache. New sessions // (no preference) and sessions whose preferred is blocked still rank, so we pick the account // with the most headroom proactively and fall back intelligently when rate-limited. const sessionPreferredIsAvailable = sessionPreferredIndex !== undefined && + sessionPreferredCanRefreshOrUse && !this.#isCredentialBlocked(provider, providerKey, sessionPreferredIndex, blockScope); const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel); const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order; @@ -3899,8 +3926,8 @@ export class AuthStorage { return configKey; } - // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored - // static api_key (which may be a stale broker-migrated copy) as a last resort. + // Precedence: a deliberate OAuth/login credential wins, then an explicit env var, + // then a stored static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthSelection = this.#selectCredentialByType(provider, "oauth"); if (oauthSelection) { const expiresAt = oauthSelection.credential.expires; @@ -3916,6 +3943,16 @@ export class AuthStorage { } } + const loginApiKeySelection = this.#selectCredentialByType( + provider, + "api_key", + undefined, + credential => credential.type === "api_key" && credential.source === "login", + ); + if (loginApiKeySelection) { + return this.#configValueResolver(loginApiKeySelection.credential.key); + } + const envKey = getEnvApiKey(provider); if (envKey) return envKey; @@ -3933,9 +3970,10 @@ export class AuthStorage { * 1. Runtime override (CLI --api-key) * 2. Config override (models.yml `providers..apiKey`) * 3. OAuth token from storage (auto-refreshed) - * 4. Environment variable - * 5. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins - * 6. Fallback resolver (models.yml custom providers, last-resort) + * 4. API key persisted by a successful `/login` + * 5. Environment variable + * 6. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins + * 7. Fallback resolver (models.yml custom providers, last-resort) */ async getApiKey(provider: string, sessionId?: string, options?: AuthApiKeyOptions): Promise { // Runtime override takes highest priority @@ -3954,13 +3992,24 @@ export class AuthStorage { return configKey; } - // Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored - // static api_key (which may be a stale broker-migrated copy) as a last resort. + // Precedence: a deliberate OAuth/login credential wins, then an explicit env var, + // then a stored static api_key (which may be a stale broker-migrated copy) as a last resort. const oauthResolved = await this.#resolveOAuthSelection(provider, sessionId, options); if (oauthResolved) { return oauthResolved.apiKey; } + const loginApiKeySelection = this.#selectCredentialByType( + provider, + "api_key", + sessionId, + credential => credential.type === "api_key" && credential.source === "login", + ); + if (loginApiKeySelection) { + this.#recordSessionCredential(provider, sessionId, "api_key", loginApiKeySelection.index); + return this.#configValueResolver(loginApiKeySelection.credential.key); + } + // Past OAuth: the session sticky (if any) is stale — the request authenticates via // env/api_key/fallback, not OAuth, so clear it now so getOAuthAccountId() correctly // suppresses account_uuid for this session. @@ -3969,7 +4018,12 @@ export class AuthStorage { const envKey = getEnvApiKey(provider); if (envKey) return envKey; - const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId); + const apiKeySelection = this.#selectCredentialByType( + provider, + "api_key", + sessionId, + credential => credential.type !== "api_key" || credential.source !== "login", + ); if (apiKeySelection) { this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index); return this.#configValueResolver(apiKeySelection.credential.key); @@ -4381,7 +4435,7 @@ export class AuthStorage { async rotateSessionCredential( provider: string, sessionId: string | undefined, - options?: { error?: unknown; modelId?: string; signal?: AbortSignal }, + options?: { error?: unknown; modelId?: string; apiKey?: string; signal?: AbortSignal }, ): Promise { const sessionCredential = this.#getSessionCredential(provider, sessionId); if (!sessionCredential) return false; @@ -4393,6 +4447,7 @@ export class AuthStorage { return ( await this.markUsageLimitReached(provider, sessionId, { modelId: options?.modelId, + apiKey: options?.apiKey, signal: options?.signal, }) ).switched; @@ -4674,9 +4729,10 @@ export class AuthStorage { * 1. Runtime override (`--api-key`). * 2. Config override (`models.yml` `providers..apiKey`). * 3. Stored OAuth credential. - * 4. Env var — overrides a stored static api_key (e.g. a stale broker copy). - * 5. Stored api_key credential. - * 6. Fallback resolver. + * 4. API key persisted by a successful `/login`. + * 5. Env var — overrides a stored static api_key (e.g. a stale broker copy). + * 6. Stored api_key credential. + * 7. Fallback resolver. * * The string is purely informational; consumers must not parse it. */ @@ -4691,14 +4747,16 @@ export class AuthStorage { const baseLabel = this.#sourceLabel ?? "local store"; const stored = this.#getStoredCredentials(provider); const session = sessionId ? this.#sessionLastCredential.get(provider)?.get(sessionId) : undefined; - // Describe the stored credential of a given type, honoring the session sticky index. - const describeStored = (type: AuthCredential["type"]): string | undefined => { + const describeStored = ( + type: AuthCredential["type"], + filter?: (credential: AuthCredential) => boolean, + ): string | undefined => { const typed = stored .map((entry, index) => ({ entry, index })) - .filter(({ entry }) => entry.credential.type === type); + .filter(({ entry }) => entry.credential.type === type && (filter?.(entry.credential) ?? true)); if (typed.length === 0) return undefined; - const index = session?.type === type ? session.index : typed[0].index; - const chosen = stored[index] ?? typed[0].entry; + const sticky = session?.type === type ? typed.find(entry => entry.index === session.index) : undefined; + const chosen = sticky?.entry ?? typed[0].entry; const credential = chosen.credential; const identity = credential.type === "oauth" @@ -4707,11 +4765,19 @@ export class AuthStorage { return `${baseLabel} · ${type} #${chosen.id} (${identity})`; }; - // A deliberate OAuth login wins; then an explicit env var; then a stored static api_key. + // Deliberate login credentials win; then an explicit env var; then a stored static api_key. const oauthSource = describeStored("oauth"); if (oauthSource) return oauthSource; + const loginApiKeySource = describeStored( + "api_key", + credential => credential.type === "api_key" && credential.source === "login", + ); + if (loginApiKeySource) return loginApiKeySource; if (getEnvApiKey(provider)) return `env (over ${baseLabel})`; - const apiKeySource = describeStored("api_key"); + const apiKeySource = describeStored( + "api_key", + credential => credential.type !== "api_key" || credential.source !== "login", + ); if (apiKeySource) return apiKeySource; if (this.#fallbackResolver?.(provider) !== undefined) return "fallback resolver"; return undefined; @@ -4776,9 +4842,10 @@ function normalizeStoredIdentityKey(identityKey: string | null | undefined): str function serializeCredential(provider: string, credential: AuthCredential): SerializedCredentialRecord | null { if (credential.type === "api_key") { + const data = credential.source === "login" ? { key: credential.key, source: "login" } : { key: credential.key }; return { credentialType: "api_key", - data: JSON.stringify({ key: credential.key }), + data: JSON.stringify(data), identityKey: null, }; } @@ -4806,7 +4873,8 @@ function deserializeCredential(row: AuthRow): AuthCredential | null { if (row.credential_type === "api_key") { const data = parsed as Record; if (typeof data.key === "string") { - return { type: "api_key", key: data.key }; + const source = data.source === "login" ? "login" : undefined; + return source ? { type: "api_key", key: data.key, source } : { type: "api_key", key: data.key }; } } if (row.credential_type === "oauth") { diff --git a/packages/ai/src/providers/anthropic.ts b/packages/ai/src/providers/anthropic.ts index 5093c7795..6b3570c23 100644 --- a/packages/ai/src/providers/anthropic.ts +++ b/packages/ai/src/providers/anthropic.ts @@ -127,12 +127,13 @@ export function buildBetaHeader(baseBetas: readonly string[], extraBetas: readon const midConversationSystemBeta = "mid-conversation-system-2026-04-07"; const contextManagementBeta = "context-management-2025-06-27"; +const structuredOutputsBeta = "structured-outputs-2025-12-15"; const claudeCodeUtilityBetaDefaults = [ "oauth-2025-04-20", "interleaved-thinking-2025-05-14", contextManagementBeta, "prompt-caching-scope-2026-01-05", - "structured-outputs-2025-12-15", + structuredOutputsBeta, ] as const; const claudeCodeAgentBetaDefaults = [ "claude-code-20250219", @@ -159,10 +160,12 @@ function buildClaudeCodeBetas( agentRequest: boolean, thinkingRequest: boolean, redactThinking: boolean, + disableStrictTools = false, ): readonly string[] { - if (!agentRequest && !redactThinking) return claudeCodeUtilityBetaDefaults; + if (!agentRequest && !redactThinking && !disableStrictTools) return claudeCodeUtilityBetaDefaults; const betas: string[] = []; for (const beta of agentRequest ? claudeCodeAgentBetaDefaults : claudeCodeUtilityBetaDefaults) { + if (disableStrictTools && beta === structuredOutputsBeta) continue; betas.push(beta); // Match CC's header order: redact-thinking immediately follows interleaved-thinking. if (redactThinking && beta === interleavedThinkingBeta) betas.push(redactThinkingBeta); @@ -1098,6 +1101,7 @@ export type AnthropicClientOptionsArgs = { hasTools?: boolean; thinkingEnabled?: boolean; thinkingDisplay?: AnthropicThinkingDisplay; + disableStrictTools?: boolean; fetch?: FetchImpl; claudeCodeSessionId?: string; }; @@ -1833,6 +1837,7 @@ const streamAnthropicOnce = ( thinkingDisplay: options?.thinkingDisplay, fetch: options?.fetch, claudeCodeSessionId: options?.sessionId ?? extractClaudeMetadataSessionId(options?.metadata?.user_id), + disableStrictTools, }); client = created.client; isOAuthToken = created.isOAuthToken; @@ -2648,8 +2653,10 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A thinkingDisplay, isOAuth, claudeCodeSessionId, + disableStrictTools: disableStrictToolsOverride, } = args; const compat = model.compat; + const disableStrictTools = disableStrictToolsOverride ?? compat.disableStrictTools; const needsInterleavedBeta = interleavedThinking && !model.thinking?.supportsDisplay; const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming; const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey); @@ -2722,7 +2729,12 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A isCloudflareAiGateway: model.provider === "cloudflare-ai-gateway", claudeCodeSessionId, claudeCodeBetas: oauthToken - ? buildClaudeCodeBetas(hasTools || thinkingEnabled, thinkingEnabled, thinkingDisplay === "omitted") + ? buildClaudeCodeBetas( + hasTools || thinkingEnabled, + thinkingEnabled, + thinkingDisplay === "omitted", + disableStrictTools, + ) : [], }); @@ -3529,6 +3541,40 @@ export function convertAnthropicMessages( }); } } + // Anthropic's replay validator rejects any non-`tool_use` block that + // appears after a `tool_use` inside an assistant turn (400: + // "tool_use ids were found without tool_result blocks immediately + // after: "). A persisted turn can violate this when a mid-turn + // server-side fallback handoff lands after the primary model already + // emitted a tool_use — the replayed content is then e.g. + // [thinking, text, tool_use, fallback, text, tool_use] — and also for + // the older cross-provider [text, tool_use, text] shape (issue #544). + // Stable-partition into [...non-tool_use, ...tool_use], preserving each + // side's relative order: the non-tool_use chain (thinking → text → + // fallback → text) carries thinking signatures and the fallback + // boundary marker whose order Anthropic verifies, while tool_use blocks + // are unsigned and safe to defer to the tail. Fast-path untouched when + // already in order so prompt-cache prefixes stay byte-identical. + let sawToolUse = false; + let needsPartition = false; + for (const block of blocks) { + if (block.type === "tool_use") { + sawToolUse = true; + } else if (sawToolUse) { + needsPartition = true; + break; + } + } + if (needsPartition) { + const nonToolUse: ContentBlockParam[] = []; + const toolUse: ContentBlockParam[] = []; + for (const block of blocks) { + if (block.type === "tool_use") toolUse.push(block); + else nonToolUse.push(block); + } + blocks.length = 0; + blocks.push(...nonToolUse, ...toolUse); + } if (blocks.length === 0) continue; params.push({ role: "assistant", diff --git a/packages/ai/src/providers/openai-codex-responses.ts b/packages/ai/src/providers/openai-codex-responses.ts index f1f71167a..80baabe8d 100644 --- a/packages/ai/src/providers/openai-codex-responses.ts +++ b/packages/ai/src/providers/openai-codex-responses.ts @@ -1232,12 +1232,27 @@ function getOutputBlockStartEventType(block: CodexOutputBlock): "thinking_start" return "toolcall_start"; } +const CODEX_STALE_PREVIOUS_RESPONSE_CODES: Record = { + // OpenAI-standard code for an expired/missing `previous_response_id` chain. + previous_response_not_found: true, + // Proxy-specific: upstream response anchor expired. Same recovery class — + // retry the turn with full context and no `previous_response_id`. + codex_previous_response_stale: true, +}; + function isCodexStalePreviousResponseError(error: unknown): boolean { - if (error instanceof CodexProviderStreamError) return error.code === "previous_response_not_found"; if (!(error instanceof Error)) return false; - if ((error as { code?: string }).code === "previous_response_not_found") return true; - // "unsupported": the backend intermittently rejects the parameter outright - // with `{"detail":"Unsupported parameter: previous_response_id"}` (no + if ( + "code" in error && + typeof error.code === "string" && + Object.hasOwn(CODEX_STALE_PREVIOUS_RESPONSE_CODES, error.code) + ) { + return true; + } + // Message-based fallback for providers/proxies that report the condition + // without a canonical code. Also covers "unsupported": the backend + // intermittently rejects the parameter outright with + // `{"detail":"Unsupported parameter: previous_response_id"}` (no // `error.code`); treat it like a stale chain so the turn replays with full // context instead of surfacing the 400. return ( diff --git a/packages/ai/src/providers/openai-responses.ts b/packages/ai/src/providers/openai-responses.ts index 92e81e7fa..759fdbfdc 100644 --- a/packages/ai/src/providers/openai-responses.ts +++ b/packages/ai/src/providers/openai-responses.ts @@ -668,7 +668,7 @@ const streamOpenAIResponsesOnce = ( } // Detect premature stream closure: the HTTP stream ended without the - // provider sending `response.completed` or `response.incomplete`. + // provider sending a recognized terminal response event. // Custom/proxy providers may drop the connection mid-stream; without // this guard the incomplete output is silently surfaced as a successful // "stop". diff --git a/packages/ai/src/providers/openai-shared.ts b/packages/ai/src/providers/openai-shared.ts index f496cf713..1d40b0ac5 100644 --- a/packages/ai/src/providers/openai-shared.ts +++ b/packages/ai/src/providers/openai-shared.ts @@ -80,6 +80,7 @@ import { import type { ChatCompletionCreateParamsStreaming } from "./openai-chat-wire"; import type { InputItem } from "./openai-codex/request-transformer"; import type { + Response as OpenAIResponse, ResponseContentPartAddedEvent, ResponseCreateParamsStreaming, ResponseCustomToolCall, @@ -1798,13 +1799,25 @@ export function finalizeCustomToolCallInputDone(block: ResponsesToolCallBlock, i block.arguments = { input }; } +type OpenAIResponsesTerminalStreamEvent = + | Extract + | { type: "response.done"; response?: Partial }; + +function getOpenAIResponsesTerminalEvent(event: ResponseStreamEvent): OpenAIResponsesTerminalStreamEvent | undefined { + const type = (event as { type?: unknown }).type; + return type === "response.completed" || type === "response.incomplete" || type === "response.done" + ? (event as OpenAIResponsesTerminalStreamEvent) + : undefined; +} + export interface ProcessResponsesStreamOptions { onFirstToken?: () => void; onOutputItemDone?: (item: ResponseOutputItem) => void; /** - * Called when a terminal `response.completed` or `response.incomplete` event - * is successfully processed. Only invoked on the successful-completion path; - * thrown failure (`response.failed`) and cancellation paths never call this. + * Called when a terminal `response.completed`, `response.incomplete`, or + * `response.done` event is successfully processed. Only invoked on the + * successful-completion path; thrown failure (`response.failed`) and + * cancellation paths never call this. * Used by callers to detect premature stream closure (i.e. the stream ended * without a recognized terminal event). */ @@ -2039,6 +2052,7 @@ export async function processResponsesStream( let sawFirstToken = false; for await (const event of openaiStream) { + const terminalEvent = getOpenAIResponsesTerminalEvent(event); if (event.type === "response.created") { output.responseId = event.response.id; } else if (event.type === "response.output_item.added") { @@ -2297,8 +2311,8 @@ export async function processResponsesStream( closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id)); stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output }); } - } else if (event.type === "response.completed" || event.type === "response.incomplete") { - const response = event.response; + } else if (terminalEvent) { + const response = terminalEvent.response; finalizePendingResponsesToolCalls(output); if (response?.id) { output.responseId = response.id; @@ -2336,7 +2350,7 @@ export async function processResponsesStream( } promoteResponsesToolUseStopReason(output, (response as { end_turn?: boolean } | undefined)?.end_turn); options?.onCompleted?.(); - // `response.completed`/`response.incomplete` is the last event of a + // `response.completed`/`response.incomplete`/`response.done` is the last event of a // Responses stream. Stop pulling instead of waiting for the server to // close the connection: misbehaving providers keep the socket open // after the terminal event, which would park this loop until the idle diff --git a/packages/ai/src/registry/oauth/callback-server.ts b/packages/ai/src/registry/oauth/callback-server.ts index afdbd1808..c3a45dc97 100644 --- a/packages/ai/src/registry/oauth/callback-server.ts +++ b/packages/ai/src/registry/oauth/callback-server.ts @@ -229,20 +229,32 @@ export abstract class OAuthCallbackFlow { /** * Build the `/launch` URL served by the callback server bound to `port`, or - * `undefined` when the configured `callbackPath` (or a `redirectUri` whose - * pathname resolves to {@link LAUNCH_PATH}) would collide with the launch - * route. Kept short (~30 chars) so UIs can advertise it as a + * `undefined` when it must not be advertised: + * - the configured `callbackPath` (or a `redirectUri` whose pathname + * resolves to {@link LAUNCH_PATH}) would collide with the launch route; + * - the flow's `redirectUri` never returns to this loopback server: fixed + * non-loopback hosts, or custom schemes like GitLab Duo's `vscode://` + * URI — which `new URL` parses without complaint, so a scheme/host check + * is required, not just the parse failure path. Advertising a localhost + * `/launch` target for such flows misrepresents the callback endpoint + * and hands remote users a URL that resolves nowhere. + * Kept short (~30 chars) so UIs can advertise it as a * viewport-truncation-safe copy target for the full authorization URL. */ #launchUrlIfSafe(port: number): string | undefined { if (this.callbackPath === LAUNCH_PATH) return undefined; if (this.redirectUri) { try { - if (new URL(this.redirectUri).pathname === LAUNCH_PATH) return undefined; + const parsed = new URL(this.redirectUri); + if (parsed.protocol !== "http:" && parsed.protocol !== "https:") return undefined; + if (parsed.hostname !== "localhost" && parsed.hostname !== "127.0.0.1" && parsed.hostname !== "[::1]") { + return undefined; + } + if (parsed.pathname === LAUNCH_PATH) return undefined; } catch { - // A non-parseable redirectUri (e.g. `vscode://...` handled elsewhere) - // can't collide with an HTTP `/launch` route — fall through and - // advertise the launch URL against the loopback server. + // A redirectUri even WHATWG URL cannot parse certainly does not + // return to this server — never advertise a launch URL for it. + return undefined; } } return `http://${this.callbackHostname}:${port}${LAUNCH_PATH}`; diff --git a/packages/ai/src/stream.ts b/packages/ai/src/stream.ts index 286ac23a4..02d17ad2f 100644 --- a/packages/ai/src/stream.ts +++ b/packages/ai/src/stream.ts @@ -1106,7 +1106,13 @@ export function streamSimple( // Caller aborted between attempts: don't mint a fresh token or fire // another doomed request — emit the captured failure instead. if (signal?.aborted) break; - const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal); + const nextKey = await resolveRetryKey( + apiKeyResolver, + AUTH_RETRY_STEPS[step]!, + failure.error, + signal, + lastKey, + ); if (nextKey === undefined || nextKey === lastKey) continue; lastKey = nextKey; const isLastStep = step === AUTH_RETRY_STEPS.length - 1; diff --git a/packages/ai/src/types.ts b/packages/ai/src/types.ts index a34df07c8..995fd454f 100644 --- a/packages/ai/src/types.ts +++ b/packages/ai/src/types.ts @@ -715,6 +715,8 @@ export interface AssistantMessage { stopReason: StopReason; stopDetails?: StopDetails | null; errorMessage?: string; + /** Per-tool abort messages used when an aborted assistant turn needs different placeholder results per tool call. */ + toolCallAbortMessages?: Record; /** HTTP status surfaced by the provider when the request failed. Populated by every provider's catch block alongside `errorMessage` so consumers (auth retry, telemetry, UI) can branch without regex-scraping the message. */ errorStatus?: number; /** Structured machine-readable error classifier; see `utils/error-id.ts` for bit layout and helpers. */ diff --git a/packages/ai/src/usage/openai-codex.ts b/packages/ai/src/usage/openai-codex.ts index 716ddd5fb..c7adebb4c 100644 --- a/packages/ai/src/usage/openai-codex.ts +++ b/packages/ai/src/usage/openai-codex.ts @@ -285,8 +285,6 @@ function buildUsageLimit(args: { label: usageWindow.label, scope: { provider: "openai-codex", - accountId: args.accountId, - tier: args.planType, windowId: usageWindow.id, shared: true, }, @@ -507,6 +505,9 @@ export const openaiCodexUsageProvider: UsageProvider = { const FIVE_HOUR_MS = 5 * 60 * 60 * 1000; export const codexRankingStrategy: CredentialRankingStrategy = { + blockScope() { + return "shared"; + }, findWindowLimits(report) { const findLimit = (key: "primary" | "secondary"): UsageLimit | undefined => { const direct = report.limits.find(l => l.id === `openai-codex:${key}`); diff --git a/packages/ai/test/anthropic-server-side-fallback.test.ts b/packages/ai/test/anthropic-server-side-fallback.test.ts index 9a1beec8c..5cae3490e 100644 --- a/packages/ai/test/anthropic-server-side-fallback.test.ts +++ b/packages/ai/test/anthropic-server-side-fallback.test.ts @@ -17,7 +17,8 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { convertAnthropicMessages, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; import { AnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic-client"; -import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types"; +import type { MessageParam } from "@oh-my-pi/pi-ai/providers/anthropic-wire"; +import type { AssistantMessage, Context, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; const fableModel: Model<"anthropic-messages"> = buildModel({ @@ -328,3 +329,255 @@ describe("anthropic fallback content-block replay policy", () => { expect(params[0]?.content).toEqual([{ type: "text", text: "continued" }]); }); }); + +describe("anthropic assistant replay block ordering (tool_use partition)", () => { + // Anthropic's replay validator rejects any assistant turn that carries a + // non-`tool_use` content block AFTER a `tool_use` block + // (`messages.N: tool_use ids were found without tool_result blocks + // immediately after`). This bites when a mid-turn server-side fallback + // (`server-side-fallback-2026-06-01`) lands AFTER the primary model already + // emitted a tool_use, leaving persisted content shaped like + // [thinking, text, tool_use, fallback, text, tool_use]. The same validator + // rejects the older cross-provider [text, tool_use, text] shape (issue + // #544). `convertAnthropicMessages` defends against both by stable- + // partitioning every assistant wire message: all non-tool_use blocks first + // (original relative order), then all tool_use blocks (original relative + // order). Already-valid messages must serialize byte-identically. + + function assistant(content: AssistantMessage["content"]): AssistantMessage { + return { + role: "assistant", + content, + api: "anthropic-messages", + provider: "anthropic", + model: "claude-fable-5", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason: "toolUse", + timestamp: 0, + }; + } + + function toolResult(id: string, name: string): ToolResultMessage { + return { + role: "toolResult", + toolCallId: id, + toolName: name, + content: [{ type: "text", text: "ok" }], + isError: false, + timestamp: 0, + }; + } + + function assistantParam(params: MessageParam[]): MessageParam { + const asst = params.filter(p => p.role === "assistant"); + if (asst.length !== 1 || !Array.isArray(asst[0]?.content)) { + throw new Error("expected exactly one assistant param with structured content blocks"); + } + return asst[0]; + } + + it("mid-turn fallback repro: defers trailing tool_use, keeps the fallback marker in place", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "after" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + { type: "toolCall", id: "call_c", name: "glob", arguments: {} }, + ]), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + toolResult("call_c", "glob"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "plan", signature: "sig-1" }, + { type: "text", text: "before" }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "after" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + { type: "tool_use", id: "call_b", name: "grep", input: {} }, + { type: "tool_use", id: "call_c", name: "glob", input: {} }, + ]); + }); + + it("opt-out drops the fallback marker but still defers tool_use to the tail", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "after" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + { type: "toolCall", id: "call_c", name: "glob", arguments: {} }, + ]), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + toolResult("call_c", "glob"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + // serverSideFallbackEnabled omitted → fallback block dropped + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "plan", signature: "sig-1" }, + { type: "text", text: "before" }, + { type: "text", text: "after" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + { type: "tool_use", id: "call_b", name: "grep", input: {} }, + { type: "tool_use", id: "call_c", name: "glob", input: {} }, + ]); + }); + + it("issue-544 family: text after a tool_use is reordered before the tool_use", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "text", text: "after" }, + ]), + toolResult("call_a", "read"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "text", text: "before" }, + { type: "text", text: "after" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + ]); + }); + + it("identity fast-path: already-valid thinking→text→tool_use serializes in unchanged order", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + ]), + toolResult("call_a", "read"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "plan", signature: "sig-1" }, + { type: "text", text: "before" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + ]); + }); + + it("interleaved signed thinking: signature-chain order preserved, tool_use deferred to the tail", () => { + const params = convertAnthropicMessages( + [ + assistant([ + { type: "thinking", thinking: "first", thinkingSignature: "sig-1" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "thinking", thinking: "second", thinkingSignature: "sig-2" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + ]), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + { role: "user", content: "next", timestamp: 0 }, + ], + fableModel, + false, + { serverSideFallbackEnabled: true }, + ); + expect(assistantParam(params).content).toEqual([ + { type: "thinking", thinking: "first", signature: "sig-1" }, + { type: "thinking", thinking: "second", signature: "sig-2" }, + { type: "tool_use", id: "call_a", name: "read", input: {} }, + { type: "tool_use", id: "call_b", name: "grep", input: {} }, + ]); + }); + + it("partition is localized: all other wire messages serialize byte-identically", () => { + // Prompt-cache contract: partitioning a poisoned assistant turn must be a + // LOCAL rewrite of that turn's own content — every OTHER wire message + // (the earlier valid assistant turn whose cached prefix must survive, its + // tool_result, the poisoned turn's tool_results, and the trailing user + // turn) must serialize to the exact same bytes. If the reorder leaked past + // the turn boundary (mutated a shared/adjacent message) or the slow path + // diverged from an already-ordered fast path, cached prefixes up to the + // reordered turn would be invalidated. Two histories, identical except the + // poisoned turn is pre-ordered to the partition result in (b): (a) fires + // the partition, (b) takes the fast path. Bytes must match everywhere but + // the poisoned param, which must converge to the same content either way. + const validAssistant = assistant([ + { type: "thinking", thinking: "plan", thinkingSignature: "sig-1" }, + { type: "text", text: "before" }, + { type: "toolCall", id: "call_v", name: "list", arguments: {} }, + ]); + const poisonedContent: AssistantMessage["content"] = [ + { type: "text", text: "poison-a" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "poison-b" }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + ]; + // Same blocks, hand-ordered to exactly what the stable partition emits: + // non-tool_use chain first (order preserved), then the tool_use tail. + const preOrderedContent: AssistantMessage["content"] = [ + { type: "text", text: "poison-a" }, + { type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } }, + { type: "text", text: "poison-b" }, + { type: "toolCall", id: "call_a", name: "read", arguments: {} }, + { type: "toolCall", id: "call_b", name: "grep", arguments: {} }, + ]; + const history = (poisoned: AssistantMessage["content"]) => [ + { role: "user", content: "start", timestamp: 0 } as const, + validAssistant, + toolResult("call_v", "list"), + assistant(poisoned), + toolResult("call_a", "read"), + toolResult("call_b", "grep"), + { role: "user", content: "next", timestamp: 0 } as const, + ]; + + const a = convertAnthropicMessages(history(poisonedContent), fableModel, false, { + serverSideFallbackEnabled: true, + }); + const b = convertAnthropicMessages(history(preOrderedContent), fableModel, false, { + serverSideFallbackEnabled: true, + }); + + expect(a.length).toBe(b.length); + const poisonedIdx = a.findIndex( + p => + p.role === "assistant" && + Array.isArray(p.content) && + p.content.some(block => block.type === "tool_use" && block.id === "call_a"), + ); + expect(poisonedIdx).toBeGreaterThanOrEqual(0); + for (let i = 0; i < a.length; i++) { + if (i === poisonedIdx) continue; + expect(JSON.stringify(a[i])).toBe(JSON.stringify(b[i])); + } + // Convergence: partition (a) and fast path (b) yield identical final content. + expect(a[poisonedIdx]).toEqual(b[poisonedIdx]); + }); +}); diff --git a/packages/ai/test/auth-retry.test.ts b/packages/ai/test/auth-retry.test.ts index 73b1f66d1..44a4a9f89 100644 --- a/packages/ai/test/auth-retry.test.ts +++ b/packages/ai/test/auth-retry.test.ts @@ -99,6 +99,28 @@ describe("withAuth", () => { ]); }); + it("switches accounts before refreshing the same account on usage limits", async () => { + const keys: string[] = []; + const contexts: ApiKeyResolveContext[] = []; + const result = await withAuth( + ctx => { + contexts.push(ctx); + return ctx.error === undefined ? "k0" : ctx.lastChance ? "k2" : "k1"; + }, + async key => { + keys.push(key); + if (key === "k2") return "success"; + throw usageLimitError(); + }, + ); + expect(result).toBe("success"); + expect(keys).toEqual(["k0", "k2"]); + expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ + { lastChance: false, hasError: false }, + { lastChance: true, hasError: true }, + ]); + }); + it("stops retrying when the resolver returns undefined", async () => { const keys: string[] = []; const original = authError(); diff --git a/packages/ai/test/auth-storage-api-key-login.test.ts b/packages/ai/test/auth-storage-api-key-login.test.ts index 879a8a9ec..1cdb9f8fa 100644 --- a/packages/ai/test/auth-storage-api-key-login.test.ts +++ b/packages/ai/test/auth-storage-api-key-login.test.ts @@ -39,9 +39,9 @@ function countCredentialRowsByDisabledState(dbPath: string, provider: string, di } describe("AuthStorage api-key login upsert", () => { - // A live env var now (correctly) overrides a stored static api_key. These tests verify that a - // freshly stored api_key resolves through AuthStorage.getApiKey, so neutralize the env leg - // entirely — this ignores every provider's ambient env key, not just the few set locally. + // Most tests neutralize the env leg so ambient shell / ~/.env keys cannot + // hide the stored credential behavior under test. Login-persisted API keys + // have their own precedence coverage below. let tempDir = ""; let dbPath = ""; let store: SqliteAuthCredentialStore | null = null; @@ -49,9 +49,10 @@ describe("AuthStorage api-key login upsert", () => { let loginDeepSeekSpy: Mock; let loginKagiSpy: Mock; let loginOllamaCloudSpy: Mock; + let getEnvApiKeySpy: Mock; beforeEach(async () => { - vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined); + getEnvApiKeySpy = vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined); tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-api-key-login-")); dbPath = path.join(tempDir, "agent.db"); store = await SqliteAuthCredentialStore.open(dbPath); @@ -118,8 +119,8 @@ describe("AuthStorage api-key login upsert", () => { const credentials = store.listAuthCredentials("kagi"); expect(credentials.map(entry => entry.credential)).toEqual([ - { type: "api_key", key: "first-kagi-key" }, - { type: "api_key", key: "second-kagi-key" }, + { type: "api_key", key: "first-kagi-key", source: "login" }, + { type: "api_key", key: "second-kagi-key", source: "login" }, ]); const rotatedKeys = [await authStorage.getApiKey("kagi"), await authStorage.getApiKey("kagi")].sort(); expect(rotatedKeys).toEqual(["first-kagi-key", "second-kagi-key"]); @@ -188,4 +189,18 @@ describe("AuthStorage api-key login upsert", () => { expect(store.getApiKey("deepseek")).toBe("same-deepseek-key"); expect(await authStorage.getApiKey("deepseek", "session-deepseek-relogin")).toBe("same-deepseek-key"); }); + + it("uses a fresh OpenCode Go login over an existing env fallback", async () => { + if (!authStorage) throw new Error("test setup failed"); + + getEnvApiKeySpy.mockImplementation(provider => (provider === "opencode-go" ? "old-opencode-key" : undefined)); + + await authStorage.login("opencode-go", { + onAuth: () => {}, + onPrompt: async () => "new-opencode-key", + }); + + expect(await authStorage.getApiKey("opencode-go", "session-opencode-go-login")).toBe("new-opencode-key"); + expect(await authStorage.peekApiKey("opencode-go")).toBe("new-opencode-key"); + }); }); diff --git a/packages/ai/test/auth-storage-codex-selection.test.ts b/packages/ai/test/auth-storage-codex-selection.test.ts index 6b803b400..b6aa79847 100644 --- a/packages/ai/test/auth-storage-codex-selection.test.ts +++ b/packages/ai/test/auth-storage-codex-selection.test.ts @@ -98,7 +98,7 @@ function createCredential(accountId: string, email: string): OAuthCredentials { return { access: `access-${accountId}`, refresh: `refresh-${accountId}`, - expires: Date.now() + HOUR_MS, + expires: Date.now() + WEEK_MS, accountId, email, }; @@ -595,6 +595,100 @@ describe("AuthStorage codex oauth ranking", () => { // flaky on loaded CI runners, so maxConcurrent is the authoritative signal. expect(maxConcurrent).toBe(3); }); + + test("skips expired access-token-only sticky credential and selects fresh sibling", async () => { + if (!authStorage) throw new Error("test setup failed"); + const sessionId = "sticky-token-only-session"; + await authStorage.set("openai-codex", [{ type: "oauth", ...createCredential("acct-k12", "k12@example.com") }]); + usageByAccount.set( + "acct-k12", + createCodexUsageReport({ + accountId: "acct-k12", + primary: { usedFraction: 0.3, resetInMs: 20 * 60 * 1000 }, + secondary: { usedFraction: 0.2, resetInMs: 5 * 24 * 60 * 60 * 1000 }, + }), + ); + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-k12"); + usageByAccount.set( + "acct-k12", + createCodexUsageReport({ + accountId: "acct-k12", + primary: { usedFraction: 1, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 0.17, resetInMs: WEEK_MS }, + }), + ); + usageByAccount.set( + "acct-plus", + createCodexUsageReport({ + accountId: "acct-plus", + primary: { usedFraction: 0.2, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 0.74, resetInMs: WEEK_MS }, + }), + ); + + await authStorage.set("openai-codex", [ + { + type: "oauth", + access: "access-acct-k12", + refresh: "", + expires: Date.now() - 1_000, + accountId: "acct-k12", + email: "k12@example.com", + }, + { + type: "oauth", + ...createCredential("acct-plus", "plus@example.com"), + }, + ]); + + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-plus"); + }); + + test("ignores legacy global Codex blocks when a scoped quota window has fresh siblings", async () => { + if (!authStorage || !store) throw new Error("test setup failed"); + await authStorage.set("openai-codex", [ + { type: "oauth", ...createCredential("acct-k12", "k12@example.com") }, + { type: "oauth", ...createCredential("acct-plus", "plus@example.com") }, + ]); + usageByAccount.set( + "acct-k12", + createCodexUsageReport({ + accountId: "acct-k12", + primary: { usedFraction: 1, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 1, resetInMs: WEEK_MS }, + }), + ); + usageByAccount.set( + "acct-plus", + createCodexUsageReport({ + accountId: "acct-plus", + primary: { usedFraction: 0.2, resetInMs: FIVE_HOUR_MS }, + secondary: { usedFraction: 0.74, resetInMs: WEEK_MS }, + }), + ); + const plus = store + .listAuthCredentials("openai-codex") + .find(row => row.credential.type === "oauth" && row.credential.accountId === "acct-plus"); + if (!plus || !store.upsertCredentialBlock) throw new Error("missing plus credential row"); + store.upsertCredentialBlock({ + credentialId: plus.id, + providerKey: "openai-codex:oauth", + blockScope: "", + blockedUntilMs: Date.now() + WEEK_MS, + }); + const k12 = store + .listAuthCredentials("openai-codex") + .find(row => row.credential.type === "oauth" && row.credential.accountId === "acct-k12"); + if (!k12 || !store.upsertCredentialBlock) throw new Error("missing k12 credential row"); + store.upsertCredentialBlock({ + credentialId: k12.id, + providerKey: "openai-codex:oauth", + blockScope: "shared", + blockedUntilMs: Date.now() + HOUR_MS, + }); + + expect(await authStorage.getApiKey("openai-codex", "session-with-legacy-global-block")).toBe("api-acct-plus"); + }); }); // ───────────────────────────────────────────────────────────────────────────── diff --git a/packages/ai/test/callback-server-launch-route.test.ts b/packages/ai/test/callback-server-launch-route.test.ts index 8f14cc59b..3d22b4205 100644 --- a/packages/ai/test/callback-server-launch-route.test.ts +++ b/packages/ai/test/callback-server-launch-route.test.ts @@ -182,4 +182,61 @@ describe("OAuthCallbackFlow /launch route", () => { abort.abort("test done"); await login; }); + + it("suppresses launchUrl for custom-scheme redirects that never return to the loopback server", async () => { + const abort = new AbortController(); + const authFired = Promise.withResolvers(); + const flow = new LaunchProbeFlow( + { + onAuth: info => { + authFired.resolve(info); + }, + signal: abort.signal, + }, + { + preferredPort: 0, + allowPortFallback: true, + // GitLab Duo shape: `new URL` parses this happily (pathname + // `/authentication`), so the guard must check scheme/host, not + // rely on a parse failure. A localhost /launch copy target for + // this flow would misrepresent the callback endpoint and point + // remote users at a URL that resolves nowhere. + redirectUri: "vscode://gitlab.gitlab-workflow/authentication", + }, + ); + const login = flow.login().catch(() => undefined) as Promise; + const info = await authFired.promise; + + expect(info.launchUrl).toBeUndefined(); + + abort.abort("test done"); + await login; + }); + + it("suppresses launchUrl for fixed non-loopback HTTP redirects", async () => { + const abort = new AbortController(); + const authFired = Promise.withResolvers(); + const flow = new LaunchProbeFlow( + { + onAuth: info => { + authFired.resolve(info); + }, + signal: abort.signal, + }, + { + preferredPort: 0, + allowPortFallback: true, + // The provider redirects to a hosted endpoint; this machine's + // callback server never sees the redirect, so no launch URL. + redirectUri: "https://auth.example.com/oauth/callback", + }, + ); + const login = flow.login().catch(() => undefined) as Promise; + const info = await authFired.promise; + + expect(info.launchUrl).toBeUndefined(); + + abort.abort("test done"); + await login; + }); }); diff --git a/packages/ai/test/issue-4679-repro.test.ts b/packages/ai/test/issue-4679-repro.test.ts new file mode 100644 index 000000000..2372bfc2d --- /dev/null +++ b/packages/ai/test/issue-4679-repro.test.ts @@ -0,0 +1,104 @@ +import { describe, expect, it } from "bun:test"; +import { buildAnthropicClientOptions, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic"; +import type { Context, Model, ModelSpec, TJsonSchema, Tool } from "@oh-my-pi/pi-ai/types"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; + +const STRUCTURED_OUTPUTS_BETA = "structured-outputs-2025-12-15"; + +const bashTool: Tool = { + name: "bash", + description: "run a bash command", + parameters: { + type: "object", + properties: { command: { type: "string" } }, + required: ["command"], + } satisfies TJsonSchema, +}; + +const toolContext: Context = { + systemPrompt: ["Stay concise."], + messages: [{ role: "user", content: "Hi", timestamp: 0 }], + tools: [bashTool], +}; + +function anthropicSpec(baseUrl: string): ModelSpec<"anthropic-messages"> { + return { + id: "claude-sonnet-5", + name: "Claude Sonnet 5", + api: "anthropic-messages", + provider: "anthropic", + baseUrl, + reasoning: false, + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 200_000, + maxTokens: 8_192, + }; +} + +function buildOAuthUtilityBetaHeader(model: Model<"anthropic-messages">): string { + const options = buildAnthropicClientOptions({ + model, + apiKey: "oauth-token", + isOAuth: true, + hasTools: false, + thinkingEnabled: false, + }); + return options.defaultHeaders["anthropic-beta"] ?? ""; +} + +function abortedSignal(): AbortSignal { + const controller = new AbortController(); + controller.abort(); + return controller.signal; +} + +async function captureToolParams( + model: Model<"anthropic-messages">, +): Promise<{ tools?: Array<{ name: string; strict?: unknown }> }> { + const { promise, resolve } = Promise.withResolvers<{ tools?: Array<{ name: string; strict?: unknown }> }>(); + void streamAnthropic(model, toolContext, { + apiKey: "sk-ant-api-test", + isOAuth: false, + signal: abortedSignal(), + onPayload: payload => { + resolve(payload as { tools?: Array<{ name: string; strict?: unknown }> }); + return undefined; + }, + }); + return promise; +} + +describe("issue #4679 Azure Foundry Anthropic strict tools", () => { + it.each([ + ["inference", "https://example.inference.ai.azure.com/anthropic/v1"], + ["services", "https://example.services.ai.azure.com/anthropic/v1"], + ])("disables strict tools and omits structured-output beta for Azure Foundry %s routes", (_kind, baseUrl) => { + const model = buildModel(anthropicSpec(baseUrl)); + + expect(model.compat.disableStrictTools).toBe(true); + expect(buildOAuthUtilityBetaHeader(model)).not.toContain(STRUCTURED_OUTPUTS_BETA); + }); + + it("keeps structured-output beta on direct Anthropic OAuth utility headers", () => { + const model = buildModel(anthropicSpec("https://api.anthropic.com")); + + expect(model.compat.disableStrictTools).toBe(false); + expect(buildOAuthUtilityBetaHeader(model)).toContain(STRUCTURED_OUTPUTS_BETA); + }); + + it("omits strict tool schemas on Azure Foundry Anthropic requests without disabling direct Anthropic", async () => { + const azureParams = await captureToolParams( + buildModel(anthropicSpec("https://example.services.ai.azure.com/anthropic/v1")), + ); + const directParams = await captureToolParams(buildModel(anthropicSpec("https://api.anthropic.com"))); + + const azureBashTool = azureParams.tools?.find(tool => tool.name === "bash"); + const directBashTool = directParams.tools?.find(tool => tool.name === "bash"); + + expect(azureBashTool).toBeDefined(); + expect(azureBashTool?.strict).toBeUndefined(); + expect(directBashTool).toBeDefined(); + expect(directBashTool?.strict).toBe(true); + }); +}); diff --git a/packages/ai/test/openai-codex-stream.test.ts b/packages/ai/test/openai-codex-stream.test.ts index 13bd10e96..e7fa92c2e 100644 --- a/packages/ai/test/openai-codex-stream.test.ts +++ b/packages/ai/test/openai-codex-stream.test.ts @@ -2708,6 +2708,114 @@ describe("openai-codex streaming", () => { lastPreviousResponseId: undefined, }); }); + it("retries websocket continuations when a proxy reports a stale previous response anchor", async () => { + const tempDir = TempDir.createSync("@pi-codex-stream-"); + setAgentDir(tempDir.path()); + const token = createCodexTestToken(); + const sentRequests: Array> = []; + const fetchMock = vi.fn(async () => { + throw new Error("SSE fallback should not be called"); + }); + + class ProxyStaleAnchorWebSocket extends MockWebSocket { + constructor(url: string, options?: { headers?: WsHeaders }) { + super(url, options); + this.scheduleOpen(); + } + + send(data: string): void { + const request = JSON.parse(data) as Record; + sentRequests.push(request); + const requestIndex = sentRequests.length; + + if (requestIndex === 1) { + this.emitCodexResponse({ + messageId: "msg_1", + responseId: "resp_1", + text: "First answer", + terminalType: "response.completed", + includeCreated: true, + }); + return; + } + + if (requestIndex === 2) { + expect(request.previous_response_id).toBe("resp_1"); + this.sendJson({ + type: "error", + code: "codex_previous_response_stale", + message: "Upstream previous response anchor expired; retry without previous_response_id.", + }); + return; + } + + if (requestIndex === 3) { + expect(request.previous_response_id).toBeUndefined(); + this.emitCodexResponse({ + messageId: "msg_3", + responseId: "resp_3", + text: "Second answer", + terminalType: "response.completed", + includeCreated: true, + }); + return; + } + + throw new Error(`Unexpected websocket request index: ${requestIndex}`); + } + } + + global.WebSocket = ProxyStaleAnchorWebSocket as unknown as typeof WebSocket; + const model = createCodexTestModel("https://chatgpt.com/backend-api"); + const providerSessionState = new Map(); + const firstContext: Context = { + systemPrompt: ["You are a helpful assistant."], + messages: [{ role: "user", content: "First question", timestamp: Date.now() }], + }; + const firstResponse = await streamOpenAICodexResponses(model, firstContext, { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-proxy-stale-anchor-session", + providerSessionState, + }).result(); + const secondContext: Context = { + systemPrompt: ["You are a helpful assistant."], + messages: [ + ...firstContext.messages, + firstResponse, + { role: "user", content: "Second question", timestamp: Date.now() + 1 }, + ], + }; + + const secondResponse = await streamOpenAICodexResponses(model, secondContext, { + fetch: fetchMock as FetchImpl, + apiKey: token, + sessionId: "ws-proxy-stale-anchor-session", + providerSessionState, + }).result(); + + expect(secondResponse.stopReason).toBe("stop"); + expect(JSON.stringify(secondResponse.content)).toContain("Second answer"); + expect(fetchMock).not.toHaveBeenCalled(); + expect(sentRequests).toHaveLength(3); + expect(sentRequests[2]?.prompt_cache_key).toBe("ws-proxy-stale-anchor-session"); + const retryInput = sentRequests[2]?.input; + expect(Array.isArray(retryInput)).toBe(true); + expect(JSON.stringify(retryInput)).toContain("First question"); + expect(JSON.stringify(retryInput)).toContain("Second question"); + + const stats = getOpenAICodexWebSocketDebugStats(model, { + sessionId: "ws-proxy-stale-anchor-session", + providerSessionState, + }); + expect(stats).toMatchObject({ + fullContextRequests: 2, + deltaRequests: 1, + lastInputItems: (retryInput as unknown[]).length, + lastDeltaInputItems: undefined, + lastPreviousResponseId: undefined, + }); + }); it("uses websocket v2 beta header when v2 mode is enabled", async () => { const tempDir = TempDir.createSync("@pi-codex-stream-"); diff --git a/packages/ai/test/openai-codex-usage.test.ts b/packages/ai/test/openai-codex-usage.test.ts index b4f385434..1027c8070 100644 --- a/packages/ai/test/openai-codex-usage.test.ts +++ b/packages/ai/test/openai-codex-usage.test.ts @@ -63,7 +63,7 @@ describe("openai-codex usage parser", () => { expect(report).not.toBeNull(); const main = report?.limits.filter(l => l.id === "openai-codex:primary" || l.id === "openai-codex:secondary"); expect(main?.map(l => l.id)).toEqual(["openai-codex:primary", "openai-codex:secondary"]); - expect(main?.[0].scope.tier).toBe("pro"); + expect(main?.[0].scope).toEqual({ provider: "openai-codex", windowId: "5h", shared: true }); expect(main?.[0].amount.usedFraction).toBeCloseTo(0.04, 5); }); diff --git a/packages/ai/test/openai-first-event-timeout.test.ts b/packages/ai/test/openai-first-event-timeout.test.ts index 34df11efe..4b5fcbbc1 100644 --- a/packages/ai/test/openai-first-event-timeout.test.ts +++ b/packages/ai/test/openai-first-event-timeout.test.ts @@ -705,6 +705,53 @@ describe("OpenAI-family first-event timeouts", () => { ]); }); + it("accepts response.done with a completed response as an OpenAI responses terminal event", async () => { + const completedResponse = createSseResponse([ + { type: "response.created", response: { id: "resp_done" } }, + { + type: "response.output_item.added", + item: { type: "message", id: "msg_done", role: "assistant", status: "in_progress", content: [] }, + }, + { type: "response.content_part.added", part: { type: "output_text", text: "" } }, + { type: "response.output_text.delta", delta: "Hello done" }, + { + type: "response.output_item.done", + item: { + type: "message", + id: "msg_done", + role: "assistant", + status: "completed", + content: [{ type: "output_text", text: "Hello done" }], + }, + }, + { + type: "response.done", + response: { + id: "resp_done", + status: "completed", + usage: { + input_tokens: 3, + output_tokens: 2, + total_tokens: 5, + }, + }, + }, + ]); + const fetch: FetchImpl = () => Promise.resolve(completedResponse); + const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), { + apiKey: "test-key", + fetch, + }).result(); + + expect(result.errorMessage).toBeUndefined(); + expect(result.stopReason).toBe("stop"); + expect(result.content as unknown[]).toContainEqual({ + type: "text", + text: "Hello done", + textSignature: '{"v":1,"id":"msg_done"}', + }); + }); + it("errors when Azure OpenAI responses stream closes without a terminal response event", async () => { const incompleteResponse = createSseResponse([ { type: "response.created", response: { id: "resp_incomplete_azure" } }, diff --git a/packages/ai/test/stream-auth-retry.test.ts b/packages/ai/test/stream-auth-retry.test.ts index 484d80f01..856fb08ae 100644 --- a/packages/ai/test/stream-auth-retry.test.ts +++ b/packages/ai/test/stream-auth-retry.test.ts @@ -405,7 +405,7 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["credential-A", "credential-B"]); expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]); - expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([false, true]); + expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([true]); } }); @@ -501,10 +501,9 @@ describe("streamSimple resolver auth retry", () => { expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]); expect(keys).toEqual(["old-key", "next-key"]); expect(retryContexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([ - { lastChance: false, hasError: true }, { lastChance: true, hasError: true }, ]); - expect((retryContexts[1]?.error as Error).message).toContain("Resource exhausted"); + expect((retryContexts[0]?.error as Error).message).toContain("Resource exhausted"); }); it("surfaces the original error when the resolver declines every retry", async () => { diff --git a/packages/catalog/CHANGELOG.md b/packages/catalog/CHANGELOG.md index 42e39218a..1a020e17d 100644 --- a/packages/catalog/CHANGELOG.md +++ b/packages/catalog/CHANGELOG.md @@ -2,6 +2,12 @@ ## [Unreleased] +### Fixed + +- Fixed LiteLLM discovery stopping at `/model_group/info` when that endpoint omitted `supports_vision`; it now continues to `/model/info` and preserves `model_info.supports_vision=true` for vision-capable proxy models. ([#4747](https://github.com/can1357/oh-my-pi/issues/4747)) +- Fixed LiteLLM discovery to fall back to bundled catalog metadata when `models.dev` lacks a model reference, preserving reasoning and thinking support for models such as `glm-5.2`. ([#4695](https://github.com/can1357/oh-my-pi/issues/4695)) +- Detected Azure AI Inference / Foundry Anthropic routes as strict-tool-incompatible so resolved Anthropic compat disables strict tools before request construction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)). + ## [16.3.11] - 2026-07-06 ### Added diff --git a/packages/catalog/src/compat/anthropic.ts b/packages/catalog/src/compat/anthropic.ts index 7bc1777c0..fafb6cac9 100644 --- a/packages/catalog/src/compat/anthropic.ts +++ b/packages/catalog/src/compat/anthropic.ts @@ -99,18 +99,15 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res // (issue #4192). const isZenmux = modelMatchesHost(spec, "zenmux"); const requiresThinkingEnabled = modelMatchesHost(spec, "moonshotNative") && matchesKimiK27CodeFamily(spec); + const isVertex = isVertexAnthropicRoute(baseUrl); + const isBedrock = isBedrockAnthropicRoute(baseUrl); + const isAzure = isAzureAnthropicRoute(baseUrl); const signingEndpoint = - official || - isCopilot || - isZenmux || - isCloudflareAnthropicGateway(baseUrl) || - isVertexAnthropicRoute(baseUrl) || - isBedrockAnthropicRoute(baseUrl) || - isAzureAnthropicRoute(baseUrl); + official || isCopilot || isZenmux || isCloudflareAnthropicGateway(baseUrl) || isVertex || isBedrock || isAzure; const compat: ResolvedAnthropicCompat = { officialEndpoint: official, signingEndpoint, - disableStrictTools: false, + disableStrictTools: isAzure, disableAdaptiveThinking: false, supportsEagerToolInputStreaming: !isCopilot, // Long cache retention is only sent to the official API by default; diff --git a/packages/catalog/src/provider-models/openai-compat.ts b/packages/catalog/src/provider-models/openai-compat.ts index d43a20143..83cd58136 100644 --- a/packages/catalog/src/provider-models/openai-compat.ts +++ b/packages/catalog/src/provider-models/openai-compat.ts @@ -3032,6 +3032,15 @@ export interface FetchLiteLLMRichModelsOptions { } type LiteLLMRichModelEntry = Record; +type LiteLLMRichEndpointModel = { + model: ModelSpec; + supportsVision: unknown; + supportsReasoning: unknown; + hasContextWindow: boolean; + hasMaxTokens: boolean; + hasToolMetadata: boolean; + hasSupportedOpenAIParams: boolean; +}; const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const; export const OPENAI_COMPAT_DISCOVERY_DEFAULT_CONTEXT_WINDOW = 128_000; @@ -3264,7 +3273,7 @@ async function fetchLiteLLMRichEndpoint( managementBaseUrl: string, runtimeBaseUrl: string, signal?: AbortSignal, -): Promise[] | null> { +): Promise<{ models: LiteLLMRichEndpointModel[]; incompleteVisionMetadata: boolean } | null> { const fetchImpl = discoveryFetch(options.fetch); const requestHeaders: Record = { Accept: "application/json", @@ -3296,17 +3305,39 @@ async function fetchLiteLLMRichEndpoint( if (!entries || entries.length === 0) { return null; } - const deduped = new Map>(); + const deduped = new Map>(); + let incompleteVisionMetadata = false; for (const entry of entries) { const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl); if (model) { - deduped.set(model.id, model); + const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision"); + const supportsReasoning = getLiteLLMMetadataValue(entry, "supports_reasoning"); + const supportsFunctionCalling = getLiteLLMMetadataValue(entry, "supports_function_calling"); + const supportedOpenAIParams = getSupportedOpenAIParams(entry); + if (supportsVision !== true && supportsVision !== false) { + incompleteVisionMetadata = true; + } + deduped.set(model.id, { + model, + supportsVision, + supportsReasoning, + hasContextWindow: toPositiveNumber(getLiteLLMMetadataValue(entry, "max_input_tokens"), null) !== null, + hasMaxTokens: toPositiveNumber(getLiteLLMMetadataValue(entry, "max_output_tokens"), null) !== null, + hasToolMetadata: + supportsFunctionCalling === true || + supportsFunctionCalling === false || + supportedOpenAIParams !== undefined, + hasSupportedOpenAIParams: supportedOpenAIParams !== undefined, + }); } } if (deduped.size === 0) { return null; } - return Array.from(deduped.values()).sort((left, right) => left.id.localeCompare(right.id)); + return { + models: Array.from(deduped.values()).sort((left, right) => left.model.id.localeCompare(right.model.id)), + incompleteVisionMetadata, + }; } export async function fetchLiteLLMRichModels( @@ -3318,13 +3349,55 @@ export async function fetchLiteLLMRichModels( return null; } const fetchModels = async (signal?: AbortSignal): Promise[] | null> => { + const deduped = new Map>(); for (const endpoint of LITELLM_RICH_ENDPOINTS) { - const models = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal); - if (models) { - return models; + const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal); + if (!result) { + continue; + } + const hadPriorModels = deduped.size > 0; + for (const next of result.models) { + const existing = deduped.get(next.model.id); + if (!existing) { + if (!hadPriorModels) { + deduped.set(next.model.id, next); + } + continue; + } + const model: ModelSpec = { + ...existing.model, + name: next.model.name === next.model.id ? existing.model.name : next.model.name, + contextWindow: next.hasContextWindow ? next.model.contextWindow : existing.model.contextWindow, + maxTokens: next.hasMaxTokens ? next.model.maxTokens : existing.model.maxTokens, + input: + next.supportsVision === true || next.supportsVision === false + ? next.model.input + : existing.model.input, + reasoning: typeof next.supportsReasoning === "boolean" ? next.model.reasoning : existing.model.reasoning, + compat: next.hasSupportedOpenAIParams ? next.model.compat : existing.model.compat, + }; + if (next.hasToolMetadata) { + model.supportsTools = next.model.supportsTools; + } + deduped.set(next.model.id, { ...next, model }); + } + let hasIncompleteVisionMetadata = false; + for (const entry of deduped.values()) { + if (entry.supportsVision !== true && entry.supportsVision !== false) { + hasIncompleteVisionMetadata = true; + break; + } + } + if (!hasIncompleteVisionMetadata) { + break; } } - return null; + if (deduped.size === 0) { + return null; + } + return Array.from(deduped.values()) + .map(entry => entry.model) + .sort((left, right) => left.id.localeCompare(right.id)); }; if (options.signal !== undefined) { return fetchModels(options.signal); @@ -3339,18 +3412,20 @@ export function litellmModelManagerOptions( const baseUrl = config?.baseUrl ?? Bun.env.LITELLM_BASE_URL ?? "http://localhost:4000/v1"; return { providerId: "litellm", - // rich-v3 invalidates rows cached before reseller usage-suffix stripping - // and placeholder-only `all-team-models` filtering; bump the version whenever - // the mappers below change, or warm authoritative caches keep serving - // pre-change rows for the full TTL. - cacheProviderId: `litellm:rich-v3:${Bun.hash(baseUrl).toString(36)}`, + // rich-v4 invalidates rows cached before LiteLLM ids gained bundled + // reference fallback and before discovery continued past `/model_group/info` + // when that endpoint omitted vision metadata. Earlier versions handled + // reseller usage-suffix stripping and placeholder-only `all-team-models` + // filtering; bump the version whenever the mappers below change, or warm + // authoritative caches keep serving pre-change rows for the full TTL. + cacheProviderId: `litellm:rich-v4:${Bun.hash(baseUrl).toString(36)}`, // litellm is a local-only proxy and is never bundled in models.json (that // would leak the machine's localhost catalog). Prefer the proxy's richer - // management metadata, then fall back to /v1/models and enrich bare ids - // against models.dev like the gateway providers (fireworks et al.) do. + // management metadata, then enrich ids against models.dev with the bundled + // catalog as a fallback before using /v1/models. fetchDynamicModels: async () => { const modelsDevReferences = await loadModelsDevReferences<"openai-completions">(config?.fetch); - const resolveReference = (id: string) => modelsDevReferences.get(id); + const resolveReference = createReferenceResolver(modelsDevReferences); const richModels = await fetchLiteLLMRichModels({ api: "openai-completions", provider: "litellm", diff --git a/packages/catalog/test/litellm-provider.test.ts b/packages/catalog/test/litellm-provider.test.ts index ec6d88704..c353ac16f 100644 --- a/packages/catalog/test/litellm-provider.test.ts +++ b/packages/catalog/test/litellm-provider.test.ts @@ -125,7 +125,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v3:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, + `litellm:rich-v4:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); @@ -148,7 +148,7 @@ describe("LiteLLM provider discovery", () => { const models = await options.fetchDynamicModels?.(); expect(options.cacheProviderId).toBe( - `litellm:rich-v3:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, + `litellm:rich-v4:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`, ); expect(fetchMock).toHaveBeenCalledTimes(6); expect(models).toHaveLength(1); @@ -238,6 +238,54 @@ describe("LiteLLM provider discovery", () => { }); }); + test("enriches LiteLLM rich models missing from models.dev with bundled reasoning metadata", async () => { + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://primary:4000/model_group/info") { + return Response.json({ + data: [ + { + model_group: "glm-5.2", + model_name: "GLM-5.2", + }, + ], + }); + } + if (url === "http://primary:4000/v1/models") { + throw new Error("/v1/models should not be called when model_group info has a real model"); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-rich", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(models).toHaveLength(1); + expect(models?.[0]).toMatchObject({ + id: "glm-5.2", + name: "GLM-5.2", + api: "openai-completions", + provider: "litellm", + baseUrl: "http://primary:4000/v1", + reasoning: true, + thinking: { + mode: "effort", + efforts: ["minimal", "low", "medium", "high", "xhigh"], + effortMap: { + minimal: "none", + xhigh: "max", + }, + }, + }); + }); + test("uses LiteLLM tool support metadata when rich endpoints succeed", async () => { const fetchMock = vi.fn(async (input: string | URL | Request) => { const url = inputUrl(input); @@ -247,8 +295,13 @@ describe("LiteLLM provider discovery", () => { if (url === "http://primary:4000/model_group/info") { return Response.json({ data: [ - { model_group: "no-tools", providers: ["openai"], supports_function_calling: false }, - { model_group: "params-tools", supported_openai_params: ["tools"] }, + { + model_group: "no-tools", + providers: ["openai"], + supports_vision: false, + supports_function_calling: false, + }, + { model_group: "params-tools", supports_vision: false, supported_openai_params: ["tools"] }, ], }); } @@ -339,6 +392,7 @@ describe("LiteLLM provider discovery", () => { max_input_tokens: 96_000, max_output_tokens: 8_000, supports_function_calling: true, + supports_vision: false, }, ], }); @@ -422,6 +476,71 @@ describe("LiteLLM provider discovery", () => { }); }); + test("continues to LiteLLM model info when model_group omits vision metadata", async () => { + const calls: string[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + calls.push(url); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if (url === "http://primary:4000/model_group/info") { + return Response.json({ + data: [ + { + model_group: "vision-proxy-model", + model_name: "Vision Proxy Model", + max_input_tokens: 64_000, + max_output_tokens: 8_000, + }, + ], + }); + } + if (url === "http://primary:4000/v2/model/info") { + return Response.json({ + data: [{ model_name: "unrelated-v2-model", model_info: { supports_vision: false } }], + }); + } + if (url === "http://primary:4000/model/info") { + return Response.json({ + data: [ + { model_name: "text-only-model", model_info: { supports_vision: false } }, + { + model_name: "vision-proxy-model", + model_info: { + supports_vision: true, + }, + }, + ], + }); + } + if (url === "http://primary:4000/v1/models") { + throw new Error("/v1/models should not be called when LiteLLM model info succeeds"); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-rich", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(calls).toContain("http://primary:4000/model_group/info"); + expect(calls).toContain("http://primary:4000/v2/model/info"); + expect(calls).toContain("http://primary:4000/model/info"); + expect(calls).not.toContain("http://primary:4000/v1/models"); + expect(models).toHaveLength(1); + expect(models?.find(model => model.id === "vision-proxy-model")).toMatchObject({ + id: "vision-proxy-model", + name: "Vision Proxy Model", + input: ["text", "image"], + contextWindow: 64_000, + maxTokens: 8_000, + }); + }); + test("falls back from v2 model info to LiteLLM model info", async () => { const calls: string[] = []; const fetchMock = vi.fn(async (input: string | URL | Request) => { @@ -431,7 +550,9 @@ describe("LiteLLM provider discovery", () => { return new Response("{}", { status: 404 }); } if (url === "http://primary:4000/model/info") { - return Response.json({ data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000 } }] }); + return Response.json({ + data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000, supports_vision: false } }], + }); } throw new Error(`Unexpected URL: ${url}`); }) as FetchImpl; @@ -547,4 +668,53 @@ describe("LiteLLM provider discovery", () => { maxTokens: 8_192, }); }); + + test("enriches LiteLLM /v1/models fallback entries missing from models.dev with bundled reasoning metadata", async () => { + const calls: string[] = []; + const fetchMock = vi.fn(async (input: string | URL | Request) => { + const url = inputUrl(input); + calls.push(url); + if (url === MODELS_DEV_URL) { + return Response.json({}); + } + if ( + url === "http://primary:4000/model_group/info" || + url === "http://primary:4000/v2/model/info" || + url === "http://primary:4000/model/info" || + url === "http://primary:4000/v1/model/info" + ) { + return new Response("{}", { status: 404 }); + } + if (url === "http://primary:4000/v1/models") { + return Response.json({ data: [{ id: "glm-5.2" }] }); + } + throw new Error(`Unexpected URL: ${url}`); + }) as FetchImpl; + const options = litellmModelManagerOptions({ + apiKey: "sk-fallback", + baseUrl: "http://primary:4000/v1", + fetch: fetchMock, + }); + + const models = await options.fetchDynamicModels?.(); + + expect(calls).toContain("http://primary:4000/v1/models"); + expect(models).toHaveLength(1); + expect(models?.[0]).toMatchObject({ + id: "glm-5.2", + name: "GLM-5.2", + api: "openai-completions", + provider: "litellm", + baseUrl: "http://primary:4000/v1", + reasoning: true, + thinking: { + mode: "effort", + efforts: ["minimal", "low", "medium", "high", "xhigh"], + effortMap: { + minimal: "none", + xhigh: "max", + }, + }, + }); + }); }); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index d9bd88a42..fd139c905 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -7,17 +7,51 @@ - Added per-advisor on/off toggle (`enabled: false` in `WATCHDOG.yml`): advisors stay in the roster but their runtime is never built — they show `○` in the status line and `/advisor status` rather than disappearing. Existing configs are backward-compatible (defaults to `true` when absent). - Added per-advisor runtime status indicators in the status line (`●` running, `○` paused/no-model, `✕` error/quota-exhausted), truncated to 4 dots + `+` when the roster exceeds 4 advisors. - Added real provider quota display (usage percent, window, reset timer) to `/advisor status` and the `/advisor configure` preview. +- Typing `#` (e.g. `#3164`) in the prompt now offers PR and Issue autocomplete candidates that rewrite to the `pr://`/`issue://` internal URL, resolved from the current repo's git remote via the existing `read` tool → InternalUrlRouter → `gh` pipeline. Naming the type (`pr #3164` / `issue #3164`) constrains the candidates to that kind, and embedded hashes like `owner/repo#N`, `foo#N`, or URL fragments are left untouched ([#3218](https://github.com/can1357/oh-my-pi/issues/3218)) ### Changed - Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn. - Enriched `/advisor status` to show per-advisor status glyphs, model, spend breakdown, and quota window for every configured advisor (including disabled ones), replacing the previous single-advisor-only summary. + ### Fixed +- Improved handling of unawaited promises in JS eval cells to prevent process crashes +- Added warning logs for unhandled rejections originating from finished eval cells - Improved advisor robustness by blocking exhausted accounts during consecutive turn failures - - Fixed advisor turns hammering the same usage-limited account: a failed advisor turn now marks the exhausted credential blocked (with the provider's retry hint and usage-report reset time), so the next retry rotates to a sibling instead of re-picking the blocked account every few seconds. Previously the in-stream auth retry rotated within a request but never blocked the last failing credential, and the advisor loop — unlike the primary retry pipeline — never called `markUsageLimitReached`. - Added the account key to the `codex-auto-reset: skipped` debug log so skip reasons (e.g. `weekly-not-exhausted`) can be attributed to the evaluated account. +- Fixed unawaited promise rejections in JS eval cells crashing the session: a floating rejection now fails the owning cell run (`Unhandled rejection (missing await?): …`) instead of escaping to the global `unhandledRejection` handler, which printed `[Unhandled Rejection]` and killed the process (inline fallback) or tore down the eval worker (dedicated worker). Rejections surfacing after a cell settled are downgraded to a warn log attributed to the finished cell. +- Fixed project `.omp/RULES.md` sticky rules being shadowed by user `~/.omp/agent/RULES.md` rules with the same synthesized `RULES` name, so both user and project sticky rules now inject ([#4739](https://github.com/can1357/oh-my-pi/issues/4739)). +- Fixed bash internal-URL expansion so unresolved literal `memory://` / `skill://` text stays verbatim instead of aborting command execution ([#4737](https://github.com/can1357/oh-my-pi/issues/4737)). +- Fixed `agent://` (and the `output()` eval helper) failing with `Not found` for a subagent spawned by another subagent (any spawn chain 2+ levels deep). `artifactsDirsFromRegistry` scanned only each ref's adopted (root-wide) `ArtifactManager` dir, but a subagent's own children are written one level deeper under its `sessionFile`-derived dir — so a live, addressable nested peer's output was unresolvable. The resolver now collects both candidate dirs per registered agent. ([#4650](https://github.com/can1357/oh-my-pi/issues/4650)) +- Fixed plan mode to document `local://` artifacts as writable session-local planning files and to carry every pre-approval `local://` artifact into the fresh session created by Approve and Execute. +- Fixed the browser tool failing to launch Microsoft Edge-only Windows installs with Puppeteer's empty `Code: 0` launch error by keeping Edge's required `--enable-automation` default while preserving Chrome/Chromium stealth launch defaults. +- Fixed bash/tool command environments inheriting Bun-autoloaded launch `.env.local` values, so nested apps can load their own dotenv values without parent deployment variables taking precedence. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723)) +- Fixed legacy plugin validation for extension graphs that import JSON with `with { type: "json" }`, leaving JSON files on Bun's native loader instead of parsing them as JavaScript ([#4687](https://github.com/can1357/oh-my-pi/issues/4687)). +- Fixed pasted terminal transcripts beginning with a shell prompt (`$ ...`) being mistaken for local Python shortcuts instead of being submitted as normal prompts ([#4678](https://github.com/can1357/oh-my-pi/issues/4678)). +- Fixed wrapped OAuth copy-URL rows corrupting on paste: continuation chunks no longer carry a leading indent, so a multi-row terminal selection reassembles to the exact authorize URL (browsers strip newlines on paste but preserve or percent-encode embedded spaces, which previously corrupted the URL at every chunk boundary). +- Fixed Windows browser-launch failures being unobservable: the opener now uses `%SystemRoot%`-resolved PowerShell `Start-Process` (via `-EncodedCommand`) instead of `rundll32`, which exits 0 unconditionally. Failures ShellExecute itself reports — missing target, no handler executable, access denied — now surface as non-zero exits and are logged; the encoded payload also keeps OAuth query strings (`&`-bearing) opaque to shell metacharacter parsing. +- Fixed system prompt date rendering to use the host local calendar date instead of UTC. +- Fixed `bash` tool `timeout: 0` so it disables the command deadline instead of falling back to the minimum timeout. +- Fixed `read` and `grep` refusing to access filesystem paths whose names end in a selector-shaped suffix (e.g. `test:1-2`, `log:raw`) by preferring a literal match over the trailing `:` peel when the raw path exists on disk ([#4618](https://github.com/can1357/oh-my-pi/issues/4618)). +- Fixed wrapped Edit-diff rows leaking inverse video into the result card's right-edge padding: a row that broke inside an intra-line highlight left inverse active at the row end, so the frame padding after it rendered as a default-foreground block ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) +- Fixed Edit-diff continuation rows escaping into the line-number column when the row's gutter was left-padded (line number narrower than the widest in the diff) or blanked by the gutter dedup (the bare `+` row of a single-line replacement); such rows now wrap behind a continuation gutter, while body lines that merely start with `|` keep wrapping generically ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103)) +- Fixed live advisors continuing to use a stale `modelRoles.advisor` selection after `/model` changed the advisor model. ([#4612](https://github.com/can1357/oh-my-pi/issues/4612)) +- Fixed Claude plugin slash commands and skills silently vanishing when the plugin manifest declares `commands`/`slash-commands`/`skills` as a JSON array — the shape the Claude plugins reference documents and real plugins like `addyosmani/agent-skills` ship. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those fields as `string` and dropped array values on the floor; it now normalizes both shapes, loads every in-root entry, and reports one out-of-plugin-root warning per bad entry. The resolver also now honours Claude's per-field merge semantic — `skills` adds to the default `skills/` scan; `commands`/`slash-commands` replace the default `commands/` — so plugins like `{"skills":["./extra-skills"]}` no longer lose their default `skills/` folder while `{"commands":["./admin"]}` still replaces `commands/` as documented. ([#4609](https://github.com/can1357/oh-my-pi/issues/4609)) +- Fixed macOS Backspace on empty search not deleting sessions in the `/resume` picker; Fn+Backspace terminals that deliver `\x7f` instead of `\e[3~` now reach the delete confirmation dialog. ([#4580](https://github.com/can1357/oh-my-pi/pull/4580) by [@JagravNaik](https://github.com/JagravNaik)) +- Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal session title text. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) +- Fixed empty session `.jsonl` files accumulating in `~/.omp/agent/sessions//` after a draft-then-clear exit cycle. `SessionManager.saveDraft(text)` materializes the session file so the draft sidecar has a parent; a subsequent `saveDraft("")` unlinked the sidecar but left the metadata-only JSONL behind (title slot + session header + startup selector entries, ~500–750 B), and `#shouldHaveSessionFile()` could no longer prune it once `#fileIsCurrent`/`#forceFileCreation` were latched. `SessionManager.close()` now drops only draft-owned metadata-only sessions with no saved draft sidecar to reattach to, while keeping real conversations, meaningful non-message entries such as handoff custom messages, explicit `ensureOnDisk()` sessions, drafts still pending for `--resume`, and never-materialized sessions untouched ([#4571](https://github.com/can1357/oh-my-pi/issues/4571)). +- Fixed the advisor being disabled for the entire session when the advisor role resolves to a reasoning model that exposes no controllable effort surface (Devin `devin/glm-5-2*`: `reasoning: true`, `thinking: undefined` — Cascade routes by sibling model id rather than a wire param). `#resolveAdvisorRuntimeDescriptors` in `packages/coding-agent/src/session/agent-session.ts` used to hardcode `ThinkingLevel.Medium`, which tripped `requireSupportedEffort` on the first advisor prompt with `Thinking effort medium is not supported by devin/glm-5-2. Supported efforts:` (empty list). The advisor descriptor now clamps the requested effort against the resolved model via `resolveThinkingLevelForModel` and forwards no explicit effort when the model has no controllable efforts — matching the `auto`-path fix (`clampAutoThinkingEffort`) and the Autonomous Memory stage fix (`clampThinkingLevelForModel`). Explicit `:off` still disables reasoning, and models that support `medium` (e.g. Anthropic) keep receiving it ([#4579](https://github.com/can1357/oh-my-pi/issues/4579)). +- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports `calculateCost` (or `modelsAreEqual` / `getBundledProviders`) from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)). +- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports relocated catalog symbols such as `calculateCost`, `modelsAreEqual`, `getBundledProviders`, `getBundledModel`, or `getBundledModels` from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)). +- Fixed legacy pi extension imports of `DefaultResourceLoader` from `@mariozechner/pi-coding-agent` / `@earendil-works/pi-coding-agent` by adding a compatibility loader shim that translates `resourceLoader` into OMP's native session discovery options. ([#4567](https://github.com/can1357/oh-my-pi/issues/4567)) +- Fixed legacy Pi extension reloads on POSIX so `loadLegacyPiModule` imports the entry through a cache-busting filesystem path, refreshes load-time graph hooks when reloads add new modules, and threads the current load's `?mtime` tag through the extension source graph — relative `./helper.ts` siblings, `#alias/*` package-imports, extension-local bare dependency entries, and their relative children all rekey per reload, so same-process re-imports pick up edits across the whole graph. ([#4565](https://github.com/can1357/oh-my-pi/issues/4565)) +- Fixed bash tool pipeline execution preserving stale upstream output when the final stage was a stripped `head`/`tail` limiter; the tool now runs the command as written so `seq 1 5 | head -n2` returns only `1` and `2`. ([#4562](https://github.com/can1357/oh-my-pi/issues/4562)) +- Fixed the status-line token-rate segment rendering as `/s`, which Ghostty auto-detected as a hyperlink on Ctrl+hover. ([#4541](https://github.com/can1357/oh-my-pi/issues/4541)) +- Fixed retry fallback model recovery by exposing `retry.fallbackChains` in `/settings`, adding a `/model` action to assign the selected default fallback model, and clearing a selected model's retry cooldown marker on manual model switches. ([#4533](https://github.com/can1357/oh-my-pi/issues/4533)) +- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting cancellable `session_before_switch` hooks and a `session_switch` with `reason: "handoff"` after the replacement session is ready ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)). +- Fixed TTSR stream interrupts so only the tool call whose stream matched a rule receives the rule-named abort result; sibling tool-call placeholders now use a neutral abort reason ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)). ## [16.3.11] - 2026-07-06 diff --git a/packages/coding-agent/src/config/api-key-resolver.ts b/packages/coding-agent/src/config/api-key-resolver.ts index f06bccc7e..d237302cb 100644 --- a/packages/coding-agent/src/config/api-key-resolver.ts +++ b/packages/coding-agent/src/config/api-key-resolver.ts @@ -49,7 +49,7 @@ export function createApiKeyResolver( options: ApiKeyResolverOptions = {}, ): ApiKeyResolver { const { sessionId, baseUrl, modelId } = options; - return async ({ lastChance, error, signal }) => { + return async ({ lastChance, error, signal, previousKey }) => { if (error === undefined) { return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId }); } @@ -59,7 +59,12 @@ export function createApiKeyResolver( // sibling exists we switch immediately; the precise no-sibling backoff // is owned by `markUsageLimitReached` (default + server usage-report // reset) and the outer whole-turn retry layer. - await registry.authStorage.rotateSessionCredential(provider, sessionId, { error, modelId, signal }); + await registry.authStorage.rotateSessionCredential(provider, sessionId, { + error, + modelId, + signal, + apiKey: previousKey, + }); return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId }); } return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId, forceRefresh: true, signal }); diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index f1303a0e3..e567dc84b 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -2266,6 +2266,15 @@ export class ModelRegistry { return true; } + /** + * Clear the cooldown suppression for one selector after an explicit user selection. + */ + clearSuppressedSelector(selector: string): void { + this.#suppressedSelectors.delete( + normalizeSuppressedSelector(selector, (provider, id) => this.find(provider, id) !== undefined), + ); + } + /** * Clear all cooldown suppressions recorded via {@link suppressSelector}. * Used to reset retry-fallback cooldown state without a full {@link refresh}. diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index 00af4ed69..c632057bf 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -1367,7 +1367,17 @@ export const SETTINGS_SCHEMA = { description: "Allow retry recovery to switch to configured fallback models", }, }, - "retry.fallbackChains": { type: "record", default: {} as Record }, + "retry.fallbackChains": { + type: "record", + default: {} as Record, + ui: { + tab: "model", + group: "Retry & Fallback", + label: "Retry Fallback Chains", + description: + 'JSON object mapping model roles to ordered fallback model selectors, e.g. {"default":["openai/gpt-4o-mini"]}.', + }, + }, "retry.fallbackRevertPolicy": { type: "enum", values: ["cooldown-expiry", "never"] as const, diff --git a/packages/coding-agent/src/config/settings.ts b/packages/coding-agent/src/config/settings.ts index 0bde19651..c8aca9a30 100644 --- a/packages/coding-agent/src/config/settings.ts +++ b/packages/coding-agent/src/config/settings.ts @@ -429,6 +429,9 @@ export class Settings { if (path === "statusLine.sessionAccent") { statusLineSessionAccentSignal.fire(); } + if (path === "modelRoles") { + modelRolesSignal.fire(); + } } /** @@ -480,11 +483,13 @@ export class Settings { async reloadForCwd(cwd: string): Promise { const normalized = path.normalize(cwd); if (normalized === this.#cwd) return; + const prevModelRoles = this.get("modelRoles"); this.#cwd = normalized; if (this.#persist) { this.#project = await this.#loadProjectSettings(); } this.#rebuildMerged(); + this.#fireEffectiveSettingChanged("modelRoles", this.get("modelRoles"), prevModelRoles); this.#fireAllHooks(); } @@ -1477,6 +1482,12 @@ const appendOnlyModeSignal = new SettingSignal<[value: string]>("provider.append */ export const onAppendOnlyModeChanged = (cb: (value: string) => void) => appendOnlyModeSignal.on(cb); +/** Fires when any model role changes at runtime. */ +const modelRolesSignal = new SettingSignal("modelRoles"); + +/** Subscribe to model role changes. Returns an unsubscribe function. */ +export const onModelRolesChanged: (cb: () => void) => () => void = modelRolesSignal.on.bind(modelRolesSignal); + /** Fires when `statusLine.sessionAccent` changes at runtime. */ const statusLineSessionAccentSignal = new SettingSignal("statusLine.sessionAccent"); diff --git a/packages/coding-agent/src/discovery/builtin.ts b/packages/coding-agent/src/discovery/builtin.ts index 94a17e2bc..42abac1d5 100644 --- a/packages/coding-agent/src/discovery/builtin.ts +++ b/packages/coding-agent/src/discovery/builtin.ts @@ -401,7 +401,8 @@ async function loadStickyRulesFile(filePath: string, level: "user" | "project"): const content = await readFile(filePath); if (!content) return null; const source = createSourceMeta(PROVIDER_ID, filePath, level); - const rule = buildRuleFromMarkdown("RULES.md", content, filePath, source, { ruleName: "RULES" }); + const ruleName = level === "project" ? "RULES@project" : "RULES"; + const rule = buildRuleFromMarkdown("RULES.md", content, filePath, source, { ruleName }); // Force alwaysApply regardless of frontmatter — the whole point of RULES.md // is to be reattached every turn. return { ...rule, alwaysApply: true }; diff --git a/packages/coding-agent/src/discovery/claude-plugins.ts b/packages/coding-agent/src/discovery/claude-plugins.ts index c980eec98..3bb09292d 100644 --- a/packages/coding-agent/src/discovery/claude-plugins.ts +++ b/packages/coding-agent/src/discovery/claude-plugins.ts @@ -4,6 +4,7 @@ * Loads configuration from ~/.claude/plugins/cache/ based on installed_plugins.json registry. * Priority: 70 (below claude.ts at 80, so user overrides in .claude/ take precedence) */ +import * as fs from "node:fs/promises"; import * as path from "node:path"; import { logger } from "@oh-my-pi/pi-utils"; import { registerProvider } from "../capability"; @@ -30,14 +31,14 @@ const DISPLAY_NAME = "Claude Code Marketplace"; const PRIORITY = 70; // Below claude.ts (80) so user .claude/ overrides win interface ClaudePluginManifest { - skills?: string; - "slash-commands"?: string; - commands?: string; + skills?: string | string[]; + "slash-commands"?: string | string[]; + commands?: string | string[]; } interface ResolvedPluginDir { - dir: string; - warning?: string; + dirs: string[]; + warnings: string[]; } async function readPluginManifest(root: ClaudePluginRoot): Promise { @@ -54,43 +55,116 @@ async function readPluginManifest(root: ClaudePluginRoot): Promise { + return value !== null && typeof value === "object" && !Array.isArray(value); +} + +async function skillsManifestReplacesFallback(root: ClaudePluginRoot): Promise { + const raw = await readFile(path.join(root.path, "marketplace.json")); + if (raw === null) return false; + + try { + const parsed: unknown = JSON.parse(raw); + if (!isRecord(parsed)) return false; + const plugins = parsed.plugins; + return ( + Array.isArray(plugins) && + plugins.some(entry => isRecord(entry) && entry.name === root.plugin && entry.source === "./") + ); + } catch { + return false; + } +} + function isWithinPluginRoot(rootPath: string, targetPath: string): boolean { const relative = path.relative(rootPath, targetPath); return relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative)); } +/** + * Resolve a manifest-declared directory field to absolute paths within the + * plugin root. + * + * Manifest path fields may be `string` or `string[]` + * (https://code.claude.com/docs/en/plugins-reference#path-behavior-rules); + * both shapes are normalized here. The first `manifestKeys` entry that + * supplies at least one non-empty path wins (later keys are ignored — used for + * the `commands` > `slash-commands` legacy fallback). + * + * `fallback` is the default subdirectory (e.g. `skills/`, `commands/`) and + * `includeFallback` controls the Claude-documented merge semantic per field: + * + * - `skills` **adds to** the default: `fallback` is always scanned, and any + * manifest entries load alongside it. Callers pass `includeFallback: true`. + * - `commands` / `slash-commands` **replace** the default: an explicit + * manifest key means the default `commands/` directory is not scanned. + * Callers pass `includeFallback: false` (the manifest itself may still + * list `./commands` explicitly to keep it). + * + * When no matching key is set, the fallback is used regardless. Entries that + * resolve outside the plugin root are dropped with a warning so misconfigured + * manifests remain observable and cannot escape via traversal. + */ async function resolvePluginDir( root: ClaudePluginRoot, manifestKeys: ReadonlyArray, fallback: string, + includeFallback: boolean, ): Promise { const manifest = await readPluginManifest(root); const fallbackDir = path.join(root.path, fallback); - let configured: string | undefined; + let configured: string[] | undefined; let matchedKey: keyof ClaudePluginManifest | undefined; for (const key of manifestKeys) { const val = manifest?.[key]; - if (typeof val === "string" && val.trim()) { - configured = val.trim(); + const candidates: string[] = []; + if (typeof val === "string") { + const trimmed = val.trim(); + if (trimmed) candidates.push(trimmed); + } else if (Array.isArray(val)) { + for (const entry of val) { + if (typeof entry !== "string") continue; + const trimmed = entry.trim(); + if (trimmed) candidates.push(trimmed); + } + } + if (candidates.length > 0) { + configured = candidates; matchedKey = key; break; } } if (configured === undefined) { - return { dir: fallbackDir }; + return { dirs: [fallbackDir], warnings: [] }; } - const resolved = path.resolve(root.path, configured); - if (isWithinPluginRoot(root.path, resolved)) { - return { dir: resolved }; + // Dedup preserves order: default entry (when included) first, then declared + // entries in manifest order. Deduping the paths themselves means a plugin + // author can still list `./commands` explicitly when they want the default + // alongside extras without producing double-loads. + const seen = new Set(); + const dirs: string[] = []; + const warnings: string[] = []; + if (includeFallback) { + seen.add(fallbackDir); + dirs.push(fallbackDir); + } + for (const entry of configured) { + const resolved = path.resolve(root.path, entry); + if (!isWithinPluginRoot(root.path, resolved)) { + warnings.push( + `[claude-plugins] Ignoring ${String(matchedKey)} path outside plugin root for ${root.id}: ${entry}`, + ); + continue; + } + if (seen.has(resolved)) continue; + seen.add(resolved); + dirs.push(resolved); } - return { - dir: fallbackDir, - warning: `[claude-plugins] Ignoring ${String(matchedKey)} path outside plugin root for ${root.id}: ${configured}`, - }; + return { dirs, warnings }; } // ============================================================================= @@ -104,24 +178,37 @@ async function loadSkills(ctx: LoadContext): Promise> { warnings.push(...rootWarnings); const results = await Promise.all( roots.map(async root => { - const { dir: skillsDir, warning } = await resolvePluginDir(root, ["skills"], "skills"); - const result = await scanSkillsFromDir(ctx, { - dir: skillsDir, - providerId: PROVIDER_ID, - level: root.scope, - }); - return { root, result, warning }; + const includeFallback = !(await skillsManifestReplacesFallback(root)); + const { dirs: skillsDirs, warnings: resolveWarnings } = await resolvePluginDir( + root, + ["skills"], + "skills", + includeFallback, + ); + const scanResults = await Promise.all( + skillsDirs.map(dir => + scanSkillsFromDir(ctx, { + dir, + providerId: PROVIDER_ID, + level: root.scope, + includeSelf: true, + }), + ), + ); + return { scanResults, resolveWarnings }; }), ); - for (const { result, warning } of results) { - if (warning) warnings.push(warning); + for (const { scanResults, resolveWarnings } of results) { + warnings.push(...resolveWarnings); // Intentionally do NOT prefix skill names with `root.plugin`. // The `plugin:name` format breaks skill:// URL parsing (colons are // ambiguous with port separators) and is unintuitive for callers. // Dedup-by-key in the capability layer already handles name collisions // across providers using priority ordering. - items.push(...result.items); - if (result.warnings) warnings.push(...result.warnings); + for (const result of scanResults) { + items.push(...result.items); + if (result.warnings) warnings.push(...result.warnings); + } } return { items, warnings }; } @@ -139,28 +226,62 @@ async function loadSlashCommands(ctx: LoadContext): Promise { - const { dir: commandsDir, warning } = await resolvePluginDir(root, ["commands", "slash-commands"], "commands"); - const commandResult = await loadFilesFromDir(ctx, commandsDir, PROVIDER_ID, root.scope, { - extensions: ["md"], - transform: (name, content, filePath, source) => { - const cmdName = name.replace(/\.md$/, ""); - return { - name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName, - path: filePath, - content, - level: root.scope, - _source: source, - }; - }, - }); - return { commandResult, warning }; + const { dirs: commandsDirs, warnings: resolveWarnings } = await resolvePluginDir( + root, + ["commands", "slash-commands"], + "commands", + false, + ); + const commandResults = await Promise.all( + commandsDirs.map(async dir => { + try { + const stats = await fs.stat(dir); + if (stats.isFile()) { + if (path.extname(dir) !== ".md") return { items: [], warnings: [] }; + const content = await readFile(dir); + if (content === null) return { items: [], warnings: [`Failed to read file: ${dir}`] }; + const cmdName = path.basename(dir).replace(/\.md$/, ""); + return { + items: [ + { + name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName, + path: dir, + content, + level: root.scope, + _source: createSourceMeta(PROVIDER_ID, dir, root.scope), + }, + ], + warnings: [], + }; + } + } catch { + // Missing entries behave like missing directories: no items, no warning. + } + return loadFilesFromDir(ctx, dir, PROVIDER_ID, root.scope, { + extensions: ["md"], + transform: (name, content, filePath, source) => { + const cmdName = name.replace(/\.md$/, ""); + return { + name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName, + path: filePath, + content, + level: root.scope, + _source: source, + }; + }, + }); + }), + ); + return { commandResults, resolveWarnings }; }), ); - for (const { commandResult, warning } of results) { - if (warning) warnings.push(warning); - items.push(...commandResult.items); - if (commandResult.warnings) warnings.push(...commandResult.warnings); + for (const { commandResults, resolveWarnings } of results) { + warnings.push(...resolveWarnings); + for (const commandResult of commandResults) { + items.push(...commandResult.items); + if (commandResult.warnings) warnings.push(...commandResult.warnings); + } } return { items, warnings }; diff --git a/packages/coding-agent/src/discovery/helpers.ts b/packages/coding-agent/src/discovery/helpers.ts index d1e6e7c91..8535fa0ad 100644 --- a/packages/coding-agent/src/discovery/helpers.ts +++ b/packages/coding-agent/src/discovery/helpers.ts @@ -312,6 +312,15 @@ export interface ScanSkillsFromDirOptions { providerId: string; level: "user" | "project"; requireDescription?: boolean; + /** + * When true, treat a `SKILL.md` sitting directly under `dir` as a single skill in addition to + * scanning `//SKILL.md` children. Matches the Claude plugin manifest convention + * that lets a skill path point at a directory containing `SKILL.md` directly (e.g. + * `"skills": ["./"]`), where the frontmatter `name` determines the invocation name and the + * directory basename is the fallback. Default `false` preserves the strict child-scan + * semantic every non-Claude provider relies on. + */ + includeSelf?: boolean; } // Stable ordering used for skill lists in prompts: name (case-insensitive), then name, then path. @@ -368,7 +377,13 @@ export async function scanSkillsFromDir( } }; - const work = []; + const work: Promise[] = []; + if (options.includeSelf) { + const selfSkillPath = path.join(dir, "SKILL.md"); + if (fs.existsSync(selfSkillPath)) { + work.push(loadSkill(selfSkillPath)); + } + } for (const entry of entries) { if (entry.name.startsWith(".")) continue; if (!entry.isDirectory() && !entry.isSymbolicLink()) continue; diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index e02af1eb7..14682a59e 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -679,21 +679,35 @@ function wrapEditRendererLine(line: string, width: number): string[] { const startAnsi = line.match(/^((?:\x1b\[[0-9;]*m)*)/)?.[1] ?? ""; const bodyWithReset = line.slice(startAnsi.length); const body = bodyWithReset.endsWith("\x1b[39m") ? bodyWithReset.slice(0, -"\x1b[39m".length) : bodyWithReset; - const diffMatch = /^([+\-\s])(\s*\d+)([|│])(.*)$/s.exec(body); + // Gutter shapes produced by formatCodeFrameLine: "-315│", " 313│", "+322│", + // plus the deduplicated forms " +│" and " │" whose repeated line number + // renderDiff blanked (single-line replacement pairs and insert-then-context + // runs) — all │-separated. ASCII "|" gutters exist only in raw canonical + // diff rows passed through by the plain fallback ("-42|old", " 42|ctx"), + // which always carry a marker column ("+"/"-"/space) and a line number. So + // the number is optional for "│", while "|" requires the full canonical + // shape; anything else (a body line merely starting with "|", error text + // like "123|…") is not a diff row and wraps generically. + const diffMatch = /^(\s*[+-]?\s*\d*)([|│])(.*)$/s.exec(body); - if (!diffMatch) { + if (!diffMatch || diffMatch[1].length === 0 || (diffMatch[2] === "|" && !/^[+\-\s]\s*\d+$/.test(diffMatch[1]))) { return wrapTextWithAnsi(line, width); } - const [, marker, lineNum, separator, content] = diffMatch; - const prefix = `${marker}${lineNum}${separator}`; + const [, gutter, separator, content] = diffMatch; + const prefix = `${gutter}${separator}`; const prefixWidth = visibleWidth(prefix); const contentWidth = Math.max(1, width - prefixWidth); const continuationPrefix = `${" ".repeat(Math.max(0, prefixWidth - 1))}${separator}`; const wrappedContent = wrapTextWithAnsi(content ?? "", contentWidth); + // Each visual row is a standalone terminal line: wrapTextWithAnsi re-opens + // active SGR state at the next row's start, so a row that breaks inside an + // intra-line diff highlight still ends with inverse video active. Close it + // alongside the foreground reset — otherwise the frame padding appended + // after the row is painted as an inverse block (default-foreground cells). return wrappedContent.map( - (segment, index) => `${startAnsi}${index === 0 ? prefix : continuationPrefix}${segment}\x1b[39m`, + (segment, index) => `${startAnsi}${index === 0 ? prefix : continuationPrefix}${segment}\x1b[27m\x1b[39m`, ); } diff --git a/packages/coding-agent/src/eval/js/worker-core.ts b/packages/coding-agent/src/eval/js/worker-core.ts index 4b7933d3c..c23517ee7 100644 --- a/packages/coding-agent/src/eval/js/worker-core.ts +++ b/packages/coding-agent/src/eval/js/worker-core.ts @@ -1,6 +1,15 @@ +import { isMainThread } from "node:worker_threads"; +import { postmortem } from "@oh-my-pi/pi-utils"; import { ToolError } from "../../tools/tool-errors"; import { JsRuntime, type RuntimeHooks } from "./shared/runtime"; -import type { RunErrorPayload, SessionSnapshot, ToolReply, Transport, WorkerInbound } from "./worker-protocol"; +import type { + RunErrorPayload, + SessionSnapshot, + ToolReply, + Transport, + WorkerInbound, + WorkerOutbound, +} from "./worker-protocol"; interface PendingTool { runId: string; @@ -10,9 +19,17 @@ interface PendingTool { interface ActiveRun { runId: string; + filename: string; pendingTools: Map; + /** Rejections floated by this run's cell code, captured before its result was sent. */ + floatingRejections: unknown[]; } +type RunResult = Extract; + +/** Finished-cell filenames retained for attributing rejections that surface after the run settled. */ +const RECENT_CELL_FILES_MAX = 256; + function errorPayload(error: unknown): RunErrorPayload { if (error instanceof Error) { return { @@ -34,15 +51,134 @@ function errorFromPayload(payload: RunErrorPayload): Error { return error; } +/** + * Fold rejections floated by cell code into the run result: an otherwise + * successful run fails with the first floating rejection (an unawaited promise + * failing is a cell failure, not a success with noise); the rest surface as + * output text so nothing is silently dropped. + */ +function foldFloatingRejections(active: ActiveRun, result: RunResult, hooks: RuntimeHooks): RunResult { + const rejections = active.floatingRejections; + if (rejections.length === 0) return result; + let folded = result; + let reported = rejections; + if (result.ok) { + const error = errorPayload(rejections[0]); + error.message = `Unhandled rejection (missing await?): ${error.message}`; + folded = { type: "result", runId: active.runId, ok: false, error }; + reported = rejections.slice(1); + } + for (const reason of reported) { + const payload = errorPayload(reason); + hooks.onText(`[unhandled rejection] ${payload.name ?? "Error"}: ${payload.message}\n`); + } + return folded; +} + export class WorkerCore { #transport: Transport; #runtime: JsRuntime | null = null; #runs = new Map(); + #recentCellFiles = new Set(); #unsubscribe: () => void; + #uninstallRejectionGuard: () => void; constructor(transport: Transport) { this.#transport = transport; this.#unsubscribe = transport.onMessage(msg => this.#handle(msg)); + this.#uninstallRejectionGuard = this.#installRejectionGuard(); + } + + /** + * Capture unhandled rejections floated by eval-cell code (unawaited async + * calls) so they fail the owning run instead of tearing down the worker or — + * via the global postmortem handler — the whole session. On the main thread + * (inline fallback) only cell-attributable rejections are consumed; in the + * dedicated worker realm a rejection during a live run is cell activity even + * without a usable stack, while anything else keeps its default fatality. + */ + #installRejectionGuard(): () => void { + if (isMainThread) { + return postmortem.interceptUnhandledRejections(reason => this.#consumeRejection(reason)); + } + const onRejection = (reason: unknown): void => { + if (this.#consumeRejection(reason)) return; + // Not cell-attributable: restore default fatality. Rethrowing from a + // timer surfaces it as an uncaught exception, which reaches the host + // as a worker `error` event exactly like an unhandled rejection did + // before this listener existed. + setTimeout(() => { + throw reason; + }, 0); + }; + process.on("unhandledRejection", onRejection); + return () => { + process.off("unhandledRejection", onRejection); + }; + } + + /** + * Attribute an unhandled rejection to eval-cell code. Live runs are stashed + * on the run (folded into its result after the settle drain); finished cells + * downgrade to a host-side warn log. Returns false when the rejection is not + * cell activity and must keep the default fatal path. + */ + #consumeRejection(reason: unknown): boolean { + const stack = reason instanceof Error && typeof reason.stack === "string" ? reason.stack : undefined; + if (stack) { + // The stack can name several cells (helper defined by an earlier cell, + // called from the live one); the outermost matching frame is the caller + // that owns the floating promise. + let owner: ActiveRun | undefined; + let ownerIndex = -1; + for (const run of this.#runs.values()) { + const index = stack.lastIndexOf(run.filename); + if (index > ownerIndex) { + ownerIndex = index; + owner = run; + } + } + if (owner) { + owner.floatingRejections.push(reason); + return true; + } + let recent: string | undefined; + let recentIndex = -1; + for (const filename of this.#recentCellFiles) { + const index = stack.lastIndexOf(filename); + if (index > recentIndex) { + recentIndex = index; + recent = filename; + } + } + if (recent) { + this.#transport.send({ + type: "log", + level: "warn", + msg: "Unhandled rejection from a finished eval cell (missing await?)", + meta: { filename: recent, error: errorPayload(reason) }, + }); + return true; + } + } + if (!isMainThread && this.#runs.size > 0) { + // Dedicated eval worker: during a live run, a rejection without a cell + // frame (e.g. `Promise.reject("msg")` or a library-created reason) is + // still cell activity — nothing else runs user code in this realm. + if (this.#runs.size === 1) { + const only = this.#runs.values().next().value; + only?.floatingRejections.push(reason); + return true; + } + this.#transport.send({ + type: "log", + level: "warn", + msg: "Unhandled rejection during concurrent eval runs; cannot attribute to a cell", + meta: { error: errorPayload(reason) }, + }); + return true; + } + return false; } #handle(msg: WorkerInbound): void { @@ -77,23 +213,42 @@ export class WorkerCore { } async #runOne(runId: string, code: string, filename: string, snapshot: SessionSnapshot): Promise { - const runtime = this.#ensureRuntime(snapshot); - runtime.setCwd(snapshot.cwd); - const active: ActiveRun = { runId, pendingTools: new Map() }; + const active: ActiveRun = { runId, filename, pendingTools: new Map(), floatingRejections: [] }; this.#runs.set(runId, active); const hooks: RuntimeHooks = { onText: chunk => this.#transport.send({ type: "text", runId, chunk }), onDisplay: output => this.#transport.send({ type: "display", runId, output }), callTool: (name, args) => this.#callTool(active, name, args), }; + let result: RunResult; try { + const runtime = this.#ensureRuntime(snapshot); + runtime.setCwd(snapshot.cwd); const value = await runtime.run(code, filename, hooks, { runId, cwd: snapshot.cwd }); runtime.displayValue(value, hooks); - this.#transport.send({ type: "result", runId, ok: true }); + result = { type: "result", runId, ok: true }; } catch (error) { - this.#transport.send({ type: "result", runId, ok: false, error: errorPayload(error) }); + result = { type: "result", runId, ok: false, error: errorPayload(error) }; + } + try { + // One event-loop turn so rejections the cell already floated surface + // while this run can still own them (rejection callbacks run before + // timers fire). + await Bun.sleep(0); + result = foldFloatingRejections(active, result, hooks); } finally { this.#runs.delete(runId); + this.#rememberCellFile(filename); + this.#transport.send(result); + } + } + + #rememberCellFile(filename: string): void { + this.#recentCellFiles.delete(filename); + this.#recentCellFiles.add(filename); + if (this.#recentCellFiles.size > RECENT_CELL_FILES_MAX) { + const oldest = this.#recentCellFiles.values().next().value; + if (oldest !== undefined) this.#recentCellFiles.delete(oldest); } } @@ -127,6 +282,7 @@ export class WorkerCore { this.#runtime?.dispose?.(); this.#runtime = null; this.#transport.send({ type: "closed" }); + this.#uninstallRejectionGuard(); this.#unsubscribe(); this.#transport.close(); } @@ -141,6 +297,7 @@ export class WorkerCore { this.#runs.clear(); this.#runtime?.dispose?.(); this.#runtime = null; + this.#uninstallRejectionGuard(); this.#unsubscribe(); try { this.#transport.close(); diff --git a/packages/coding-agent/src/exec/bash-executor.ts b/packages/coding-agent/src/exec/bash-executor.ts index dfdcee8a3..fc7d7dc69 100644 --- a/packages/coding-agent/src/exec/bash-executor.ts +++ b/packages/coding-agent/src/exec/bash-executor.ts @@ -14,6 +14,7 @@ import { buildNonInteractiveEnv } from "./non-interactive-env"; export interface BashExecutorOptions { cwd?: string; + /** Milliseconds before aborting the command; 0 disables the executor deadline. */ timeout?: number; onChunk?: (chunk: string) => void; chunkThrottleMs?: number; @@ -296,11 +297,15 @@ export async function executeBash(command: string, options?: BashExecutorOptions let timeoutTimer: NodeJS.Timeout | undefined; const timeoutDeferred = Promise.withResolvers<"timeout">(); - const baseTimeoutMs = Math.max(1_000, options?.timeout ?? 300_000); - timeoutTimer = setTimeout(() => { - abortCurrentExecution(); - timeoutDeferred.resolve("timeout"); - }, baseTimeoutMs); + const requestedTimeoutMs = options?.timeout; + const deadlineTimeoutMs = requestedTimeoutMs === 0 ? undefined : Math.max(1_000, requestedTimeoutMs ?? 300_000); + const nativeTimeoutMs = requestedTimeoutMs !== undefined && requestedTimeoutMs > 0 ? requestedTimeoutMs : undefined; + if (deadlineTimeoutMs !== undefined) { + timeoutTimer = setTimeout(() => { + abortCurrentExecution(); + timeoutDeferred.resolve("timeout"); + }, deadlineTimeoutMs); + } let resetSession = false; @@ -311,7 +316,7 @@ export async function executeBash(command: string, options?: BashExecutorOptions command: finalCommand, cwd: commandCwd, env: commandEnv, - timeoutMs: options?.timeout, + timeoutMs: nativeTimeoutMs, signal: runAbortController.signal, }, (err, chunk) => { @@ -328,7 +333,7 @@ export async function executeBash(command: string, options?: BashExecutorOptions sessionEnv: shellEnv, snapshotPath: snapshotPath ?? undefined, minimizer, - timeoutMs: options?.timeout, + timeoutMs: nativeTimeoutMs, signal: runAbortController.signal, }, (err, chunk) => { @@ -359,8 +364,8 @@ export async function executeBash(command: string, options?: BashExecutorOptions exitCode: undefined, cancelled: true, ...(await sink.dump( - winner.kind === "timeout" - ? `Command timed out after ${Math.round(baseTimeoutMs / 1000)} seconds` + winner.kind === "timeout" && deadlineTimeoutMs !== undefined + ? `Command timed out after ${Math.round(deadlineTimeoutMs / 1000)} seconds` : "Command cancelled", )), }; diff --git a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts index 1399be92e..20f55ad19 100644 --- a/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts +++ b/packages/coding-agent/src/extensibility/plugins/legacy-pi-compat.ts @@ -1059,7 +1059,8 @@ async function collectExtensionModules(entryRealPath: string): Promise { +async function readMarketplaceCatalog( + root: string, + options: { relativeDisplayPaths?: boolean } = {}, +): Promise<{ catalogPath: string; displayPath: string; content: string }> { const tried: string[] = []; for (const rel of CATALOG_RELATIVE_PATHS) { - const catalogPath = path.join(root, rel); - tried.push(catalogPath); + const catalogPath = path.join(root, ...rel.split("/")); + const displayPath = options.relativeDisplayPaths ? rel : catalogPath; + tried.push(displayPath); try { const content = await Bun.file(catalogPath).text(); - return { catalogPath, content }; + return { catalogPath, displayPath, content }; } catch (err) { if (isEnoent(err)) continue; throw err; @@ -252,11 +253,11 @@ export async function fetchMarketplace(source: string, cacheDir: string): Promis if (type === "github") { const url = `https://github.com/${source}.git`; - return cloneAndReadCatalog(url, cacheDir); + return cloneAndReadCatalog(url, source, cacheDir); } if (type === "git") { - return cloneAndReadCatalog(source, cacheDir); + return cloneAndReadCatalog(source, source, cacheDir); } // type === "url" @@ -284,7 +285,7 @@ export async function fetchMarketplace(source: string, cacheDir: string): Promis * responsible for promoting the clone to its final cache location via * `promoteCloneToCache` after any duplicate/drift checks pass. */ -async function cloneAndReadCatalog(url: string, cacheDir: string): Promise { +async function cloneAndReadCatalog(url: string, source: string, cacheDir: string): Promise { const tmpDir = path.join(cacheDir, `.tmp-clone-${Date.now()}`); await fs.mkdir(cacheDir, { recursive: true }); @@ -292,12 +293,12 @@ async function cloneAndReadCatalog(url: string, cacheDir: string): Promise {}); - throw new Error(`Cloned repository ${url}: ${(err as Error).message}`, { cause: err }); + throw new Error(`Cloned repository ${url}: ${(err as Error).message} (source: ${source})`, { cause: err }); } } diff --git a/packages/coding-agent/src/extensibility/shared-events.ts b/packages/coding-agent/src/extensibility/shared-events.ts index e3bb7dcbb..e1d22f50c 100644 --- a/packages/coding-agent/src/extensibility/shared-events.ts +++ b/packages/coding-agent/src/extensibility/shared-events.ts @@ -33,7 +33,7 @@ export interface SessionStartEvent { export interface SessionBeforeSwitchEvent { type: "session_before_switch"; /** Reason for the switch */ - reason: "new" | "resume" | "fork"; + reason: "new" | "resume" | "fork" | "handoff"; /** Session file we're switching to (only for "resume") */ targetSessionFile?: string; } @@ -42,7 +42,7 @@ export interface SessionBeforeSwitchEvent { export interface SessionSwitchEvent { type: "session_switch"; /** Reason for the switch */ - reason: "new" | "resume" | "fork"; + reason: "new" | "resume" | "fork" | "handoff"; /** Session file we came from */ previousSessionFile: string | undefined; } diff --git a/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts b/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts new file mode 100644 index 000000000..7829557f5 --- /dev/null +++ b/packages/coding-agent/src/internal-urls/__tests__/agent-protocol-nested.test.ts @@ -0,0 +1,68 @@ +import { afterAll, afterEach, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as path from "node:path"; +import { TempDir } from "@oh-my-pi/pi-utils"; +import { AgentRegistry } from "../../registry/agent-registry"; +import type { AgentSession } from "../../session/agent-session"; +import { ArtifactManager } from "../../session/artifacts"; +import { AgentProtocolHandler } from "../agent-protocol"; +import { resetRegisteredArtifactDirsForTests } from "../registry-helpers"; + +const tempDir = TempDir.createSync("omp-nested-agent-repro-"); +afterEach(() => { + AgentRegistry.resetGlobalForTests(); + resetRegisteredArtifactDirsForTests(); +}); +afterAll(() => { + tempDir.removeSync(); +}); + +it("agent:// resolves a depth-2 subagent's .md output while its session is live and artifact-manager-adopted", async () => { + const root = tempDir.path(); + const rootSessionFile = path.join(root, "session.jsonl"); + const rootArtifactsDir = rootSessionFile.slice(0, -6); + await fs.mkdir(rootArtifactsDir, { recursive: true }); + // Every subagent adopts the root ArtifactManager and reports its dir. + const sharedArtifactManager = new ArtifactManager(rootArtifactsDir); + + // A depth-1 subagent's OWN children are written under its own + // sessionFile.slice(0, -6) (task/index.ts), i.e. one level deeper. + const midSessionFile = path.join(rootArtifactsDir, "CodexDeepDive.jsonl"); + const midOwnArtifactsDir = midSessionFile.slice(0, -6); + await fs.mkdir(midOwnArtifactsDir, { recursive: true }); + + const grandchildId = "CodexDeepDive.GraphStore"; + const grandchildSessionFile = path.join(midOwnArtifactsDir, `${grandchildId}.jsonl`); + await fs.writeFile(path.join(midOwnArtifactsDir, `${grandchildId}.md`), "full report content"); + + const fakeSession = { + sessionManager: { getArtifactsDir: () => sharedArtifactManager.dir }, + } as unknown as AgentSession; + const registry = AgentRegistry.global(); + registry.register({ + id: "Main", + displayName: "main", + kind: "main", + session: fakeSession, + sessionFile: rootSessionFile, + }); + registry.register({ + id: "CodexDeepDive", + displayName: "sub", + kind: "sub", + parentId: "Main", + session: fakeSession, + sessionFile: midSessionFile, + }); + registry.register({ + id: grandchildId, + displayName: "sub", + kind: "sub", + parentId: "CodexDeepDive", + session: fakeSession, + sessionFile: grandchildSessionFile, + }); + + const resource = await new AgentProtocolHandler().resolve(new URL(`agent://${grandchildId}`) as never); + expect(resource.content).toBe("full report content"); +}); diff --git a/packages/coding-agent/src/internal-urls/registry-helpers.ts b/packages/coding-agent/src/internal-urls/registry-helpers.ts index 02648f76a..701ed685f 100644 --- a/packages/coding-agent/src/internal-urls/registry-helpers.ts +++ b/packages/coding-agent/src/internal-urls/registry-helpers.ts @@ -20,11 +20,13 @@ export function resetRegisteredArtifactDirsForTests(): void { /** * Snapshot of artifacts dirs for every registered session, deduped. * - * Prefers `sessionManager.getArtifactsDir()` because subagents adopt their - * parent's `ArtifactManager` and report the parent's dir there; dedup then - * collapses parent + N subagents (the whole agent tree) to one entry. Falls - * back to the raw session file (with the `.jsonl` suffix stripped) when no - * live session reference is attached. + * Collects TWO candidate dirs per ref, because a subagent reads from its + * adopted (root-wide) `ArtifactManager.dir` but its own children are written + * one level deeper, under `sessionFile.slice(0, -6)` (`task/index.ts`). A + * depth-2+ subagent's output therefore lives in the write-time dir, not the + * adopted one, so `agent://` must scan both or it 404s a live nested peer. + * `addDir` dedup collapses the depth-0 case (both formulas agree) back to a + * single entry. */ export function artifactsDirsFromRegistry(): string[] { const dirs: string[] = []; @@ -33,7 +35,8 @@ export function artifactsDirsFromRegistry(): string[] { if (!dirs.includes(dir)) dirs.push(dir); }; for (const ref of AgentRegistry.global().list()) { - addDir(ref.session?.sessionManager.getArtifactsDir() ?? (ref.sessionFile ? ref.sessionFile.slice(0, -6) : null)); + addDir(ref.session?.sessionManager.getArtifactsDir()); + if (ref.sessionFile) addDir(ref.sessionFile.slice(0, -6)); } for (const dir of extraArtifactsDirs) addDir(dir); return dirs; diff --git a/packages/coding-agent/src/modes/components/model-selector.ts b/packages/coding-agent/src/modes/components/model-selector.ts index aa6e0c451..156ba7d81 100644 --- a/packages/coding-agent/src/modes/components/model-selector.ts +++ b/packages/coding-agent/src/modes/components/model-selector.ts @@ -90,16 +90,20 @@ interface RoleAssignment { autoSelected: boolean; } +type ModelSelectorAction = "modelRole" | "retryFallback"; + type RoleSelectCallback = ( model: Model, role: string | null, thinkingLevel?: ConfiguredThinkingLevel, selector?: string, + action?: ModelSelectorAction, ) => void; type CancelCallback = () => void; interface MenuRoleAction { label: string; - role: string; // now accepts custom role strings + role: string; + action: ModelSelectorAction; } interface ProviderTabState { @@ -284,14 +288,19 @@ export class ModelSelectorComponent extends Container { } #buildMenuRoleActions(): void { - this.#menuRoleActions = getKnownRoleIds(this.#settings).map(role => { + const roleActions = getKnownRoleIds(this.#settings).map(role => { const roleInfo = getRoleInfo(role, this.#settings); const roleLabel = roleInfo.tag ? `${roleInfo.tag} (${roleInfo.name})` : roleInfo.name; return { label: `Set as ${roleLabel}`, role, + action: "modelRole" as const, }; }); + this.#menuRoleActions = [ + ...roleActions, + { label: "Set as DEFAULT retry fallback", role: "default", action: "retryFallback" }, + ]; } #loadRoleModels(autoCandidateModels?: ReadonlyArray): void { @@ -1195,6 +1204,11 @@ export class ModelSelectorComponent extends Container { if (this.#menuStep === "role") { const action = this.#menuRoleActions[this.#menuSelectedIndex]; if (!action) return; + if (action.action === "retryFallback") { + this.#handleSelect(selectedItem, action.role, undefined, action.action); + this.#closeMenu(); + return; + } this.#menuSelectedRole = action.role; this.#menuStep = "thinking"; this.#menuSelectedIndex = this.#getThinkingPreselectIndex(action.role, selectedItem.model); @@ -1206,7 +1220,7 @@ export class ModelSelectorComponent extends Container { const thinkingOptions = this.#getThinkingLevelsForModel(selectedItem.model); const thinkingLevel = thinkingOptions[this.#menuSelectedIndex]; if (!thinkingLevel) return; - this.#handleSelect(selectedItem, this.#menuSelectedRole, thinkingLevel); + this.#handleSelect(selectedItem, this.#menuSelectedRole, thinkingLevel, "modelRole"); this.#closeMenu(); return; } @@ -1225,13 +1239,23 @@ export class ModelSelectorComponent extends Container { } } - #handleSelect(item: ModelItem, role: string | null, thinkingLevel?: ConfiguredThinkingLevel): void { + #handleSelect( + item: ModelItem, + role: string | null, + thinkingLevel?: ConfiguredThinkingLevel, + action: ModelSelectorAction = "modelRole", + ): void { if (this.#isItemDisabled(item)) { return; } // For temporary role, don't save to settings - just notify caller if (role === null) { - this.#onSelectCallback(item.model, null, undefined, item.selector); + this.#onSelectCallback(item.model, null, undefined, item.selector, action); + return; + } + + if (action === "retryFallback") { + this.#onSelectCallback(item.model, role, undefined, item.selector, action); return; } @@ -1241,7 +1265,7 @@ export class ModelSelectorComponent extends Container { this.#roles[role] = { model: item.model, thinkingLevel: selectedThinkingLevel, autoSelected: false }; // Notify caller (for updating agent state if needed) - this.#onSelectCallback(item.model, role, selectedThinkingLevel, item.selector); + this.#onSelectCallback(item.model, role, selectedThinkingLevel, item.selector, action); // Update list to show new badges this.#updateList(); diff --git a/packages/coding-agent/src/modes/components/settings-defs.ts b/packages/coding-agent/src/modes/components/settings-defs.ts index 582877281..dcad6310d 100644 --- a/packages/coding-agent/src/modes/components/settings-defs.ts +++ b/packages/coding-agent/src/modes/components/settings-defs.ts @@ -180,7 +180,7 @@ function pathToSettingDef(path: SettingPath): SettingDef | null { } if (schemaType === "record") { - return path === "providers.maxInFlightRequests" ? { ...base, type: "providerLimits" } : null; + return path === "providers.maxInFlightRequests" ? { ...base, type: "providerLimits" } : { ...base, type: "text" }; } return null; diff --git a/packages/coding-agent/src/modes/components/status-line/component.ts b/packages/coding-agent/src/modes/components/status-line/component.ts index 380e0d5e6..195539a46 100644 --- a/packages/coding-agent/src/modes/components/status-line/component.ts +++ b/packages/coding-agent/src/modes/components/status-line/component.ts @@ -1235,9 +1235,21 @@ export class StatusLineComponent implements Component { } } } + const leftOverflowDropIndex = (): number => { + // Preserve the current working directory as long as possible. The + // previous right-to-left pop could collapse a normal-width bar to + // just the model segment, hiding the path before less-critical left + // segments such as model/mode/collab were removed. + for (let i = leftSegIds.length - 1; i >= 0; i--) { + if (leftSegIds[i] !== "path") return i; + } + return left.length - 1; + }; + while (totalWidth() > topFillWidth && left.length > 0) { - left.pop(); - leftSegIds.pop(); + const dropIdx = leftOverflowDropIndex(); + left.splice(dropIdx, 1); + leftSegIds.splice(dropIdx, 1); leftWidth = groupWidth(left, leftCapWidth, leftSepWidth); } } diff --git a/packages/coding-agent/src/modes/controllers/command-controller.ts b/packages/coding-agent/src/modes/controllers/command-controller.ts index 8b7363b18..b9c63f42e 100644 --- a/packages/coding-agent/src/modes/controllers/command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/command-controller.ts @@ -898,11 +898,7 @@ export class CommandController { } async #runNewSessionFlow(options?: NewSessionOptions, label: string = "New session started"): Promise { - if (this.ctx.loadingAnimation) { - this.ctx.loadingAnimation.stop(); - this.ctx.loadingAnimation = undefined; - } - this.ctx.statusContainer.clear(); + this.ctx.clearTransientSessionUi(); if (this.ctx.session.isCompacting) { this.ctx.session.abortCompaction(); @@ -916,14 +912,9 @@ export class CommandController { this.ctx.statusLine.invalidate(); this.ctx.statusLine.resetActiveTime(); - this.ctx.ui.requestRender(); this.ctx.updateEditorBorderColor(); - this.ctx.chatContainer.clear(); - this.ctx.pendingMessagesContainer.clear(); - this.ctx.compactionQueuedMessages = []; - this.ctx.streamingComponent = undefined; - this.ctx.streamingMessage = undefined; - this.ctx.pendingTools.clear(); + this.ctx.clearTransientSessionUi(); + this.ctx.resetTranscript(); this.ctx.present([new Spacer(1), new Text(`${theme.fg("accent", `${theme.status.success} ${label}`)}`, 1, 1)]); await this.ctx.reloadTodos(); @@ -963,7 +954,7 @@ export class CommandController { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; } - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const success = await this.ctx.session.fork(); if (!success) { @@ -1226,7 +1217,7 @@ export class CommandController { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; } - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const label = isAuto ? "Auto-compacting context... (esc to cancel)" : "Compacting context... (esc to cancel)"; const compactingLoader = new Loader( @@ -1256,7 +1247,7 @@ export class CommandController { await this.ctx.session.compact(instructions, options); compactingLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); this.ctx.rebuildChatFromMessages(); this.ctx.statusLine.invalidate(); @@ -1272,7 +1263,7 @@ export class CommandController { } } finally { compactingLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } // Run the caller's pre-flush hook (e.g. the plan-approval model transition) // before queued user input is dispatched, so any turn queued during @@ -1301,7 +1292,7 @@ export class CommandController { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; } - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const handoffLoader = new Loader( this.ctx.ui, @@ -1322,11 +1313,10 @@ export class CommandController { return; } - // Rebuild chat from the new session (which now contains the handoff document) - this.ctx.rebuildChatFromMessages(); - + // Rebuild chat from the new session (which now contains the handoff document). + this.ctx.clearTransientSessionUi(); + this.ctx.renderInitialMessages(); this.ctx.statusLine.invalidate(); - this.ctx.ui.requestRender(); this.ctx.updateEditorBorderColor(); await this.ctx.reloadTodos(); @@ -1346,9 +1336,9 @@ export class CommandController { } } finally { handoffLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } - this.ctx.ui.requestRender(); + this.ctx.ui.requestRender(true, { clearScrollback: true }); } } diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index 88a66022b..e2baa993e 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -379,7 +379,7 @@ export class EventController { if (this.ctx.retryLoader) { this.ctx.retryLoader.stop(); this.ctx.retryLoader = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } this.#cancelIdleCompaction(); this.#cancelIdleRecap(); @@ -1083,7 +1083,7 @@ export class EventController { if (this.ctx.loadingAnimation) { this.ctx.loadingAnimation.stop(); this.ctx.loadingAnimation = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } if (this.ctx.streamingComponent) { this.ctx.chatContainer.removeChild(this.ctx.streamingComponent); @@ -1125,9 +1125,9 @@ export class EventController { /** * Tear down the live "Working…" loader: stop its animation timer AND clear the - * reference. A transient overlay (auto-compaction / auto-retry) that only ran - * `statusContainer.clear()` detached the loader from the container but left - * `ctx.loadingAnimation` set, so the resumed turn's `agent_start` → + * reference. A transient overlay (auto-compaction / auto-retry) can remove the + * loader from the container while leaving `ctx.loadingAnimation` set, so the + * resumed turn's `agent_start` → * `ensureLoadingAnimation()` (guarded by `if (!this.loadingAnimation)`) skipped * re-adding it and the spinner vanished while the agent kept streaming. Nulling * the reference here lets the next `agent_start` recreate and re-attach it. @@ -1168,7 +1168,7 @@ export class EventController { this.#cancelIdleRecap(); this.#setTerminalProgress(true); this.#stopWorkingLoader(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); const reasonText = event.reason === "overflow" ? "Context overflow detected, " @@ -1203,7 +1203,7 @@ export class EventController { if (this.ctx.autoCompactionLoader) { this.ctx.autoCompactionLoader.stop(); this.ctx.autoCompactionLoader = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } const isHandoffAction = event.action === "handoff"; const isShakeAction = event.action === "shake"; @@ -1245,12 +1245,12 @@ export class EventController { } else if (event.errorMessage) { this.ctx.showWarning(event.errorMessage); } else if (isHandoffAction) { - this.ctx.chatContainer.clear(); + this.ctx.clearTransientSessionUi(); this.ctx.lastAssistantUsage = undefined; - this.ctx.rebuildChatFromMessages(); + this.ctx.renderInitialMessages(); this.ctx.statusLine.invalidate(); - this.ctx.ui.requestRender(); await this.ctx.reloadTodos(); + this.ctx.ui.requestRender(true, { clearScrollback: true }); this.ctx.showStatus("Auto-handoff completed"); } else if (event.skipped) { // Benign skip: no model selected, no candidate models available, or nothing @@ -1268,7 +1268,7 @@ export class EventController { async #handleAutoRetryStart(event: Extract): Promise { this.#trackRetrySupersededAssistantComponent(this.#lastAssistantComponent); this.#stopWorkingLoader(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); if (AIError.is(event.errorId, AIError.Flag.ThinkingLoop)) { // The retry path drops the failed assistant from runtime context. Do not // restore its inline Error row; just unpin the fixed-region banner so the @@ -1292,7 +1292,7 @@ export class EventController { if (this.ctx.retryLoader) { this.ctx.retryLoader.stop(); this.ctx.retryLoader = undefined; - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } if (event.success) { let appliedRecovered = false; diff --git a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts index 455f8de17..baea76ffd 100644 --- a/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts +++ b/packages/coding-agent/src/modes/controllers/extension-ui-controller.ts @@ -162,18 +162,12 @@ export class ExtensionUiController { waitForIdle: () => this.ctx.session.agent.waitForIdle(), reload: async () => { await this.ctx.session.reload(); - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.showStatus("Reloaded session"); }, newSession: async options => { - // Stop any loading animation - if (this.ctx.loadingAnimation) { - this.ctx.loadingAnimation.stop(); - this.ctx.loadingAnimation = undefined; - } - this.ctx.statusContainer.clear(); + this.ctx.clearTransientSessionUi(); // Create new session this.clearExtensionTerminalInputListeners(); @@ -192,15 +186,8 @@ export class ExtensionUiController { // Reset and update status line this.ctx.statusLine.invalidate(); this.ctx.statusLine.resetActiveTime(); - this.ctx.ui.requestRender(); - - // Clear UI state - this.ctx.chatContainer.clear(); - this.ctx.pendingMessagesContainer.clear(); - this.ctx.compactionQueuedMessages = []; - this.ctx.streamingComponent = undefined; - this.ctx.streamingMessage = undefined; - this.ctx.pendingTools.clear(); + this.ctx.clearTransientSessionUi(); + this.ctx.resetTranscript(); this.ctx.present([ new Spacer(1), @@ -218,7 +205,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.editor.setText(result.selectedText); @@ -233,7 +219,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); if (result.editorText && !this.ctx.editor.getText().trim()) { @@ -251,7 +236,6 @@ export class ExtensionUiController { return { cancelled: true }; } setSessionTerminalTitle(this.ctx.sessionManager.getSessionName(), this.ctx.sessionManager.getCwd()); - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); return { cancelled: false }; @@ -398,18 +382,12 @@ export class ExtensionUiController { waitForIdle: () => this.ctx.session.agent.waitForIdle(), reload: async () => { await this.ctx.session.reload(); - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.showStatus("Reloaded session"); }, newSession: async options => { - // Stop any loading animation - if (this.ctx.loadingAnimation) { - this.ctx.loadingAnimation.stop(); - this.ctx.loadingAnimation = undefined; - } - this.ctx.statusContainer.clear(); + this.ctx.clearTransientSessionUi(); // Create new session this.clearExtensionTerminalInputListeners(); @@ -425,12 +403,8 @@ export class ExtensionUiController { } // Clear UI state - this.ctx.chatContainer.clear(); - this.ctx.pendingMessagesContainer.clear(); - this.ctx.compactionQueuedMessages = []; - this.ctx.streamingComponent = undefined; - this.ctx.streamingMessage = undefined; - this.ctx.pendingTools.clear(); + this.ctx.clearTransientSessionUi(); + this.ctx.resetTranscript(); this.ctx.present([ new Spacer(1), @@ -448,7 +422,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.editor.setText(result.selectedText); @@ -463,7 +436,6 @@ export class ExtensionUiController { } // Update UI - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); if (result.editorText && !this.ctx.editor.getText().trim()) { @@ -480,7 +452,6 @@ export class ExtensionUiController { if (!result) { return { cancelled: true }; } - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); return { cancelled: false }; diff --git a/packages/coding-agent/src/modes/controllers/input-controller.ts b/packages/coding-agent/src/modes/controllers/input-controller.ts index ed16eb36c..aa70e59ca 100644 --- a/packages/coding-agent/src/modes/controllers/input-controller.ts +++ b/packages/coding-agent/src/modes/controllers/input-controller.ts @@ -86,6 +86,20 @@ function hasPasteText(value: unknown): value is PasteTarget { return typeof value === "object" && value !== null && typeof (value as PasteTarget).pasteText === "function"; } +const SHELL_PROMPT_COMMAND_RE = + /^(?:\.{0,2}\/|~\/|cd(?:\s|$)|sudo(?:\s|$)|git(?:\s|$)|bun(?:\s|$)|npm(?:\s|$)|pnpm(?:\s|$)|yarn(?:\s|$)|node(?:\s|$)|python\d*(?:\s|$)|cargo(?:\s|$)|go(?:\s|$)|make(?:\s|$)|docker(?:\s|$)|kubectl(?:\s|$))/; +const SHELL_PROMPT_OPERATOR_RE = /(?:^|\s)(?:&&|\|\||\||2>&1|[<>]{1,2})(?:\s|$)/; +const OMP_STATUS_LINE_RE = /^\s*in:\s+\d+\s+out:\s+\d+(?:\s+cache\s+\S+)?\s+t:\s+\S+\s+tok\/s:\s+\S+/m; + +function looksLikePastedShellPrompt(code: string): boolean { + const firstLine = code.split("\n", 1)[0]?.trimStart() ?? ""; + return ( + SHELL_PROMPT_COMMAND_RE.test(firstLine) || + SHELL_PROMPT_OPERATOR_RE.test(firstLine) || + OMP_STATUS_LINE_RE.test(code) + ); +} + function pythonCommandPrefixLength(trimmedText: string): 0 | 1 | 2 { if (trimmedText.charCodeAt(0) !== 36 /* $ */) return 0; if (trimmedText.charCodeAt(1) === 123 /* { */) return 0; @@ -100,8 +114,10 @@ function parsePythonCommandInput(text: string): { code: string; isExcluded: bool const trimmed = text.trimStart(); const prefixLength = pythonCommandPrefixLength(trimmed); if (prefixLength === 0) return undefined; + const code = trimmed.slice(prefixLength).trim(); + if (prefixLength === 1 && looksLikePastedShellPrompt(code)) return undefined; return { - code: trimmed.slice(prefixLength).trim(), + code, isExcluded: prefixLength === 2, }; } @@ -536,7 +552,7 @@ export class InputController { const wasPythonMode = this.ctx.isPythonMode; const trimmed = text.trimStart(); this.ctx.isBashMode = trimmed.startsWith("!"); - this.ctx.isPythonMode = pythonCommandPrefixLength(trimmed) > 0; + this.ctx.isPythonMode = parsePythonCommandInput(trimmed) !== undefined; if (wasBashMode !== this.ctx.isBashMode || wasPythonMode !== this.ctx.isPythonMode) { this.ctx.updateEditorBorderColor(); } diff --git a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts index 90040b394..5d0491a10 100644 --- a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts @@ -88,12 +88,14 @@ function raceAbortSignal(promise: Promise, signal: AbortSignal, createErro const MCP_AUTH_MIN_WRAP_WIDTH = 16; /** - * Wrap `url` into rows that each fit inside `width`, prefixed by a shared - * single-column indent so nested composition doesn't touch column 0. When the - * label + URL fit on one line, returns a single row; otherwise puts the label - * on its own row and slices the URL into fixed-width chunks. URL chunks are - * plain code points — browsers strip whitespace when pasted into the address - * bar, so a multi-row selection copies back to the intact URL. + * Wrap `url` into rows that each fit inside `width`. When the label + URL fit + * on one line, returns a single indented row; otherwise puts the label on its + * own indented row and slices the URL into fixed-width chunks that start at + * column 0. Continuation chunks carry ZERO leading bytes on purpose: a + * multi-row terminal selection includes the newline plus any leading indent, + * and while address bars strip newlines they preserve or percent-encode + * embedded spaces — an indent would corrupt the URL at every chunk boundary + * (silently, when the damage lands inside a query value). */ function wrapUrlRows(label: string, url: string, width: number): string[] { const indent = " "; @@ -103,10 +105,9 @@ function wrapUrlRows(label: string, url: string, width: number): string[] { if (inlineWidth <= effective) { return [`${indent}${theme.fg("muted", `${label} ${sanitized}`)}`]; } - const chunkWidth = Math.max(1, effective - indent.length); const rows: string[] = [`${indent}${theme.fg("muted", label)}`]; - for (let i = 0; i < sanitized.length; i += chunkWidth) { - rows.push(`${indent}${theme.fg("muted", sanitized.slice(i, i + chunkWidth))}`); + for (let i = 0; i < sanitized.length; i += effective) { + rows.push(theme.fg("muted", sanitized.slice(i, i + effective))); } return rows; } diff --git a/packages/coding-agent/src/modes/controllers/selector-controller.ts b/packages/coding-agent/src/modes/controllers/selector-controller.ts index 74789e251..510ef2812 100644 --- a/packages/coding-agent/src/modes/controllers/selector-controller.ts +++ b/packages/coding-agent/src/modes/controllers/selector-controller.ts @@ -595,13 +595,27 @@ export class SelectorController { this.ctx.settings, this.ctx.session.modelRegistry, this.ctx.session.scopedModels, - async (model, role, thinkingLevel, selector) => { + async (model, role, thinkingLevel, selector, action) => { // `auto` is session-global: never baked into a per-role model value // (it can't round-trip through `model:`). Apply it to the session // separately and persist via `defaultThinkingLevel`. const isAuto = thinkingLevel === AUTO_THINKING; const concreteThinking = isAuto ? undefined : thinkingLevel; + const selectorValue = selector ?? `${model.provider}/${model.id}`; try { + if (action === "retryFallback" && role !== null) { + const fallbackSelector = formatModelSelectorValue(selectorValue, concreteThinking); + const fallbackChains = this.ctx.settings.get("retry.fallbackChains"); + const chain = Array.isArray(fallbackChains[role]) ? fallbackChains[role] : []; + this.ctx.settings.set("retry.fallbackChains", { + ...fallbackChains, + [role]: [fallbackSelector, ...chain.filter(existing => existing !== fallbackSelector)], + }); + const roleInfo = getRoleInfo(role, settings); + const roleLabel = roleInfo?.name ?? role; + this.ctx.showStatus(`${roleLabel} fallback model: ${fallbackSelector}`); + return; + } if (role === null) { // Temporary: update agent state but don't persist the model to settings await this.ctx.session.setModelTemporary(model); @@ -773,7 +787,6 @@ export class SelectorController { return; } - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); this.ctx.editor.setText(result.selectedText); done(); @@ -917,7 +930,6 @@ export class SelectorController { // Update UI — rebuild the display transcript for the new leaf (the // context from navigateTree is the LLM context, not the transcript). - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); if (result.editorText && !this.ctx.editor.getText().trim()) { @@ -929,7 +941,7 @@ export class SelectorController { } finally { if (summaryLoader) { summaryLoader.stop(); - this.ctx.statusContainer.clear(); + this.ctx.statusContainer.disposeChildren(); } this.ctx.editor.onEscape = originalOnEscape; } @@ -1069,7 +1081,6 @@ export class SelectorController { this.ctx.updateEditorBorderColor(); // Clear and re-render the chat - this.ctx.chatContainer.clear(); this.ctx.renderInitialMessages({ clearTerminalHistory: true }); await this.ctx.reloadTodos(); this.ctx.showStatus(movedProject ? `Resumed session in ${shortenPath(newCwd)}` : "Resumed session"); diff --git a/packages/coding-agent/src/modes/github-ref-autocomplete.ts b/packages/coding-agent/src/modes/github-ref-autocomplete.ts new file mode 100644 index 000000000..0b261f480 --- /dev/null +++ b/packages/coding-agent/src/modes/github-ref-autocomplete.ts @@ -0,0 +1,75 @@ +/** + * Autocomplete for GitHub issue/PR references typed as `#` (e.g. `#3164`). + * + * Mirrors the `@` file-reference and `scheme://` internal-url conventions: the + * token is rewritten to an internal URL (`pr://3164` or `issue://3164`) plus a + * trailing space, and the existing tool-mediated pipeline (the `read` tool → + * InternalUrlRouter → `gh`) resolves it from the session cwd's git remote. + * + * No network at suggestion time — candidates are generated locally. GitHub + * shares the issue/PR number space and there is no cheap way to tell which a + * given number is while typing, so both a PR and an Issue candidate are offered + * by default. Naming the type first (`pr #3164` / `issue #3164`) constrains the + * candidates to that kind. Anything that is not a standalone `#` token + * keeps falling through to the existing prompt-action menu. + */ +import type { AutocompleteItem } from "@oh-my-pi/pi-tui"; + +/** Candidate kinds, in default display order. */ +const GITHUB_REF_KINDS = [ + { qualifier: "pr", scheme: "pr", label: "PR", description: "GitHub pull request" }, + { qualifier: "issue", scheme: "issue", label: "Issue", description: "GitHub issue" }, +] as const; + +export interface GithubRefContext { + /** Text to replace on accept: `#3164`, or `pr #3164` when a qualifier precedes it. */ + prefix: string; + /** Type the user named (`pr`/`pull` → `pr`, `issue` → `issue`), or null to offer both. */ + qualifier: "pr" | "issue" | null; + /** The numeric reference, e.g. `3164`. */ + number: string; +} + +/** + * A standalone `#` token ending at the cursor. The `#` must be + * preceded by a token boundary (start, whitespace, or an opening quote/paren/`<`/`=`, + * matching the internal-URL boundary set) so embedded hashes like `owner/repo#N`, + * `foo#N`, `C#12`, or a URL fragment do not match. An optional `pr`/`pull`/`issue` + * qualifier word (case-insensitive) immediately before the `#` constrains the kind. + */ +const GITHUB_REF_TOKEN_RE = /(?:^|[\s"'`(<=])(?:(pr|pull|issue)(\s+))?#([1-9]\d*)$/i; + +export function getGithubRefContext(textBeforeCursor: string): GithubRefContext | null { + const match = textBeforeCursor.match(GITHUB_REF_TOKEN_RE); + if (!match) return null; + const qualifierWord = match[1]; + const whitespace = match[2] ?? ""; + const number = match[3] ?? ""; + return { + prefix: qualifierWord ? `${qualifierWord}${whitespace}#${number}` : `#${number}`, + qualifier: !qualifierWord ? null : qualifierWord.toLowerCase() === "issue" ? "issue" : "pr", + number, + }; +} + +/** + * Suggestions for a `#` token. Both kinds are offered unless the user + * named a type (`pr #3164` / `issue #3164`), in which case only that kind is + * offered. Returns `null` when the text before the cursor is not a standalone + * `#` token. + */ +export function getGithubRefSuggestions( + textBeforeCursor: string, +): { items: AutocompleteItem[]; prefix: string } | null { + const context = getGithubRefContext(textBeforeCursor); + if (!context) return null; + const kinds = context.qualifier + ? GITHUB_REF_KINDS.filter(kind => kind.qualifier === context.qualifier) + : GITHUB_REF_KINDS; + const items: AutocompleteItem[] = kinds.map(kind => ({ + value: `${kind.scheme}://${context.number}`, + label: `${kind.label} #${context.number}`, + description: kind.description, + })); + return { items, prefix: context.prefix }; +} diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index 8e064d8ee..78bec222b 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -579,10 +579,10 @@ export class InteractiveMode implements InteractiveModeContext { this.retryLoader.stop(); this.retryLoader = undefined; } - this.statusContainer.clear(); - this.pendingMessagesContainer.clear(); + this.statusContainer.disposeChildren(); + this.pendingMessagesContainer.disposeChildren(); this.#cancelModelCycleClearTimer(); - this.modelCycleContainer.clear(); + this.modelCycleContainer.disposeChildren(); this.compactionQueuedMessages = []; this.streamingComponent = undefined; this.streamingMessage = undefined; @@ -2602,6 +2602,49 @@ export class InteractiveMode implements InteractiveModeContext { } } + #resolveLocalRoot(): string { + return resolveLocalUrlToPath("local://", { + getArtifactsDir: () => this.sessionManager.getArtifactsDir(), + getSessionId: () => this.sessionManager.getSessionId(), + }); + } + + async #copyLocalArtifactsForFreshSession(sourceRoot: string, destinationRoot: string): Promise { + if (sourceRoot === destinationRoot) return; + + let sourceRootStat: { isDirectory(): boolean }; + try { + sourceRootStat = await fs.lstat(sourceRoot); + } catch (error) { + if (isEnoent(error)) return; + throw error; + } + + if (!sourceRootStat.isDirectory()) return; + + await fs.mkdir(destinationRoot, { recursive: true }); + await this.#copyLocalArtifactEntries(sourceRoot, destinationRoot); + } + + async #copyLocalArtifactEntries(sourceDir: string, destinationDir: string): Promise { + const entries = await fs.readdir(sourceDir, { withFileTypes: true }); + for (const entry of entries) { + const sourcePath = path.join(sourceDir, entry.name); + const destinationPath = path.join(destinationDir, entry.name); + + if (entry.isDirectory()) { + await fs.mkdir(destinationPath, { recursive: true }); + await this.#copyLocalArtifactEntries(sourcePath, destinationPath); + continue; + } + + if (entry.isFile()) { + await fs.mkdir(path.dirname(destinationPath), { recursive: true }); + await fs.copyFile(sourcePath, destinationPath); + } + } + } + async #approvePlan( planContent: string, options: { @@ -2632,14 +2675,16 @@ export class InteractiveMode implements InteractiveModeContext { }); if (!options.preserveContext) { + const oldLocalRoot = this.#resolveLocalRoot(); await this.handleClearCommand(); - // The new session has a fresh local:// root — persist the approved plan there - // so `local://-plan.md` resolves correctly in the execution session. + const newLocalRoot = this.#resolveLocalRoot(); + await this.#copyLocalArtifactsForFreshSession(oldLocalRoot, newLocalRoot); const newLocalPath = resolveLocalUrlToPath(options.planFilePath, { getArtifactsDir: () => this.sessionManager.getArtifactsDir(), getSessionId: () => this.sessionManager.getSessionId(), }); - await Bun.write(newLocalPath, planContent); + await fs.mkdir(path.dirname(newLocalPath), { recursive: true }); + await fs.writeFile(newLocalPath, planContent); } else if (options.compactBeforeExecute) { // Distill the plan-mode transcript before the execution turn is queued so // the plan-approved synthetic prompt lands as a fresh cache anchor. @@ -3626,7 +3671,7 @@ export class InteractiveMode implements InteractiveModeContext { ensureLoadingAnimation(): void { if (!this.loadingAnimation) { this.#clearWorkingMessageAccentCache(); - this.statusContainer.clear(); + this.statusContainer.disposeChildren(); const messageColorFn = ((message: string) => renderWorkingMessage(message, this.#getWorkingMessageAccent())) as LoaderMessageColorFn & { animated?: true; @@ -3647,7 +3692,7 @@ export class InteractiveMode implements InteractiveModeContext { ); this.statusContainer.addChild(this.loadingAnimation); } else if (!this.statusContainer.children.includes(this.loadingAnimation)) { - this.statusContainer.clear(); + this.statusContainer.disposeChildren(); this.statusContainer.addChild(this.loadingAnimation); this.ui.requestRender(); } @@ -3660,7 +3705,7 @@ export class InteractiveMode implements InteractiveModeContext { this.loadingAnimation = undefined; this.#clearWorkingMessageAccentCache(); if (clearStatusContainer) { - this.statusContainer.clear(); + this.statusContainer.disposeChildren(); } } @@ -4123,7 +4168,6 @@ export class InteractiveMode implements InteractiveModeContext { } this.#btwController.dispose(); this.#omfgController.dispose(); - this.chatContainer.clear(); this.renderInitialMessages({ clearTerminalHistory: true }); this.updateEditorBorderColor(); this.showStatus( diff --git a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts index 9adcfc057..f5135bddc 100644 --- a/packages/coding-agent/src/modes/prompt-action-autocomplete.ts +++ b/packages/coding-agent/src/modes/prompt-action-autocomplete.ts @@ -9,6 +9,7 @@ import { import { formatKeyHints, type KeybindingsManager } from "../config/keybindings"; import { isSettingsInitialized, settings } from "../config/settings"; import { applyEmojiCompletion, getEmojiSuggestions, isEmojiPrefix, tryEmojiInlineReplace } from "./emoji-autocomplete"; +import { getGithubRefContext, getGithubRefSuggestions } from "./github-ref-autocomplete"; import { applyInternalUrlCompletion, getInternalUrlSuggestions, @@ -94,6 +95,36 @@ function getPromptActionPrefix(textBeforeCursor: string): string | null { return textBeforeCursor.slice(hashIndex); } +function applyGithubRefCompletion( + lines: string[], + cursorLine: number, + cursorCol: number, + item: AutocompleteItem, + prefix: string, +): { lines: string[]; cursorLine: number; cursorCol: number } | null { + if (!getGithubRefContext(prefix)) return null; + const scheme: "pr" | "issue" | null = item.value.startsWith("pr://") + ? "pr" + : item.value.startsWith("issue://") + ? "issue" + : null; + if (!scheme) return { lines, cursorLine, cursorCol }; + + const currentLine = lines[cursorLine] || ""; + const liveContext = getGithubRefContext(currentLine.slice(0, cursorCol)); + if (!liveContext || (liveContext.qualifier && liveContext.qualifier !== scheme)) { + return { lines, cursorLine, cursorCol }; + } + + return applyInternalUrlCompletion( + lines, + cursorLine, + cursorCol, + { ...item, value: `${scheme}://${liveContext.number}` }, + liveContext.prefix, + ); +} + export class PromptActionAutocompleteProvider implements AutocompleteProvider { #commands: SlashCommand[]; #baseProvider: CombinedAutocompleteProvider; @@ -129,6 +160,8 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider { } } + const githubRefSuggestions = getGithubRefSuggestions(textBeforeCursor); + if (githubRefSuggestions) return githubRefSuggestions; const promptActionPrefix = getPromptActionPrefix(textBeforeCursor); if (promptActionPrefix) { const query = promptActionPrefix.slice(1).toLowerCase(); @@ -176,6 +209,8 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider { cursorCol: number; onApplied?: () => void; } { + const githubRefCompletion = applyGithubRefCompletion(lines, cursorLine, cursorCol, item, prefix); + if (githubRefCompletion) return githubRefCompletion; if (prefix.startsWith("#") && isPromptActionItem(item)) { if (item.actionId === "undo") { return { diff --git a/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts b/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts index f523732a1..419e91792 100644 --- a/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts +++ b/packages/coding-agent/src/modes/utils/hotkeys-markdown.ts @@ -52,7 +52,8 @@ export function buildHotkeysMarkdown(bindings: HotkeysMarkdownBindings): string `| \`${appKey(bindings, "app.clipboard.pasteImage")}\` | Paste image or text from clipboard |`, "| Hold `Space` | Speech-to-text (push-to-talk): hold to record, release to transcribe |", `| \`${appKey(bindings, "app.agents.hub")}\` / \`${appKey(bindings, "app.session.observe")}\` / double-tap \`←\` (empty editor) | Open the agent hub |`, - "| `#` | Open prompt actions |", + "| `#` | GitHub issue/PR reference (e.g. `#3164` → `pr://`/`issue://`) |", + "| `#` / `#` | Prompt actions (copy / undo / move cursor) |", "| `/` | Slash commands |", "| `!` | Run bash command |", "| `!!` | Run bash command (excluded from context) |", diff --git a/packages/coding-agent/src/modes/utils/ui-helpers.ts b/packages/coding-agent/src/modes/utils/ui-helpers.ts index d5f250824..bfc89ae00 100644 --- a/packages/coding-agent/src/modes/utils/ui-helpers.ts +++ b/packages/coding-agent/src/modes/utils/ui-helpers.ts @@ -581,7 +581,7 @@ export class UiHelpers { } else { this.ctx.resetTranscript(); } - this.ctx.pendingMessagesContainer.clear(); + this.ctx.pendingMessagesContainer.disposeChildren(); this.ctx.pendingBashComponents = []; this.ctx.pendingPythonComponents = []; @@ -647,7 +647,7 @@ export class UiHelpers { } updatePendingMessagesDisplay(): void { - this.ctx.pendingMessagesContainer.clear(); + this.ctx.pendingMessagesContainer.disposeChildren(); const queuedMessages = this.ctx.viewSession.getQueuedMessages() as QueuedMessages; const steeringMessages: Array<{ message: string; label: string }> = []; diff --git a/packages/coding-agent/src/modes/workflow.ts b/packages/coding-agent/src/modes/workflow.ts index ab7ae17fa..e8d4832cc 100644 --- a/packages/coding-agent/src/modes/workflow.ts +++ b/packages/coding-agent/src/modes/workflow.ts @@ -1,4 +1,5 @@ -import workflowNotice from "../prompts/system/workflow-notice.md" with { type: "text" }; +import { prompt } from "@oh-my-pi/pi-utils"; +import workflowNoticeTemplate from "../prompts/system/workflow-notice.md" with { type: "text" }; import { createGradientHighlighter, type KeywordHighlighter } from "./gradient-highlight"; import { keywordInProse } from "./markdown-prose"; @@ -7,18 +8,23 @@ import { keywordInProse } from "./markdown-prose"; * * Typing the standalone word in the input editor paints it with a warm * amber→green gradient ({@link highlightWorkflow}); submitting a message that - * mentions it appends a hidden {@link WORKFLOW_NOTICE} that steers the model to - * author a deterministic multi-subagent workflow in eval cells (agent/parallel/ - * pipeline). Matching is whitespace-delimited and case-sensitive (lowercase - * only) — "workflowz" triggers, but "workflowzed", "Workflowz", and - * "workflowz.ts" never do. + * mentions it appends a hidden workflow notice that steers the model to author + * a deterministic multi-subagent workflow through the active task schema. + * Matching is whitespace-delimited and case-sensitive (lowercase only) — + * "workflowz" triggers, but "workflowzed", "Workflowz", and "workflowz.ts" + * never do. */ // Detection: lowercase keyword flanked by whitespace or a string edge. Non-global so `.test` stays stateless. const WORKFLOW_WORD = /(? -Plan mode is active. You MUST perform READ-ONLY work only: -- You NEVER create, edit, or delete files — except the single plan file named below. +Plan mode is active. You MUST preserve read-only working-tree and system semantics: +- You NEVER create, edit, delete, or rename working-tree files. - You NEVER run state-changing commands (`git commit`, `npm install`, migrations) or make any other system change. +- `local://` artifacts are session-local planning artifacts. You MAY create or update them when explicitly requested or needed for the plan. +- You NEVER delete or rename `local://` artifacts. +- You MUST write the canonical plan to `local://-plan.md`. To leave plan mode and implement: call `resolve` with `action: "apply"`, a `reason`, and `extra: { title: "" }`, where `` matches your `local://-plan.md`. The user then picks an execution option and full write access is restored. `` may contain only letters, numbers, underscores, and hyphens. diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index e852b653a..e90d57c5e 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -110,9 +110,8 @@ You MUST use the specialized tool over its shell equivalent: {{#has tools "lsp"}}- Code intelligence → `{{toolRefs.lsp}}`.{{/has}} {{#has tools "grep"}}- Regex search → `{{toolRefs.grep}}`, not `grep`, `rg`, or `awk`.{{/has}} {{#has tools "glob"}}- Globbing → `{{toolRefs.glob}}`, not `ls **/*.ext` or `fd`.{{/has}} -{{#has tools "eval"}}- Default for any compute: `{{toolRefs.eval}}` cells. Bash is the EXCEPTION — only single binary calls or short fact-computing pipelines (`wc -l`, `sort | uniq -c`, `diff`, checksums). The moment a command grows a loop, conditional, heredoc, `-e`/`-c` script, `$(…)` nesting, or >2 pipe stages, it's a program → `{{toolRefs.eval}}`. NEVER write multiline or inline-script bash.{{/has}} {{#has tools "bash"}}- `{{toolRefs.bash}}`: real binaries and short fact pipelines only. Commands shadowing the specialized tools above are blocked.{{/has}} -{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash.{{#has tools "eval"}} Needs control flow, state, or fights shell quoting → `{{toolRefs.eval}}`.{{/has}} Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}} +{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash. Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}} {{#has tools "report_tool_issue"}} diff --git a/packages/coding-agent/src/prompts/system/workflow-notice.md b/packages/coding-agent/src/prompts/system/workflow-notice.md index 4071c9df5..74eddee21 100644 --- a/packages/coding-agent/src/prompts/system/workflow-notice.md +++ b/packages/coding-agent/src/prompts/system/workflow-notice.md @@ -1,70 +1,89 @@ -The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Author the orchestration as Python in the `eval` tool and fan out subagents — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough. +The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Use the `task` tool {{#if taskBatch}}for batched fan-out{{else}}once per independent subagent{{/if}} — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough. -Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline FIRST (list the files, scope the diff, find the call sites) to discover the work-list, then fan out over it — you don't need to know the shape before the *task*, only before the *fan-out*. Common shapes, each a well-scoped `eval` call you can chain across turns: -- **Understand** — parallel readers over subsystems → structured map -- **Design** — judge panel of N independent approaches → scored synthesis -- **Review** — split into dimensions → find per dimension → adversarially verify each finding -- **Research** — multi-modal sweep → deep-read the hits → synthesize -- **Migrate** — discover sites → transform each → verify +Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline first (list the files, scope the diff, find the call sites) to discover the work list, then fan out over it. Common shapes: +- **Understand** — parallel readers over subsystems → structured map. +- **Design** — independent approaches → scored synthesis. +- **Review** — split dimensions → find per dimension → adversarially verify each finding. +- **Research** — multi-modal sweep → deep-read the hits → synthesize. +- **Migrate** — discover sites → transform each → verify. - -State persists across eval calls, so scout in one call and fan out in the next. Every eval call has: + +{{#if taskBatch}} +Call `task` once per independent fan-out batch. Put shared background in `context`, and put each independent work item in `tasks[]`. Do not emulate batching with shell loops or eval helper APIs. -- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes. Recursion follows `task.maxRecursionDepth` (default 2; `-1` uses eval's hard cap 3): main agent depth = 0, each `agent()` child increments depth by 1, and a spawner may call `agent()` only while its current `taskDepth < effective cap`. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts. -- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one. -- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`. -- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out. -- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it. -- `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget. +`context` must carry the shared contract: -Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across calls and turns for multi-phase work, reading each result before you decide the next phase. - + # Goal + What the batch accomplishes. + # Constraints + Rules, non-goals, permissions, and verification limits. + # Contract + Shared interfaces, output shape, branch/base assumptions, and coordination rules. + +Each task assignment must be self-contained: + + # Target + Exact files, symbols, subsystem, or evidence surface; explicit non-goals. + # Change + What to inspect or modify, step by step, including APIs and patterns to reuse. + # Acceptance + Observable result, return packet, and local verification. Subagents skip formatters, + linters, and project-wide tests; the parent runs shared proof once. +{{else}} +Call `task` once per independent subagent. Put the full shared background and the leaf work in that call's `assignment`. Do not pass `context` or `tasks[]`: the flat task schema rejects them when batch calls are disabled. + +Each assignment must be self-contained: + + # Target + Exact files, symbols, subsystem, or evidence surface; explicit non-goals. + # Change + Shared background plus what to inspect or modify, step by step, including APIs and patterns to reuse. + # Acceptance + Observable result, return packet, and local verification. Subagents skip formatters, + linters, and project-wide tests; the parent runs shared proof once. +{{/if}} -For independent per-item chains (review → verify, fetch → extract → score), wrap the WHOLE chain in one function and run it with `parallel()` — then each item flows through its own steps without waiting on the others: +Decompose first, then {{#if taskBatch}}batch the independent leaves{{else}}issue one independent task call per leaf in the same turn{{/if}}: - DIMENSIONS = [{"key": "bugs", "prompt": "…"}, {"key": "perf", "prompt": "…"}] - def review_and_verify(d): - found = agent(d["prompt"], label=f"review:{d['key']}", schema=FINDINGS_SCHEMA) - return parallel([lambda f=f: {**f, "verdict": agent( - f"Refute if you can (default refuted when unsure): {f['title']}", - label=f"verify:{f['file']}", schema=VERDICT_SCHEMA)} for f in found["findings"]]) - phase("Review") - results = parallel([lambda d=d: review_and_verify(d) for d in DIMENSIONS]) - confirmed = [f for group in results for f in group if f["verdict"]["is_real"]] +{{#if taskBatch}} + task( + context: "# Goal\nReview the auth diff...\n# Constraints\nRead-only...\n# Contract\nReturn findings as severity/file/line/fix...", + tasks: [ + { id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection...\n# Acceptance\nReturn confirmed findings only..." }, + { id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance...\n# Acceptance\nReturn mismatches and exact prompt lines..." }, + ] + ) +{{else}} + task( + role: "Auth Storage Reviewer", + assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only..." + ) + task( + role: "Prompt Contract Reviewer", + assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only..." + ) +{{/if}} -Reach for `pipeline()` only when a stage genuinely needs ALL of the previous stage first — dedup/merge across the whole set, early-exit on zero, or "compare against the other findings" — because its inter-stage barrier makes every item wait for the slowest peer: - - phase("Find") - found = parallel([lambda d=d: agent(d["prompt"], schema=FINDINGS_SCHEMA) for d in DIMENSIONS]) - findings = dedupe([f for r in found for f in r["findings"]]) # needs everything at once - phase("Verify") - verdicts = parallel([lambda f=f: agent(verify_prompt(f), schema=VERDICT_SCHEMA) for f in findings]) - -Don't add a barrier just to flatten/map/filter — do that with plain Python between calls. Nested `parallel()` pools each cap independently, so keep total fan-out sane. +{{#if taskBatch}}Prefer one wide batch over serial subagent calls when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing.{{else}}Prefer issuing all independent task calls in one assistant turn over serial dispatch when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing.{{/if}} -Compose the harness the task calls for: -- **Adversarial verify** — N independent skeptics per finding, each prompted to REFUTE; keep it only if a majority survive. `votes = parallel([lambda i=i: agent(f"Refute: {claim}. refuted=true if unsure.", schema=VERDICT) for i in range(3)])`, then keep when `sum(not v["refuted"] for v in votes) ≥ 2`. -- **Perspective-diverse verify** — give each verifier a distinct lens (correctness, security, perf, does-it-reproduce) instead of N identical refuters. -- **Judge panel** — N attempts from different angles, scored by parallel judges; synthesize from the winner, graft the best of the rest. -- **Loop-until-dry** — for unknown-size discovery, keep spawning finders until K consecutive rounds surface nothing new; dedup against everything SEEN, not just what was confirmed, or it never converges. -- **Multi-modal sweep** — parallel finders each searching a different way (by-container, by-content, by-entity, by-time), each blind to the others. -- **Completeness critic** — a final agent that asks "what's missing — modality not run, claim unverified, file unread?"; its answer is the next round. -- **Budget/count loops** — `while len(bugs) < 10:` to hit a target, or `while budget.total and budget.remaining() > 50_000:` to scale depth to the turn budget; `log()` each round. -- **No silent caps** — if you bound coverage (top-N, no-retry, sampling), `log()` what you dropped; silent truncation reads as "covered everything" when it didn't. - -Scale to the ask: "find any bugs" → a few finders, single-vote verify. "thoroughly audit / be comprehensive" → larger finder pool, 3–5-vote adversarial pass, a synthesis stage. +- **Adversarial verify** — dispatch skeptical reviewers with distinct targets, then keep only findings the parent can verify against source. +- **Perspective-diverse review** — use separate correctness, security, performance, and maintainability roles instead of identical reviewers. +- **Completeness critic** — after the first batch, dispatch one read-only critic that asks what modality, file, claim, or proof was missed. +- **No silent caps** — if you bound coverage (top-N, no retry, sampling), state what was dropped and why before acting. +- **Parent owns closure** — subagents return evidence; the parent reads it, resolves contradictions, runs proof, and makes the final decision. -- Decompose the surface first; capture it in `todo` when it spans phases. -- Prefer `schema=` for any agent whose output you branch on. -- After a fan-out returns, YOU own correctness: read the artifacts, run the gate, verify before acting. Subagents do the legwork; they don't get the last word. -- Keep going until the task is closed — a returned fan-out is a step, not a stopping point. +- Capture multi-phase workflow state in the visible todo system when available. +{{#if taskBatch}}- Batch independent subagents in one `task` call.{{else}}- Dispatch independent subagents as separate `task` calls in the same turn.{{/if}} +- Give every subagent a narrow target, explicit non-goals, and a concrete return packet. +- After fan-out returns, read the artifacts, patch or decide, and run the shared gate. +- Keep going until the task is closed — returned fan-out is a step, not a stopping point. diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index 04bcca23c..83ad3ba0a 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -6,14 +6,22 @@ The shell invokes **real binaries** with simple args. It is NOT full GNU Bash. Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a fact and does not depend on shell-specific regex/quoting (`wc -l`, `sort | uniq -c`, `comm`, `diff`, a checksum, `git status`). -Anything below → `eval` cell, not bash: +{{#if hasEval}}Anything below → `eval` cell, not bash: - Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language - Heredocs (`< - `cwd` sets the working dir, not `cd dir && …` @@ -23,14 +31,17 @@ Anything below → `eval` cell, not bash: - `;` only when later commands should run despite earlier failures - Multiple bash calls per message run concurrently. NEVER split order-dependent commands across parallel calls — chain with `&&` in one call. - Internal URIs (`skill://`, `agent://`, …) auto-resolve to FS paths -- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly. +{{#if hasEval}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly.{{else}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Use a checked-in script, purpose-built tool, or single command that owns the output shape.{{/if}} {{#if asyncEnabled}} - `async: true` for long-running commands when you don't need immediate output: returns a background job ID; result delivered as a follow-up. {{/if}} -- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps. +{{#if hasEval}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.{{else}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a shell program; use a purpose-built tool or checked-in script instead.{{/if}} +{{#if hasGrep}}- NEVER shell out to search content or files: `grep/rg` → `grep`.{{else}}- Avoid shelling out for broad content search; use an active search/read tool when one is available.{{/if}} +{{#if hasRead}}{{#if hasGlob}}- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing.{{else}}- Prefer `read` for known file and directory reads. Only use shell listing when no file-listing tool is active.{{/if}}{{else}}{{#if hasGlob}}- Prefer `glob` for file discovery; avoid `find` when `glob` is active.{{else}}- If no file read/listing tool is active, keep shell inspection narrow and state that limitation.{{/if}}{{/if}} +- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://`. @@ -41,9 +52,9 @@ Anything below → `eval` cell, not bash: {{#if asyncEnabled}} # Timeout and async -- `timeout` is seconds, clamped to `1..3600`; the process is killed on elapse. -- `async: true` defers only reporting — it does NOT extend the timeout; a daemon with `async: true` is still killed at the clamped timeout. -- Need >3600s? Detach/manage lifecycle yourself (`cmd &`, supervisor, self-restarting script). The shell session persists across calls. +- `timeout` is seconds; nonzero values are clamped to `1..3600` and the process is killed on elapse. Set `timeout: 0` only for commands that must run until completion or explicit cancellation. +- `async: true` defers only reporting — it does NOT extend a nonzero timeout; use `timeout: 0` when a daemon or watcher must be cancellation-owned. +- Need a daemon or >3600s run? Use `async: true` with `timeout: 0` when the harness should keep it alive until cancellation, or detach/manage lifecycle yourself (`cmd &`, supervisor, self-restarting script). The shell session persists across calls. {{/if}} {{#if autoBackgroundEnabled}} diff --git a/packages/coding-agent/src/prompts/tools/grep.md b/packages/coding-agent/src/prompts/tools/grep.md index 78467d100..eef17d10e 100644 --- a/packages/coding-agent/src/prompts/tools/grep.md +++ b/packages/coding-agent/src/prompts/tools/grep.md @@ -2,7 +2,7 @@ Greps files using regex. - Rust regex (RE2-style): alternation is `foo|bar`, not GNU BRE-style `foo\|bar`; Rust word boundaries like `\bword\b` are supported. Use line anchors or post-filters instead of lookaround/backreferences. -- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). +- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). Literal colon filename + line range? Use `selector` (e.g. `{"path":"test:1-2","selector":"1-2"}`), not recursive `path:"test:1-2:1-2"`. - Cross-line patterns detected from literal `\n` or `\\n` in `pattern`. diff --git a/packages/coding-agent/src/prompts/tools/read.md b/packages/coding-agent/src/prompts/tools/read.md index 554dc4cab..a54242eff 100644 --- a/packages/coding-agent/src/prompts/tools/read.md +++ b/packages/coding-agent/src/prompts/tools/read.md @@ -1,4 +1,4 @@ -Read files, directories, archives, SQLite, images, documents, internal resources, and web URLs via one `path`. +Read files, directories, archives, SQLite, images, documents, internal resources, and web URLs via `path` plus optional `selector`. - SHOULD parallelize independent reads. @@ -7,7 +7,8 @@ Read files, directories, archives, SQLite, images, documents, internal resources ## Parameters -- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`, `ssh://`), or URL. Append `:` for ranges/modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`). +- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`, `ssh://`), or URL. Inline `:` still works for ranges/modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`). +- `selector` — optional selector without leading `:` (e.g. `"50-200"`, `"raw"`, `"raw:50-100"`, `"conflicts"`). Use when `path` contains literal colons: `{"path":"test:1-2","selector":"1-2"}`. ## Selectors @@ -72,6 +73,6 @@ All URI schemes take the same line selectors. `artifact://` recovers spilled `ssh://host/` reads a remote text file (UTF-8, ≤1 MiB) or lists a directory one level deep, on a pre-configured SSH host or `~/.ssh/config` alias; `ssh://host/` lists the remote root and bare `ssh://` lists the configured hosts. Files are also writable via `write` and searchable via `search`; a directory only lists (`search` refuses a directory, `write` refuses to overwrite one). A literal `:`, `?`, or `#` in the remote path must be percent-encoded (`%3A`/`%3F`/`%23`) — a trailing `:sel` is read as a line selector, and `?`/`#` start a URL query/fragment. Requires a POSIX login shell (`sh`/`bash`/`zsh`); a Windows host or a non-POSIX shell (fish, csh/tcsh) is rejected — use the `ssh` tool there. -- Line ranges go in the selector: `path="src/foo.ts:50-200"`. +- Literal colon filename + selector? Use `selector`, not recursive `path:"file:sel:sel"`. - Summary footer names elided ranges? Re-issue ONLY those ranges. NEVER guess `..`/`…` content. diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index afbbfa42d..0d27eac07 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -1525,10 +1525,19 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} // entries capture it at fetch time and are dropped at injection if a newer // mutation (any tool) bumped it in the meantime. const fileMutationVersions = new Map(); + const activeToolNames = new Set(); + const setActiveToolNames = (names: Iterable): void => { + activeToolNames.clear(); + for (const name of names) { + activeToolNames.add(name); + } + }; const toolSession: ToolSession = { get cwd() { return sessionManager.getCwd(); }, + isToolActive: name => activeToolNames.has(name), + setActiveToolNames, hasUI: options.hasUI ?? false, enableLsp, get hasEditTool() { @@ -2558,6 +2567,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} }); hasRegistered = true; + setActiveToolNames(initialToolNames); const { systemPrompt } = await logger.time( "buildSystemPrompt", rebuildSystemPrompt, @@ -2852,6 +2862,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} rebuildSystemPrompt, reloadSshTool, requestedToolNames: requestedToolNameSet, + setActiveToolNames, getMcpServerInstructions: mcpManager ? () => { const raw = mcpManager.getServerInstructions(); diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 37054c2a2..654e2f63b 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -35,6 +35,7 @@ import { type AsideMessage, type CompactionSummaryMessage, countTokens, + createToolScopedAbortReason, resolveTelemetry, type StreamFn, ThinkingLevel, @@ -182,7 +183,12 @@ import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles"; import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates"; import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier"; import type { Settings, SkillsSettings } from "../config/settings"; -import { getDefault, onAppendOnlyModeChanged, validateProviderMaxInFlightRequests } from "../config/settings"; +import { + getDefault, + onAppendOnlyModeChanged, + onModelRolesChanged, + validateProviderMaxInFlightRequests, +} from "../config/settings"; import { RawSseDebugBuffer } from "../debug/raw-sse-buffer"; import { loadCapability } from "../discovery"; import { expandApplyPatchToEntries, normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit"; @@ -240,7 +246,7 @@ import { theme } from "../modes/theme/theme"; import { parseTurnBudget } from "../modes/turn-budget"; import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink"; import { computeNonMessageBreakdown, computeNonMessageTokens } from "../modes/utils/context-usage"; -import { containsWorkflow, WORKFLOW_NOTICE } from "../modes/workflow"; +import { containsWorkflow, renderWorkflowNotice } from "../modes/workflow"; import { createPlanReadMatcher } from "../plan-mode/plan-protection"; import type { PlanModeState } from "../plan-mode/state"; import advisorSystemPrompt from "../prompts/advisor/system.md" with { type: "text" }; @@ -317,6 +323,7 @@ import { resolveFileDisplayMode } from "../utils/file-display-mode"; import { extractFileMentions, generateFileMentionMessages } from "../utils/file-mentions"; import { normalizeModelContextImages } from "../utils/image-loading"; import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback"; +import { formatLocalCalendarDate } from "../utils/local-date"; import { generateSessionTitle } from "../utils/title-generator"; import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice"; import type { AuthStorage } from "./auth-storage"; @@ -689,6 +696,8 @@ export interface AgentSessionConfig { toolRegistry?: Map; /** Tool names whose current registry entry is still the built-in implementation. */ builtInToolNames?: Iterable; + /** Update tool-session predicates that render guidance from the live active tool set. */ + setActiveToolNames?: (names: Iterable) => void; /** Current session pre-LLM message transform pipeline */ transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise; /** @@ -727,6 +736,8 @@ export interface AgentSessionConfig { convertToLlm?: (messages: AgentMessage[]) => Message[] | Promise; /** System prompt builder that can consider tool availability. Returns ordered provider-facing blocks. */ rebuildSystemPrompt?: (toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>; + /** Local calendar date provider used by prompt-cache invalidation. Defaults to the host local date. */ + getLocalCalendarDate?: () => string; /** Rebuild the SSH tool from current capability discovery results. */ reloadSshTool?: () => Promise; requestedToolNames?: ReadonlySet; @@ -873,6 +884,7 @@ export interface HandoffResult { export interface SessionHandoffOptions { autoTriggered?: boolean; signal?: AbortSignal; + onSwitchCancelled?: () => void; } /** Result from cycleModel() */ @@ -1562,6 +1574,7 @@ export class AgentSession { #cancelExitRecorder?: () => void; #exitRecorded = false; #unsubscribeAppendOnly?: () => void; + #unsubscribeModelRoles?: () => void; /** Last (enable, providerId) tuple resolved by `#syncAppendOnlyContext` — used to skip no-op invalidations. */ #lastAppendOnlyResolution?: { enable: boolean; providerId: string | undefined }; #eventListeners: AgentSessionEventListener[] = []; @@ -1728,8 +1741,10 @@ export class AgentSession { #rebuildSystemPrompt: | ((toolNames: string[], tools: Map) => Promise<{ systemPrompt: string[] }>) | undefined; + #getLocalCalendarDate: () => string; #getMcpServerInstructions: (() => Map | undefined) | undefined; #reloadSshTool: (() => Promise) | undefined; + #setActiveToolNames: ((names: Iterable) => void) | undefined; #disconnectOwnedMcpManager: (() => Promise) | undefined; #requestedToolNames: ReadonlySet | undefined; #baseSystemPrompt: string[]; @@ -2173,8 +2188,10 @@ export class AgentSession { }); this.#convertToLlm = config.convertToLlm ?? convertToLlm; this.#rebuildSystemPrompt = config.rebuildSystemPrompt; + this.#getLocalCalendarDate = config.getLocalCalendarDate ?? formatLocalCalendarDate; this.#getMcpServerInstructions = config.getMcpServerInstructions; this.#reloadSshTool = config.reloadSshTool; + this.#setActiveToolNames = config.setActiveToolNames; this.#disconnectOwnedMcpManager = config.disconnectOwnedMcpManager; this.#baseSystemPrompt = this.agent.state.systemPrompt; this.#promptModelKey = this.#currentPromptModelKey(); @@ -2271,6 +2288,11 @@ export class AgentSession { this.#unsubscribeAgent = this.agent.subscribe(this.#handleAgentEvent); // Re-evaluate append-only context mode when the setting changes at runtime. this.#unsubscribeAppendOnly = onAppendOnlyModeChanged(_value => this.#syncAppendOnlyContext(this.model)); + this.#unsubscribeModelRoles = onModelRolesChanged(() => { + if (!this.#advisorEnabled || this.#isDisposed) return; + if (this.#advisors.length > 0 && !this.#advisorRuntimeMatchesCurrentConfig()) this.#stopAdvisorRuntime(); + this.#buildAdvisorRuntime(true); + }); } // ------------------------------------------------------------------------- // Advisor runtime lifecycle @@ -2411,7 +2433,9 @@ export class AgentSession { #advisorRuntimeSignature(config: AdvisorConfig, slug: string, model: Model, thinkingLevel: ThinkingLevel): string { const tools = config.tools?.length ? config.tools.join("\u001e") : ""; const instructions = config.instructions?.trim() ?? ""; - return [config.name, slug, model.provider, model.id, thinkingLevel, tools, instructions].join("\u001f"); + return [config.name, slug, formatModelStringWithRouting(model), thinkingLevel, tools, instructions].join( + "\u001f", + ); } #advisorRuntimeMatchesCurrentConfig(): boolean { @@ -4711,7 +4735,8 @@ export class AgentSession { // Decide first: a non-interrupting tool-source match attaches to the // specific tool call's result instead of driving a loop-wide follow-up. const shouldInterrupt = this.#shouldInterruptForTtsrMatch(matches, matchContext); - const perToolId = shouldInterrupt ? undefined : this.#extractTtsrToolCallId(matchContext); + const matchedToolId = this.#extractTtsrToolCallId(matchContext); + const perToolId = shouldInterrupt ? undefined : matchedToolId; if (perToolId) { this.#addPerToolTtsrInjections(perToolId, matches); this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); @@ -4727,7 +4752,16 @@ export class AgentSession { // Abort the stream immediately — do not gate on extension callbacks this.#ttsrAbortPending = true; this.#ensureTtsrResumePromise(); - this.agent.abort(this.#formatTtsrAbortReason(matches)); + const abortReason = this.#formatTtsrAbortReason(matches); + this.agent.abort( + matchedToolId + ? createToolScopedAbortReason( + abortReason, + { [matchedToolId]: abortReason }, + "TTSR interrupt on another tool call", + ) + : abortReason, + ); // Notify extensions (fire-and-forget, does not block abort) this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {}); // Schedule retry after a short delay @@ -5800,6 +5834,10 @@ export class AgentSession { this.#unsubscribeAppendOnly(); this.#unsubscribeAppendOnly = undefined; } + if (this.#unsubscribeModelRoles) { + this.#unsubscribeModelRoles(); + this.#unsubscribeModelRoles = undefined; + } this.#eventListeners = []; } @@ -6343,6 +6381,7 @@ export class AgentSession { ), ); } + this.#setActiveToolNames?.(validToolNames); const activeNameSet = new Set(validToolNames); for (const name of Array.from(this.#selectedDiscoveredToolNames)) { if (!activeNameSet.has(name) || isMCPToolName(name) || !this.#toolRegistry.has(name)) { @@ -6448,6 +6487,7 @@ export class AgentSession { async refreshBaseSystemPrompt(): Promise { if (!this.#rebuildSystemPrompt) return; const activeToolNames = this.getActiveToolNames(); + this.#setActiveToolNames?.(activeToolNames); const built = await this.#rebuildSystemPrompt(activeToolNames, this.#toolRegistry); this.#baseSystemPrompt = built.systemPrompt; this.#baseSystemPromptBeforeMemoryPromotion = undefined; @@ -6570,7 +6610,7 @@ export class AgentSession { entries.sort(); instructionsSegment = entries.join("\u0006"); } - const date = new Date().toISOString().slice(0, 10); + const date = this.#getLocalCalendarDate(); return `${nameSegment}\u0003${descriptionSegment}\u0005${registrySegment}\u0007${instructionsSegment}|${date}`; } @@ -7387,11 +7427,15 @@ export class AgentSession { timestamp, }); } - if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text)) { + if ( + this.#magicKeywordEnabled("workflow") && + containsWorkflow(text) && + this.getActiveToolNames().includes("task") + ) { keywordNotices.push({ role: "custom", customType: "workflow-notice", - content: WORKFLOW_NOTICE, + content: renderWorkflowNotice({ taskBatch: this.settings.get("task.batch") }), display: false, attribution: "user", timestamp, @@ -8832,6 +8876,7 @@ export class AgentSession { const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model); + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(targetModel); this.sessionManager.appendModelChange(`${targetModel.provider}/${targetModel.id}`, role); @@ -8869,6 +8914,7 @@ export class AgentSession { const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model); + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(targetModel); this.sessionManager.appendModelChange( @@ -9028,6 +9074,7 @@ export class AgentSession { const next = scopedModels[nextIndex]; // Apply model + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(next.model)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(next.model); this.sessionManager.appendModelChange(`${next.model.provider}/${next.model.id}`); @@ -9058,6 +9105,7 @@ export class AgentSession { throw new Error(`No API key for ${nextModel.provider}/${nextModel.id}`); } + this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(nextModel)); this.#clearActiveRetryFallback(); this.#setModelWithProviderSessionReset(nextModel); this.sessionManager.appendModelChange(`${nextModel.provider}/${nextModel.id}`); @@ -10086,6 +10134,17 @@ export class AgentSession { // Start a new session const previousSessionFile = this.sessionFile; + if (this.#extensionRunner?.hasHandlers("session_before_switch")) { + const result = (await this.#extensionRunner.emit({ + type: "session_before_switch", + reason: "handoff", + })) as SessionBeforeSwitchResult | undefined; + + if (result?.cancel) { + options?.onSwitchCancelled?.(); + return undefined; + } + } await this.sessionManager.flush(); this.#cancelOwnAsyncJobs(); await this.sessionManager.newSession(previousSessionFile ? { parentSession: previousSessionFile } : undefined); @@ -10142,6 +10201,13 @@ export class AgentSession { this.agent.replaceMessages(sessionContext.messages); this.#resetAllAdvisorRuntimes(); this.#syncTodoPhasesFromBranch(); + if (this.#extensionRunner) { + await this.#extensionRunner.emit({ + type: "session_switch", + reason: "handoff", + previousSessionFile, + }); + } return { document: handoffText, savedPath }; } catch (error) { @@ -12358,13 +12424,17 @@ export class AgentSession { // queue, not the core steering queue (which handoff's agent.reset() would wipe). await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action }); if (action === "handoff") { + let handoffSwitchCancelled = false; const handoffFocus = AUTO_HANDOFF_THRESHOLD_FOCUS; const handoffResult = await this.handoff(handoffFocus, { autoTriggered: true, signal: autoCompactionSignal, + onSwitchCancelled: () => { + handoffSwitchCancelled = true; + }, }); if (!handoffResult) { - const aborted = autoCompactionSignal.aborted; + const aborted = autoCompactionSignal.aborted || handoffSwitchCancelled; if (aborted) { await this.#emitSessionEvent({ type: "auto_compaction_end", @@ -13265,6 +13335,7 @@ export class AgentSession { #resolveRetryFallbackRole(currentSelector: string): string | undefined { const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); if (!parsedCurrent) return undefined; + const chains = this.#getRetryFallbackChains(); const currentBaseSelector = formatRetryFallbackBaseSelector(parsedCurrent); const currentPlainSelector = this.model ? formatModelSelectorValue(formatModelString(this.model), parsedCurrent.thinkingLevel) @@ -13274,11 +13345,11 @@ export class AgentSession { ? formatRetryFallbackBaseSelector(parseRetryFallbackSelector(currentPlainSelector) ?? parsedCurrent) : undefined; - for (const role of Object.keys(this.#getRetryFallbackChains())) { + for (const role of Object.keys(chains)) { const primarySelector = this.#getRetryFallbackPrimarySelector(role); if (primarySelector?.raw === currentSelector) return role; } - for (const role of Object.keys(this.#getRetryFallbackChains())) { + for (const role of Object.keys(chains)) { const primarySelector = this.#getRetryFallbackPrimarySelector(role); if (!primarySelector) continue; if (currentPlainSelector && primarySelector.raw === currentPlainSelector) return role; @@ -13286,6 +13357,14 @@ export class AgentSession { if (primaryBaseSelector === currentBaseSelector) return role; if (currentPlainBaseSelector && primaryBaseSelector === currentPlainBaseSelector) return role; } + const defaultChain = chains.default; + if ( + Array.isArray(defaultChain) && + defaultChain.length > 0 && + this.#getRetryFallbackPrimarySelector("default") === undefined + ) { + return "default"; + } return undefined; } @@ -13304,9 +13383,27 @@ export class AgentSession { } #findRetryFallbackCandidates(role: string, currentSelector: string): RetryFallbackSelector[] { - const chain = this.#getRetryFallbackEffectiveChain(role); - if (chain.length <= 1) return []; + let chain = this.#getRetryFallbackEffectiveChain(role); const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry); + if (chain.length === 0 && role === "default" && parsedCurrent) { + const chains = this.#getRetryFallbackChains(); + const defaultChain = chains.default; + if ( + Array.isArray(defaultChain) && + defaultChain.length > 0 && + this.#getRetryFallbackPrimarySelector("default") === undefined + ) { + const seen = new Set([parsedCurrent.raw]); + chain = [parsedCurrent]; + for (const selector of defaultChain) { + const parsed = parseRetryFallbackSelector(selector, this.#modelRegistry); + if (!parsed || seen.has(parsed.raw)) continue; + seen.add(parsed.raw); + chain.push(parsed); + } + } + } + if (chain.length <= 1) return []; const currentBaseSelector = parsedCurrent ? formatRetryFallbackBaseSelector(parsedCurrent) : undefined; const currentPlainSelector = this.model && parsedCurrent diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index e64a98587..f8ddd76ca 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -24,6 +24,7 @@ import projectPromptTemplate from "./prompts/system/project-prompt.md" with { ty import systemPromptTemplate from "./prompts/system/system-prompt.md" with { type: "text" }; import { shortenPath } from "./tools/render-utils"; import { type ActiveRepoContext, resolveActiveRepoContext } from "./utils/active-repo-context"; +import { formatLocalCalendarDate } from "./utils/local-date"; import { normalizePromptPath } from "./utils/prompt-path"; import { AGENTS_MD_LIMIT, buildWorkspaceTree, type WorkspaceTree } from "./workspace-tree"; @@ -400,7 +401,7 @@ export async function loadSystemPromptFiles(options: LoadContextFilesOptions = { return userLevel?.content ?? null; } -export const DEFAULT_SYSTEM_PROMPT_TOOL_NAMES = ["read", "bash", "eval", "edit", "write"] as const; +export const DEFAULT_SYSTEM_PROMPT_TOOL_NAMES = ["read", "bash", "edit", "write"] as const; export interface SystemPromptToolMetadata { label: string; @@ -693,7 +694,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): } } - const date = new Date().toISOString().slice(0, 10); + const date = formatLocalCalendarDate(); const dateTime = date; const promptCwd = shortenPath(normalizePromptPath(resolvedCwd)); const activeRepoContextPrompt = renderActiveRepoContextPrompt(activeRepoContext); diff --git a/packages/coding-agent/src/tools/bash-interactive.ts b/packages/coding-agent/src/tools/bash-interactive.ts index 3c1a57e6c..fc39cb0c8 100644 --- a/packages/coding-agent/src/tools/bash-interactive.ts +++ b/packages/coding-agent/src/tools/bash-interactive.ts @@ -300,7 +300,7 @@ export async function runInteractiveBashPty( options: { command: string; cwd: string; - timeoutMs: number; + timeoutMs?: number; signal?: AbortSignal; env?: Record; artifactPath?: string; diff --git a/packages/coding-agent/src/tools/bash-skill-urls.ts b/packages/coding-agent/src/tools/bash-skill-urls.ts index 82e9e5f8e..585882a9c 100644 --- a/packages/coding-agent/src/tools/bash-skill-urls.ts +++ b/packages/coding-agent/src/tools/bash-skill-urls.ts @@ -140,6 +140,30 @@ function unquoteToken(token: string): string { return token; } +function isInsideShellQuote(command: string, index: number): boolean { + let quote: "'" | '"' | undefined; + for (let i = 0; i < index; i++) { + const char = command[i]; + if (char === "\\" && quote !== "'") { + i++; + continue; + } + if (char === "'" && quote !== '"') { + quote = quote === "'" ? undefined : "'"; + continue; + } + if (char === '"' && quote !== "'") { + quote = quote === '"' ? undefined : '"'; + } + } + return quote !== undefined; +} + +function isEmbeddedInQuotedText(command: string, token: string, index: number): boolean { + if (token.startsWith("'") || token.startsWith('"')) return false; + return isInsideShellQuote(command, index); +} + /** Shell-escape a path using single quotes. */ function shellEscape(p: string): string { return `'${p.replace(/'/g, "'\\''")}'`; @@ -216,6 +240,7 @@ export function expandSkillUrls(command: string, skills: readonly Skill[]): stri /** * Expand supported internal URLs in a bash command string to shell-escaped absolute paths. + * Unresolvable URLs and literal mentions inside larger quoted text are left unchanged. * Supported schemes: skill://, agent://, artifact://, memory://, rule://, local:// */ export async function expandInternalUrls(command: string, options: InternalUrlExpansionOptions): Promise { @@ -231,15 +256,22 @@ export async function expandInternalUrls(command: string, options: InternalUrlEx const index = match.index; if (index === undefined) continue; + if (isEmbeddedInQuotedText(command, token, index)) continue; + const rawUrl = unquoteToken(token); const url = normalizeLocalScheme(rawUrl); - const resolvedPath = await resolveInternalUrlToPath( - url, - options.skills, - options.internalRouter, - options.localOptions, - options.ensureLocalParentDirs, - ); + let resolvedPath: string; + try { + resolvedPath = await resolveInternalUrlToPath( + url, + options.skills, + options.internalRouter, + options.localOptions, + options.ensureLocalParentDirs, + ); + } catch { + continue; + } const replacement = options.noEscape ? resolvedPath : shellEscape(resolvedPath); expanded = `${expanded.slice(0, index)}${replacement}${expanded.slice(index + token.length)}`; } diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index ad8d8faff..14e67375b 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -27,6 +27,7 @@ import { type BashInteractiveResult, runInteractiveBashPty } from "./bash-intera import { checkBashInterception } from "./bash-interceptor"; import { canUseInteractiveBashPty } from "./bash-pty-selection"; import { expandInternalUrls, type InternalUrlExpansionOptions } from "./bash-skill-urls"; +import { resolveEvalBackends } from "./eval-backends"; import { invalidateGithubCacheForBashCommand } from "./gh-cache-invalidation"; import { formatStyledTruncationWarning, @@ -131,7 +132,7 @@ async function saveBashOriginalArtifact(session: ToolSession, originalText: stri } } -const BASH_TIMEOUT_DESCRIPTION = `timeout in seconds; clamped to ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}`; +const BASH_TIMEOUT_DESCRIPTION = `timeout in seconds; 0 disables the command deadline; nonzero values are clamped to ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}`; const bashSchemaBase = type({ command: type("string").describe("command to execute"), @@ -166,6 +167,7 @@ export interface BashToolDetails { meta?: OutputMeta; timeoutSeconds?: number; requestedTimeoutSeconds?: number; + timeoutDisabled?: boolean; wallTimeMs?: number; /** Exit code of a command that ran to completion but failed (non-zero). */ exitCode?: number; @@ -375,7 +377,24 @@ export class BashTool implements AgentTool this.session.isToolActive?.(name) ?? fallback; + return prompt.render(bashDescription, { + asyncEnabled: this.#asyncEnabled, + autoBackgroundEnabled: this.#autoBackgroundEnabled, + autoBackgroundThresholdSeconds: Math.max(0, Math.floor(this.#autoBackgroundThresholdMs / 1000)), + hasAstGrep: isToolActive("ast_grep", this.session.settings.get("astGrep.enabled")), + hasAstEdit: isToolActive("ast_edit", this.session.settings.get("astEdit.enabled")), + hasGrep: isToolActive("grep", this.session.settings.get("grep.enabled")), + hasGlob: isToolActive("glob", this.session.settings.get("glob.enabled")), + hasRead: isToolActive("read", true), + hasEval: isToolActive( + "eval", + evalBackends.python || evalBackends.js || evalBackends.ruby || evalBackends.julia, + ), + }); + } readonly parameters: BashToolSchema; // Non-pty calls run alongside each other (the executor isolates overlapping // runs on the same shell session); pty takes over the terminal UI and must @@ -397,15 +416,6 @@ export class BashTool implements AgentTool { const details: BashToolDetails = { - timeoutSeconds: timeoutSec, async: { state: "running", jobId, type: "bash" }, }; + if (timeoutSec === undefined) { + details.timeoutDisabled = true; + } else { + details.timeoutSeconds = timeoutSec; + } if (options.requestedTimeoutSec !== undefined && options.requestedTimeoutSec !== timeoutSec) { details.requestedTimeoutSeconds = options.requestedTimeoutSec; } @@ -539,8 +560,8 @@ export class BashTool implements AgentTool ({ kind: "timeout" as const })); + const timeoutPromise = timeoutMs + ? Bun.sleep(timeoutMs).then(() => ({ kind: "timeout" as const })) + : undefined; // Poll until the process exits, times out, or the caller aborts. for (;;) { const racers: Array> = [ exitPromise.then(s => ({ kind: "exit" as const, status: s })), - timeoutPromise, Bun.sleep(250).then(() => ({ kind: "poll" as const })), ]; + if (timeoutPromise) racers.push(timeoutPromise); if (signal) { racers.push(abortedP.then(() => ({ kind: "aborted" as const }))); } @@ -1053,7 +1081,7 @@ export class BashTool implements AgentTool(config: ShellRendererConfig) { const showingFullOutput = expanded && renderContext?.isFullOutput === true; // Build truncation warning - const timeoutSeconds = details?.timeoutSeconds ?? renderContext?.timeout; + const timeoutDisabled = details?.timeoutDisabled === true || renderContext?.timeout === 0; + const timeoutSeconds = timeoutDisabled ? undefined : (details?.timeoutSeconds ?? renderContext?.timeout); const requestedTimeoutSeconds = details?.requestedTimeoutSeconds; const wallTimeMs = details?.wallTimeMs; const statsParts: string[] = []; if (wallTimeMs !== undefined) { statsParts.push(`Wall: ${formatWallTimeSeconds(wallTimeMs)}s`); } + if (timeoutDisabled) { + statsParts.push("Timeout: disabled"); + } if (typeof timeoutSeconds === "number") { statsParts.push( requestedTimeoutSeconds !== undefined && requestedTimeoutSeconds !== timeoutSeconds diff --git a/packages/coding-agent/src/tools/browser/launch.ts b/packages/coding-agent/src/tools/browser/launch.ts index d9c9bb846..38344bff7 100644 --- a/packages/coding-agent/src/tools/browser/launch.ts +++ b/packages/coding-agent/src/tools/browser/launch.ts @@ -29,15 +29,20 @@ export const DEFAULT_VIEWPORT = { width: 1365, height: 768, deviceScaleFactor: 1 * connection dropped, etc.). */ export const BROWSER_PROTOCOL_TIMEOUT_MS = 60_000; +const ENABLE_AUTOMATION_FLAG = "--enable-automation"; // Automation-tell launch flags that puppeteer-core adds by default. We suppress // them via `ignoreDefaultArgs` (the supported escape hatch) to mirror xxxx's -// chromiumSwitches patch. `--enable-automation` is the loudest: it sets +// chromiumSwitches patch. `--enable-automation` is the loudest: it normally sets // navigator.webdriver=true and shows the "controlled by automated software" infobar. +// Edge is the launch-stability exception: it can exit before CDP opens when this +// default flag is stripped, so Edge keeps Puppeteer's flag while our explicit +// `--disable-blink-features=AutomationControlled` launch arg still handles +// navigator.webdriver. // `ignoreDefaultArgs` does exact-string matching, so each entry must be a flag that // puppeteer emits verbatim. The default `--disable-features=...` string can't be // matched this way; it is neutralized in the puppeteer-core patch (ChromeLauncher). const STEALTH_IGNORE_DEFAULT_ARGS = [ - "--enable-automation", + ENABLE_AUTOMATION_FLAG, "--disable-extensions", "--disable-default-apps", "--disable-component-extensions-with-background-pages", @@ -47,6 +52,23 @@ const STEALTH_IGNORE_DEFAULT_ARGS = [ "--disable-ipc-flooding-protection", "--metrics-recording-only", ]; + +function isMicrosoftEdgeExecutable(executablePath: string | undefined): boolean { + if (!executablePath) return false; + const normalizedPath = executablePath.replaceAll("\\", "/").toLowerCase(); + const executableName = normalizedPath.slice(normalizedPath.lastIndexOf("/") + 1); + return ( + executableName === "msedge.exe" || + executableName === "microsoft edge" || + executableName.startsWith("microsoft-edge") + ); +} + +function stealthIgnoreDefaultArgs(executablePath: string | undefined): string[] { + if (!isMicrosoftEdgeExecutable(executablePath)) return [...STEALTH_IGNORE_DEFAULT_ARGS]; + return STEALTH_IGNORE_DEFAULT_ARGS.filter(arg => arg !== ENABLE_AUTOMATION_FLAG); +} + const STEALTH_ACCEPT_LANGUAGE = "en-US,en"; const USER_AGENT_TARGET_TIMEOUT_MS = 5_000; @@ -282,12 +304,13 @@ export async function launchHeadlessBrowser(opts: LaunchHeadlessOptions): Promis if (ignoreCert === "true" || ignoreCert === "1" || ignoreCert === "yes" || ignoreCert === "on") { launchArgs.push("--ignore-certificate-errors"); } + const executablePath = await ensureChromiumExecutable(); return await puppeteer.launch({ headless: opts.headless, defaultViewport: opts.headless ? initialViewport : null, - executablePath: await ensureChromiumExecutable(), + executablePath, args: launchArgs, - ignoreDefaultArgs: [...STEALTH_IGNORE_DEFAULT_ARGS], + ignoreDefaultArgs: stealthIgnoreDefaultArgs(executablePath), protocolTimeout: BROWSER_PROTOCOL_TIMEOUT_MS, }); } @@ -737,6 +760,10 @@ export async function applyStealthPatches( await injectStealthScripts(page); } +export function stealthIgnoreDefaultArgsForTest(executablePath: string | undefined): string[] { + return stealthIgnoreDefaultArgs(executablePath); +} + export function targetSupportsUserAgentOverrideForTest(target: Target): boolean { return targetSupportsUserAgentOverride(target); } diff --git a/packages/coding-agent/src/tools/grep.ts b/packages/coding-agent/src/tools/grep.ts index dbc5f3371..393078133 100644 --- a/packages/coding-agent/src/tools/grep.ts +++ b/packages/coding-agent/src/tools/grep.ts @@ -48,12 +48,14 @@ import { type LineRange, parseLineRanges, pathTargetsSsh, + probeLiteralPathExists, type ResolvedSearchTarget, resolveReadPath, resolveToolSearchScope, selectorLineRanges, splitInternalUrlSel, splitPathAndSel, + splitPathAndSelPreferringLiteral, toPathList, } from "./path-utils"; import { @@ -77,6 +79,9 @@ const searchSchema = type({ "path?": searchPathEntry.describe( 'file, directory, glob, internal URL, or ":" selector to search; pass several as a semicolon-delimited list ("src; tests"). Omitted -> searches the workspace root (".")', ), + "selector?": type("string").describe( + 'line selector without a leading colon (e.g. "50-100", "50+10", "50-100,200-300"); keeps `path` literal when filenames contain colons', + ), "case?": type("boolean").describe("case-sensitive search"), "gitignore?": type("boolean").describe("respect gitignore"), "skip?": type("number") @@ -119,6 +124,7 @@ const SEARCH_GREP_TIMEOUT_MS = 30_000; interface GrepPathSpec { original: string; clean: string; + literalFilesystemMatch?: boolean; ranges?: [LineRange, ...LineRange[]]; } @@ -147,9 +153,38 @@ function isReadSelectorGrammar(sel: string): boolean { return lower === "raw" || lower === "conflicts" || parseLineRanges(sel) !== null; } -function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] { +async function parsePathSpecs( + rawEntries: readonly string[], + cwd: string, + explicitSelector?: string, +): Promise { + const explicitRanges = + explicitSelector === undefined || explicitSelector.length === 0 ? undefined : parseLineRanges(explicitSelector); + if (explicitSelector !== undefined && !explicitRanges) { + throw new ToolError( + `selector "${explicitSelector}" is invalid — use line ranges like "50-100", "50+10", or "50-100,200-300" without a leading colon`, + ); + } const specs: GrepPathSpec[] = []; for (const entry of rawEntries) { + if (explicitRanges) { + // Separate selector parameter makes `path` deterministic: first try the + // exact local filesystem path (with read-path normalization), then let + // archive/internal/URL resolution handle non-literal structured paths. + const rawPathHasScheme = /^[a-z][a-z0-9+.-]*:\/\//i.test(entry); + const probe = rawPathHasScheme ? "missing" : await probeLiteralPathExists(entry, cwd); + // `"unknown"` covers EACCES/IO where we cannot confirm existence — treat + // it as a literal so a real file such as `test:1-2` under an unreadable + // parent is never silently reinterpreted as `test` + selector. + const literalMatch = probe !== "missing"; + specs.push({ + original: entry, + clean: literalMatch && !rawPathHasScheme ? resolveReadPath(entry, cwd) : entry, + literalFilesystemMatch: literalMatch, + ranges: explicitRanges, + }); + continue; + } // Internal URLs (`artifact://`, `skill://`, …) use the URL-aware splitter, // which peels selector-shaped tails only for selector-capable schemes and // leaves opaque ones (`mcp://`) intact. Unlike filesystem paths, their @@ -168,10 +203,14 @@ function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] { specs.push({ original: entry, clean: internalSplit.path, ranges: selectorLineRanges(internalSplit.sel) }); continue; } - const split = splitPathAndSel(entry); - let clean = entry; + // Prefer a literal filesystem match when one exists — a real file named + // `test:1-2` outranks the `:1-2` selector interpretation (issue #4618). + const strictSplit = splitPathAndSel(entry); + const split = await splitPathAndSelPreferringLiteral(entry, cwd); + const literalFilesystemMatch = strictSplit.sel !== undefined && split.sel === undefined; + let clean = literalFilesystemMatch ? resolveReadPath(entry, cwd) : entry; let ranges: [LineRange, ...LineRange[]] | undefined; - if (split.sel) { + if (!literalFilesystemMatch && split.sel) { const parsed = parseLineRanges(split.sel); if (!parsed) { throw new ToolError( @@ -184,7 +223,7 @@ function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] { clean = split.path; ranges = parsed; } - specs.push({ original: entry, clean, ranges }); + specs.push({ original: entry, clean, literalFilesystemMatch, ranges }); } return specs; } @@ -220,7 +259,7 @@ function matchAbsolutePath(matchPath: string, searchPath: string): string { * cleanup hook the caller MUST invoke in a `finally`. */ async function resolveArchiveSearchPaths( - paths: string[], + pathSpecs: readonly GrepPathSpec[], cwd: string, ): Promise<{ resolvedPaths: string[]; @@ -229,17 +268,18 @@ async function resolveArchiveSearchPaths( unreadable: string[]; cleanup: () => Promise; }> { - const resolvedPaths = paths.slice(); + const resolvedPaths = pathSpecs.map(spec => spec.clean); const displayMap = new Map(); const displaySet = new Set(); const unreadable: string[] = []; let tempDir: string | undefined; const archiveCache = new Map(); - for (let idx = 0; idx < paths.length; idx++) { - const entry = paths[idx]; + for (let idx = 0; idx < pathSpecs.length; idx++) { + const spec = pathSpecs[idx]; + if (!spec || spec.literalFilesystemMatch) continue; + const entry = spec.clean; const candidates = parseArchivePathCandidates(entry); - // Longest archive prefix first; we want the one whose member portion is non-empty. const member = candidates.find(c => c.subPath !== "" && c.archivePath !== entry); if (!member) continue; @@ -879,7 +919,7 @@ export class GrepTool implements AgentTool _onUpdate?: AgentToolUpdateCallback, _toolContext?: AgentToolContext, ): Promise> { - const { pattern, path: rawPath, case: caseSensitive, gitignore, skip } = params; + const { pattern, path: rawPath, selector, case: caseSensitive, gitignore, skip } = params; return untilAborted(signal, async () => { // Preserve the pattern verbatim — leading/trailing whitespace is @@ -897,8 +937,7 @@ export class GrepTool implements AgentTool const scopedPaths = toPathList(rawPath); const effectivePaths = scopedPaths.length > 0 ? scopedPaths : ["."]; const rawEntries = await expandDelimitedPathEntries(effectivePaths, this.session.cwd); - const pathSpecs = parsePathSpecs(rawEntries); - const paths = pathSpecs.map(spec => spec.clean); + const pathSpecs = await parsePathSpecs(rawEntries, this.session.cwd, selector); const materializedExternalPaths = new Map(); const materializeExternalUrlForSearch = async (rawPath: string) => { const target = parseReadUrlTarget(rawPath); @@ -917,7 +956,7 @@ export class GrepTool implements AgentTool displaySet: archiveDisplaySet, unreadable: archiveUnreadable, cleanup: cleanupArchiveScratch, - } = await resolveArchiveSearchPaths(paths, this.session.cwd); + } = await resolveArchiveSearchPaths(pathSpecs, this.session.cwd); try { const internalResolution = await resolveInternalSearchInputs({ pathSpecs, diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index f42e49741..8ea30c85d 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -224,6 +224,10 @@ export interface ToolSession { getAgentId?: () => string | null; /** Look up a registered tool by name (used by the eval js backend's tool bridge). */ getToolByName?: (name: string) => AgentTool | undefined; + /** Return whether a built-in tool is active in this turn's tool set. */ + isToolActive?: (name: string) => boolean; + /** Update the active built-in tool predicate when a session changes tools mid-run. */ + setActiveToolNames?: (names: Iterable) => void; /** Agent registry for IRC routing across live sessions. */ agentRegistry?: AgentRegistry; /** Get artifacts directory for artifact:// URLs */ @@ -647,6 +651,13 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P ...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []), ]; + const activeToolNames = new Set(baseEntries.map(([name]) => name)); + if (session.setActiveToolNames) { + session.setActiveToolNames(activeToolNames); + } else { + session.isToolActive = name => activeToolNames.has(name); + } + const baseResults = await Promise.all( baseEntries.map(async ([name, factory]) => { const tool = await logger.time(`createTools:${name}`, factory as ToolFactory, session); diff --git a/packages/coding-agent/src/tools/path-utils.ts b/packages/coding-agent/src/tools/path-utils.ts index b0e51eb3a..cc5b787c6 100644 --- a/packages/coding-agent/src/tools/path-utils.ts +++ b/packages/coding-agent/src/tools/path-utils.ts @@ -315,6 +315,46 @@ export function splitPathAndSel(rawPath: string): { path: string; sel?: string } return { path: basePath, sel }; } +/** + * Three-way probe for whether the exact filesystem entry named by `filePath` + * exists. `stat` (used earlier) failed for reasons other than "no such file" + * (dangling symlink, `EACCES` on a parent, transient I/O), and each of those + * silently reinterpreted a real literal path such as `test:1-2` as `test` + * plus selector `1-2` (issue #4618). `lstat` inspects the entry itself, so a + * dangling symlink is still detected as present; ambiguous errors resolve to + * `"unknown"` so callers keep the raw path instead of guessing. + */ +export async function probeLiteralPathExists(filePath: string, cwd: string): Promise<"exists" | "missing" | "unknown"> { + const resolved = resolveReadPath(filePath, cwd); + try { + await fs.promises.lstat(resolved); + return "exists"; + } catch (err) { + if (isEnoent(err) || isEnotdir(err)) return "missing"; + return "unknown"; + } +} + +/** + * Async sibling of {@link splitPathAndSel} that prefers a literal filesystem + * path over selector interpretation. Filenames whose tail matches the selector + * grammar (e.g. `test:1-2`, `log:raw`) are legal on POSIX; without this the + * strict splitter peels the tail and both `read` and `grep` refuse to open the + * real file (issue #4618). The literal wins on a confirmed `lstat`, and also + * on `"unknown"` (`EACCES` on a parent, transient I/O), so an unreachable + * literal is never silently reinterpreted as `path + selector`. Only a + * definitive `ENOENT`/`ENOTDIR` falls back to the strict split. + */ +export async function splitPathAndSelPreferringLiteral( + rawPath: string, + cwd: string, +): Promise<{ path: string; sel?: string }> { + const strict = splitPathAndSel(rawPath); + if (strict.sel === undefined) return strict; + const probe = await probeLiteralPathExists(rawPath, cwd); + return probe === "missing" ? strict : { path: rawPath }; +} + /** * Variant of {@link splitPathAndSel} for internal URLs (`scheme://...`). * @@ -669,7 +709,12 @@ export async function splitDelimitedPathEntry( const normalizedEntry = normalizePathLikeInput(entry); if (!hasTopLevelPathDelimiter(normalizedEntry)) return null; if (isInternalUrlPath(normalizedEntry)) return null; - + // A real POSIX file may contain the delimiter and a selector-shaped tail + // (`a;b:1-2`, `a b:1-2`). Preserve the raw entry whenever the full literal + // resolves — or is only ambiguous — so downstream literal-preferring + // splitters see it before delimiter expansion peels or splits (issue #4618 + // reviewer feedback: delimited expansion ran before the literal check). + if ((await probeLiteralPathExists(normalizedEntry, cwd)) !== "missing") return null; const splitter = options.splitter ?? parseSearchPath; const peeledEntry = splitPathAndSel(normalizedEntry).path; if (!hasGlobPathChars(peeledEntry) && (await delimitedPathPartResolves(normalizedEntry, cwd, splitter))) { diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index cd838ca42..14edfad2b 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -99,10 +99,12 @@ import { type LineRange, parseLineRanges, pathTargetsSsh, + probeLiteralPathExists, resolveReadPath, splitDelimitedPathEntry, splitInternalUrlSel, splitPathAndSel, + splitPathAndSelPreferringLiteral, } from "./path-utils"; import { formatBytes, replaceTabs, shortenPath, wrapBrackets } from "./render-utils"; import { @@ -746,7 +748,10 @@ function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; membe const readSchema = type({ path: type("string").describe( - 'Local path, internal URI (e.g. "omp://", "issue://123", "pr://123"), or URL; append : for line ranges or raw mode (e.g. "src/foo.ts:50-100")', + 'Local path, internal URI (e.g. "omp://", "issue://123", "pr://123"), or URL. Inline : is still accepted for compatibility.', + ), + "selector?": type("string").describe( + 'selector without a leading colon (e.g. "50-100", "raw", "raw:50-100", "conflicts"); keeps `path` literal when filenames contain colons', ), }); @@ -2113,6 +2118,14 @@ export class ReadTool implements AgentTool { _toolContext?: AgentToolContext, ): Promise> { let { path: readPath } = params; + let explicitSelector = params.selector?.trim(); + let explicitParsedSelector = explicitSelector === undefined ? undefined : parseSel(explicitSelector); + if ( + params.selector !== undefined && + (explicitSelector === undefined || explicitSelector.length === 0 || explicitParsedSelector?.kind === "none") + ) { + throw invalidSelector(params.selector); + } if (readPath.startsWith("file://")) { readPath = expandPath(readPath); } @@ -2133,40 +2146,55 @@ export class ReadTool implements AgentTool { if (!this.session.settings.get("fetch.enabled")) { throw new ToolError("URL reads are disabled by settings."); } - if (parsedUrlTarget.ranges !== undefined) { + if (explicitParsedSelector?.kind === "conflicts") { + throw new ToolError("The explicit read selector `conflicts` is only supported for local files."); + } + const urlRaw = + explicitParsedSelector === undefined ? parsedUrlTarget.raw : isRawSelector(explicitParsedSelector); + const urlRanges = + explicitParsedSelector?.kind === "lines" ? explicitParsedSelector.ranges : parsedUrlTarget.ranges; + if (urlRanges !== undefined && urlRanges.length > 1) { const cached = await loadReadUrlCacheEntry( this.session, - { path: parsedUrlTarget.path, raw: parsedUrlTarget.raw }, + { path: parsedUrlTarget.path, raw: urlRaw }, signal, { ensureArtifact: true, preferCached: true }, ); - return this.#buildInMemoryMultiRangeResult(cached.output, parsedUrlTarget.ranges, { + return this.#buildInMemoryMultiRangeResult(cached.output, urlRanges, { details: { ...cached.details }, sourceUrl: cached.details.finalUrl, entityLabel: "URL output", - raw: parsedUrlTarget.raw, + raw: urlRaw, immutable: true, }); } - if (parsedUrlTarget.offset !== undefined || parsedUrlTarget.limit !== undefined) { + const urlRange = urlRanges?.[0]; + const urlOffset = explicitParsedSelector?.kind === "lines" ? urlRange?.startLine : parsedUrlTarget.offset; + const urlLimit = + explicitParsedSelector?.kind === "lines" && urlRange + ? urlRange.endLine !== undefined + ? urlRange.endLine - urlRange.startLine + 1 + : undefined + : parsedUrlTarget.limit; + if (urlOffset !== undefined || urlLimit !== undefined) { const cached = await loadReadUrlCacheEntry( this.session, - { path: parsedUrlTarget.path, raw: parsedUrlTarget.raw }, + { path: parsedUrlTarget.path, raw: urlRaw }, signal, { ensureArtifact: true, preferCached: true, }, ); - return this.#buildInMemoryTextResult(cached.output, parsedUrlTarget.offset, parsedUrlTarget.limit, { + return this.#buildInMemoryTextResult(cached.output, urlOffset, urlLimit, { details: { ...cached.details }, sourceUrl: cached.details.finalUrl, entityLabel: "URL output", - raw: parsedUrlTarget.raw, + raw: urlRaw, immutable: true, }); } - return executeReadUrl(this.session, { path: parsedUrlTarget.path, raw: parsedUrlTarget.raw }, signal); + return executeReadUrl(this.session, { path: parsedUrlTarget.path, raw: urlRaw }, signal); } // Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://, omp://, issue://, pr://). @@ -2174,8 +2202,9 @@ export class ReadTool implements AgentTool { // off the URL and surfaced via parseSel rather than confusing handlers. const internalRouter = InternalUrlRouter.instance(); if (internalRouter.canHandle(readPath)) { - const internalTarget = splitInternalUrlSel(readPath); - const parsed = parseSel(internalTarget.sel); + const internalTarget = + explicitSelector === undefined ? splitInternalUrlSel(readPath) : { path: readPath, sel: explicitSelector }; + const parsed = explicitParsedSelector ?? parseSel(internalTarget.sel); if (internalTarget.sel !== undefined && parsed.kind === "none") { throw new ToolError( `Invalid selector ':${internalTarget.sel}' on '${internalTarget.path}'. Use :N, :N-M, :N+K, :N- (open-ended), a comma-separated list of ranges, :raw, or a range combined with raw (e.g. :raw:50-100).`, @@ -2192,7 +2221,16 @@ export class ReadTool implements AgentTool { skills: this.session.skills, }); if (localFile) { - readPath = internalTarget.sel === undefined ? localFile.path : `${localFile.path}:${internalTarget.sel}`; + readPath = localFile.path; + // Promote the URL-embedded selector into the explicit-selector state so + // downstream literal-preferring routing does NOT re-split the synthesized + // `${localFile.path}:${sel}` string — a sibling literal file at that name + // would otherwise shadow the intended local:// URL selector semantics + // (issue #4618 reviewer feedback on c493d12). + if (explicitSelector === undefined && internalTarget.sel !== undefined) { + explicitSelector = internalTarget.sel; + explicitParsedSelector = parsed; + } } else { return this.#handleInternalUrl(internalTarget.path, parsed, signal); } @@ -2205,48 +2243,68 @@ export class ReadTool implements AgentTool { // resolution share misses instead of re-globbing the workspace. const suffixCache: SuffixMatchCache = new Map(); - const archivePath = await this.#resolveArchiveReadPath(readPath, suffixCache, signal); - if (archivePath) { - const archiveSubPath = splitPathAndSel(archivePath.archiveSubPath); - const archiveParsed = parseSel(archiveSubPath.sel); - return this.#readArchive( - readPath, - archiveParsed, - { ...archivePath, archiveSubPath: archiveSubPath.path }, - signal, - ); - } + // Prefer a literal filesystem match over selector interpretation so real + // POSIX filenames containing selector-looking suffixes win over structured + // archive / sqlite / pdf-image dispatch. With explicit `selector`, `path` + // is exact: `path: "test:1-2", selector: "1-2"` means "lines 1-2 from + // the literal file test:1-2", without recursively depending on whether a + // longer `test:1-2:1-2` filename also exists (issue #4618). + const literalSplit = + explicitSelector === undefined + ? await splitPathAndSelPreferringLiteral(readPath, this.session.cwd) + : { path: readPath, sel: explicitSelector }; + const rawPathIsLiteral = + explicitSelector !== undefined + ? readPath.includes(":") && (await probeLiteralPathExists(readPath, this.session.cwd)) !== "missing" + : literalSplit.sel === undefined && splitPathAndSel(readPath).sel !== undefined; - const sqlitePath = await this.#resolveSqliteReadPath(readPath, suffixCache, signal); - if (sqlitePath) { - return this.#readSqlite(sqlitePath, signal); - } - - const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath); - if (pdfImageMemberPath) { - let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd); - let suffixResolution: { from: string; to: string } | undefined; - try { - const stat = await Bun.file(absolutePdfPath).stat(); - if (stat.isDirectory()) - throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`); - } catch (error) { - if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error; - const suffixMatch = await this.#findSuffixMatchCached(suffixCache, pdfImageMemberPath.pdfPath, signal); - if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`); - absolutePdfPath = suffixMatch.absolutePath; - suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath }; + if (!rawPathIsLiteral) { + const archivePath = await this.#resolveArchiveReadPath(readPath, suffixCache, signal); + if (archivePath) { + const archiveSubPath = + explicitSelector === undefined + ? splitPathAndSel(archivePath.archiveSubPath) + : { path: archivePath.archiveSubPath, sel: explicitSelector }; + const archiveParsed = parseSel(archiveSubPath.sel); + return this.#readArchive( + readPath, + archiveParsed, + { ...archivePath, archiveSubPath: archiveSubPath.path }, + signal, + ); + } + + const sqlitePath = await this.#resolveSqliteReadPath(readPath, suffixCache, signal); + if (sqlitePath) { + return this.#readSqlite(sqlitePath, signal); + } + + const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath); + if (pdfImageMemberPath) { + let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd); + let suffixResolution: { from: string; to: string } | undefined; + try { + const stat = await Bun.file(absolutePdfPath).stat(); + if (stat.isDirectory()) + throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`); + } catch (error) { + if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error; + const suffixMatch = await this.#findSuffixMatchCached(suffixCache, pdfImageMemberPath.pdfPath, signal); + if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`); + absolutePdfPath = suffixMatch.absolutePath; + suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath }; + } + return this.#readPdfImageMember( + absolutePdfPath, + pdfImageMemberPath.pdfPath, + pdfImageMemberPath.member, + suffixResolution, + signal, + ); } - return this.#readPdfImageMember( - absolutePdfPath, - pdfImageMemberPath.pdfPath, - pdfImageMemberPath.member, - suffixResolution, - signal, - ); } - const localTarget = splitPathAndSel(readPath); + const localTarget = literalSplit; const localReadPath = localTarget.path; const parsed = parseSel(localTarget.sel); diff --git a/packages/coding-agent/src/utils/local-date.ts b/packages/coding-agent/src/utils/local-date.ts new file mode 100644 index 000000000..80962f102 --- /dev/null +++ b/packages/coding-agent/src/utils/local-date.ts @@ -0,0 +1,7 @@ +/** formatLocalCalendarDate formats a Date as YYYY-MM-DD in the host local timezone. */ +export function formatLocalCalendarDate(date: Date = new Date()): string { + const year = date.getFullYear(); + const month = String(date.getMonth() + 1).padStart(2, "0"); + const day = String(date.getDate()).padStart(2, "0"); + return `${year}-${month}-${day}`; +} diff --git a/packages/coding-agent/src/utils/open.ts b/packages/coding-agent/src/utils/open.ts index ca3390438..413e9ca2e 100644 --- a/packages/coding-agent/src/utils/open.ts +++ b/packages/coding-agent/src/utils/open.ts @@ -32,22 +32,48 @@ function getExistingWslLocalPath(urlOrPath: string): string | undefined { } /** - * Resolve the Windows `rundll32.exe` command used to hand a URL/path to the - * user's registered protocol handler. Anchoring to `%SystemRoot%\System32` - * (rather than relying on `rundll32` being on `PATH`) survives environments - * where the machine `PATH` no longer references `System32` — a common - * real-world misconfiguration where `System32\Wbem` / `WindowsPowerShell` / - * `OpenSSH` survive but `System32` itself is dropped. Bare `rundll32` on - * such boxes throws `Executable not found in $PATH: "rundll32"` from - * `Bun.spawn` before ShellExecute ever sees the URL. + * Resolve the Windows opener used to hand a URL/path to the user's registered + * protocol handler. PowerShell's `Start-Process` goes through ShellExecute + * like the previous `rundll32 url.dll,FileProtocolHandler`, with two + * advantages that make the delayed-failure telemetry in {@link openPath} + * actually observable on Windows: + * + * - `rundll32` exits 0 unconditionally, so no launch failure ever reaches the + * non-zero-exit logging below. `Start-Process` surfaces the failures + * ShellExecute itself reports — missing target file, no handler executable, + * access denied — as exit code 1 (verified live: a nonexistent file path + * exits 1; `$ErrorActionPreference='Stop'` additionally promotes any + * non-terminating error classes). Known limitation shared by every opener: + * an unregistered URL scheme exits 0 because Windows "handles" it by + * offering the app-picker. + * - `-EncodedCommand` carries the target as a UTF-16LE/base64 payload, so no + * cmd/PowerShell metacharacter parsing ever sees it (OAuth authorize URLs + * carry `&`); inside the decoded script the target is a single-quoted + * literal (no `$` expansion) with embedded quotes doubled. + * + * PowerShell is anchored to `%SystemRoot%\System32` for the same reason the + * previous revision anchored `rundll32`: machine PATHs that dropped + * `System32` are a real-world occurrence, and bare names throw + * `Executable not found in $PATH` from `Bun.spawn`. A bare-name fallback + * remains for exotic SystemRoot layouts. */ function windowsOpenerCommand(target: string): string[] { const systemRoot = process.env.SystemRoot?.trim() || process.env.SYSTEMROOT?.trim() || "C:\\Windows"; // `path.win32` (not the platform-adaptive `path.join`) keeps Windows path // separators when tests run under a POSIX host and matches Windows call // conventions on the real target. - const rundll32 = path.win32.join(systemRoot, "System32", "rundll32.exe"); - return [rundll32, "url.dll,FileProtocolHandler", target]; + const absolute = path.win32.join(systemRoot, "System32", "WindowsPowerShell", "v1.0", "powershell.exe"); + const powershell = fs.existsSync(absolute) ? absolute : "powershell.exe"; + const script = `$ErrorActionPreference='Stop';Start-Process '${target.replaceAll("'", "''")}'`; + return [ + powershell, + "-NoProfile", + "-NonInteractive", + "-WindowStyle", + "Hidden", + "-EncodedCommand", + Buffer.from(script, "utf16le").toString("base64"), + ]; } /** Open a URL or file path in the default browser/application. Best-effort, never throws. */ export function openPath(urlOrPath: string): void { diff --git a/packages/coding-agent/test/advisor-toggle.test.ts b/packages/coding-agent/test/advisor-toggle.test.ts index 0812a895b..b22067375 100644 --- a/packages/coding-agent/test/advisor-toggle.test.ts +++ b/packages/coding-agent/test/advisor-toggle.test.ts @@ -1,4 +1,5 @@ import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs"; import * as path from "node:path"; import { Agent, type AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { Model } from "@oh-my-pi/pi-ai"; @@ -6,9 +7,10 @@ import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; -import { TempDir } from "@oh-my-pi/pi-utils"; +import { getProjectAgentDir, TempDir } from "@oh-my-pi/pi-utils"; describe("AgentSession advisor toggle", () => { let sharedDir: TempDir; @@ -22,6 +24,7 @@ describe("AgentSession advisor toggle", () => { authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db")); authStorage.setRuntimeApiKey("anthropic", "test-key"); authStorage.setRuntimeApiKey("openai", "test-key"); + authStorage.setRuntimeApiKey("openrouter", "test-key"); modelRegistry = new ModelRegistry(authStorage); const bundled = getBundledModel("anthropic", "claude-sonnet-4-5"); const replacement = getBundledModel("openai", "gpt-4o-mini"); @@ -98,6 +101,89 @@ describe("AgentSession advisor toggle", () => { expect(session.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id); }); + it("refreshes the live advisor when the advisor role setting changes", () => { + session.settings.setModelRole("advisor", `${model.provider}/${model.id}`); + expect(session.setAdvisorEnabled(true)).toBe(true); + expect(session.getAdvisorAgent()?.state.model.provider).toBe(model.provider); + expect(session.getAdvisorAgent()?.state.model.id).toBe(model.id); + + session.settings.setModelRole("advisor", `${replacementModel.provider}/${replacementModel.id}`); + + expect(session.getAdvisorAgent()?.state.model.provider).toBe(replacementModel.provider); + expect(session.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id); + }); + + it("refreshes the live advisor when only the advisor route changes", () => { + session.settings.setModelRole("advisor", "openrouter/z-ai/glm-4.7@cerebras"); + expect(session.setAdvisorEnabled(true)).toBe(true); + expect(session.getAdvisorAgent()?.state.model.provider).toBe("openrouter"); + expect(session.getAdvisorAgent()?.state.model.id).toBe("z-ai/glm-4.7"); + expect( + (session.getAdvisorAgent()?.state.model.compat as { openRouterRouting?: { only?: string[] } } | undefined) + ?.openRouterRouting?.only, + ).toEqual(["cerebras"]); + + session.settings.setModelRole("advisor", "openrouter/z-ai/glm-4.7@fireworks"); + + expect(session.getAdvisorAgent()?.state.model.provider).toBe("openrouter"); + expect(session.getAdvisorAgent()?.state.model.id).toBe("z-ai/glm-4.7"); + expect( + (session.getAdvisorAgent()?.state.model.compat as { openRouterRouting?: { only?: string[] } } | undefined) + ?.openRouterRouting?.only, + ).toEqual(["fireworks"]); + }); + + it("refreshes the live advisor after project model-role reloads", async () => { + const projectA = path.join(tempDir.path(), "project-a"); + const projectB = path.join(tempDir.path(), "project-b"); + const agentDir = path.join(tempDir.path(), "agent"); + fs.mkdirSync(getProjectAgentDir(projectA), { recursive: true }); + fs.mkdirSync(getProjectAgentDir(projectB), { recursive: true }); + fs.mkdirSync(agentDir, { recursive: true }); + await Bun.write( + path.join(getProjectAgentDir(projectA), "settings.json"), + JSON.stringify({ modelRoles: { advisor: `${model.provider}/${model.id}` } }), + ); + await Bun.write( + path.join(getProjectAgentDir(projectB), "settings.json"), + JSON.stringify({ modelRoles: { advisor: `${replacementModel.provider}/${replacementModel.id}` } }), + ); + + const settings = await Settings.loadIsolated({ + cwd: projectA, + agentDir, + overrides: { "compaction.enabled": false }, + }); + const customSession = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }), + sessionManager: SessionManager.create(tempDir.path(), tempDir.path()), + settings, + modelRegistry, + advisorTools: [], + }); + + try { + expect(customSession.setAdvisorEnabled(true)).toBe(true); + expect(customSession.getAdvisorAgent()?.state.model.provider).toBe(model.provider); + expect(customSession.getAdvisorAgent()?.state.model.id).toBe(model.id); + + await settings.reloadForCwd(projectB); + + expect(customSession.getAdvisorAgent()?.state.model.provider).toBe(replacementModel.provider); + expect(customSession.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id); + } finally { + await customSession.dispose(); + AgentStorage.resetInstance(); + } + }); + it("keeps explicit enable idempotent when the advisor config is unchanged", () => { session.settings.setModelRole("advisor", `${model.provider}/${model.id}`); expect(session.setAdvisorEnabled(true)).toBe(true); diff --git a/packages/coding-agent/test/agent-session-concurrent.test.ts b/packages/coding-agent/test/agent-session-concurrent.test.ts index 2660f7d84..7b4b045d6 100644 --- a/packages/coding-agent/test/agent-session-concurrent.test.ts +++ b/packages/coding-agent/test/agent-session-concurrent.test.ts @@ -1235,6 +1235,126 @@ describe("AgentSession TTSR resume gate", () => { expect(text).not.toContain("Request was aborted"); }); + it("labels only the matching aborted tool placeholder with the TTSR rule reason", async () => { + collapseSchedulerSettleDelays(); + const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; + let streamCallCount = 0; + + const ttsrManager = new TtsrManager({ + enabled: true, + contextMode: "discard", + interruptMode: "always", + repeatMode: "once", + repeatGap: 10, + }); + ttsrManager.addRule(testRule); + + const readToolCallContent: ToolCall = { + type: "toolCall", + id: "call_innocent_read", + name: "read", + arguments: { path: "history://Eval1WithSkill" }, + }; + const matchedToolCallContent: ToolCall = { + type: "toolCall", + id: "call_ttsr_abort_reason", + name: "mock_edit", + arguments: { snippet: "let val = result.unwrap(" }, + }; + + const makeToolCallMsg = (stopReason: "toolUse" | "aborted" = "toolUse"): AssistantMessage => ({ + role: "assistant", + content: [readToolCallContent, matchedToolCallContent], + api: "anthropic-messages", + provider: "anthropic", + model: "mock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason, + timestamp: Date.now(), + }); + + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model, systemPrompt: ["Test"], tools: [] }, + streamFn: (_model, _context, options) => { + streamCallCount++; + const stream = new AssistantMessageEventStream(); + const signal = options?.signal; + if (streamCallCount === 1) { + queueMicrotask(() => { + const partial = makeToolCallMsg(); + if (signal) { + signal.addEventListener( + "abort", + () => { + stream.push({ + type: "error", + reason: "aborted", + error: makeToolCallMsg("aborted"), + }); + }, + { once: true }, + ); + } + stream.push({ type: "start", partial }); + stream.push({ type: "toolcall_start", contentIndex: 1, partial }); + stream.push({ + type: "toolcall_delta", + contentIndex: 1, + delta: 'let val = result.unwrap("oops")', + partial, + }); + // The abort placeholder is only minted for tool calls that reached + // `toolcall_end`: the agent loop drops incomplete tool calls from an + // aborted turn (partial args are unsafe to replay). Complete the + // innocent read before the rule-driven abort fires so its placeholder + // survives and can carry the neutral sibling label. + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: readToolCallContent, partial }); + }); + } else { + pushContinuationStream(stream, () => {}); + } + return stream; + }, + }); + + const sessionManager = SessionManager.inMemory(); + const settings = Settings.isolated(); + const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-abort-reason.db")); + authStorages.push(authStorage); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + session = new AgentSession({ agent, sessionManager, settings, modelRegistry, ttsrManager }); + + await session.prompt("Write some Rust code"); + + const toolResults = sessionManager + .getEntries() + .filter(entry => entry.type === "message" && entry.message.role === "toolResult") + .map(entry => (entry.type === "message" && entry.message.role === "toolResult" ? entry.message : undefined)) + .filter(message => message !== undefined); + const toolResultText = (toolCallId: string): string => + toolResults + .find(message => message.toolCallId === toolCallId) + ?.content.find((part): part is { type: "text"; text: string } => part.type === "text")?.text ?? ""; + + const readText = toolResultText(readToolCallContent.id); + expect(readText).toContain("Tool execution was aborted: TTSR interrupt on another tool call"); + expect(readText).not.toContain("TTSR matched rule: no-unwrap"); + // The matching call never reached `toolcall_end`, so the loop drops it from + // the aborted turn (partial args are unsafe to replay) and no placeholder is + // minted. The rule label for a completed matching call is covered by the + // single-call test above. + expect(toolResultText(matchedToolCallContent.id)).toBe(""); + }); + it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => { collapseSchedulerSettleDelays(); const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; diff --git a/packages/coding-agent/test/agent-session-handoff.test.ts b/packages/coding-agent/test/agent-session-handoff.test.ts index 3cb9a6288..47c30fe9c 100644 --- a/packages/coding-agent/test/agent-session-handoff.test.ts +++ b/packages/coding-agent/test/agent-session-handoff.test.ts @@ -158,6 +158,90 @@ describe("AgentSession handoff", () => { expect(sessionManager.getEntries().filter(entry => entry.type === "compaction")).toHaveLength(0); }); + it("emits handoff lifecycle hooks on the outgoing and replacement sessions", async () => { + const extensionsResult = await loadExtensions([], tempDir.path()); + const extensionRunner = new ExtensionRunner( + extensionsResult.extensions, + extensionsResult.runtime, + tempDir.path(), + sessionManager, + modelRegistry, + ); + const observedEvents: Array<{ + type: "session_before_switch" | "session_switch"; + reason: string; + previousSessionFile: string | undefined; + activeSessionFile: string | undefined; + messageCount: number; + handoffEntryCount: number; + }> = []; + vi.spyOn(extensionRunner, "hasHandlers").mockImplementation(eventName => eventName === "session_before_switch"); + const emit = extensionRunner.emit.bind(extensionRunner); + vi.spyOn(extensionRunner, "emit").mockImplementation(event => { + if (event.type === "session_before_switch" || event.type === "session_switch") { + observedEvents.push({ + type: event.type, + reason: event.reason, + previousSessionFile: event.type === "session_switch" ? event.previousSessionFile : undefined, + activeSessionFile: session.sessionFile, + messageCount: sessionManager.getBranch().filter(entry => entry.type === "message").length, + handoffEntryCount: sessionManager + .getBranch() + .filter(entry => entry.type === "custom_message" && entry.customType === "handoff").length, + }); + } + return emit(event); + }); + + await session.dispose(); + session = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }), + sessionManager, + settings: Settings.isolated({ + "compaction.enabled": true, + "compaction.autoContinue": false, + }), + modelRegistry, + extensionRunner, + obfuscator, + }); + const previousSessionFile = session.sessionFile; + const generateHandoffSpy = vi + .spyOn(compactionModule, "generateHandoffFromContext") + .mockResolvedValue("## Goal\nContinue from here"); + + await session.handoff(); + + const nextSessionFile = session.sessionFile; + expect(generateHandoffSpy).toHaveBeenCalledTimes(1); + expect(nextSessionFile).not.toBe(previousSessionFile); + expect(observedEvents).toEqual([ + { + type: "session_before_switch", + reason: "handoff", + previousSessionFile: undefined, + activeSessionFile: previousSessionFile, + messageCount: 2, + handoffEntryCount: 0, + }, + { + type: "session_switch", + reason: "handoff", + previousSessionFile, + activeSessionFile: nextSessionFile, + messageCount: 0, + handoffEntryCount: 1, + }, + ]); + }); + it("runs handoff generation through the configured side stream function", async () => { const handoffText = "## Goal\nContinue via side stream"; let sideStreamCalls = 0; @@ -1336,6 +1420,7 @@ describe("AgentSession handoff", () => { expect(handoffSpy).toHaveBeenCalledWith(expect.stringContaining("Threshold-triggered maintenance"), { autoTriggered: true, signal: expect.anything(), + onSwitchCancelled: expect.any(Function), }); expect(events.filter(event => event.type === "auto_compaction_start")).toHaveLength(1); const endEvents = events.filter(event => event.type === "auto_compaction_end"); @@ -1598,6 +1683,89 @@ describe("AgentSession handoff", () => { }); }); + it("treats a vetoed auto-handoff switch as cancelled instead of falling back", async () => { + session.settings.set("compaction.strategy", "handoff"); + session.settings.set("compaction.thresholdPercent", 1); + session.settings.set("contextPromotion.enabled", false); + + const model = session.model; + if (!model) { + throw new Error("Expected model to be set"); + } + + const extensionsResult = await loadExtensions([], tempDir.path()); + const extensionRunner = new ExtensionRunner( + extensionsResult.extensions, + extensionsResult.runtime, + tempDir.path(), + sessionManager, + modelRegistry, + ); + vi.spyOn(extensionRunner, "hasHandlers").mockImplementation(eventName => eventName === "session_before_switch"); + const emitSpy = vi.spyOn(extensionRunner, "emit").mockImplementation((async () => ({ + cancel: true, + })) as ExtensionRunner["emit"]); + + await session.dispose(); + session = new AgentSession({ + agent: new Agent({ + initialState: { + model, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + }), + sessionManager, + settings: session.settings, + modelRegistry, + extensionRunner, + obfuscator, + }); + session.subscribe(event => { + events.push(event); + }); + const previousSessionFile = session.sessionFile; + const generateHandoffSpy = vi + .spyOn(compactionModule, "generateHandoffFromContext") + .mockResolvedValue("## Goal\nContinue from here"); + const assistantMessage: AssistantMessage = { + role: "assistant", + content: [{ type: "text", text: "maintenance trigger" }], + api: model.api, + provider: model.provider, + model: model.id, + stopReason: "stop", + usage: { + input: 10_000, + output: 1_000, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 11_000, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + timestamp: Date.now(), + }; + + session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage }); + session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] }); + await waitFor(() => events.filter(event => event.type === "auto_compaction_end").length === 1); + + expect(generateHandoffSpy).toHaveBeenCalledTimes(1); + expect(emitSpy).toHaveBeenCalledWith({ type: "session_before_switch", reason: "handoff" }); + expect(emitSpy).not.toHaveBeenCalledWith(expect.objectContaining({ type: "session_switch" })); + expect(session.sessionFile).toBe(previousSessionFile); + expect(sessionManager.getEntries().filter(entry => entry.type === "compaction")).toHaveLength(0); + const endEvents = events.filter(event => event.type === "auto_compaction_end"); + expect(endEvents).toHaveLength(1); + expect(endEvents[0]).toMatchObject({ + type: "auto_compaction_end", + action: "handoff", + aborted: true, + willRetry: false, + }); + }); + it("resets to the base system prompt before generating a handoff", async () => { const model = session.model; if (!model) { diff --git a/packages/coding-agent/test/agent-session-magic-keywords.test.ts b/packages/coding-agent/test/agent-session-magic-keywords.test.ts index d0e09a7ae..3a10dfe8b 100644 --- a/packages/coding-agent/test/agent-session-magic-keywords.test.ts +++ b/packages/coding-agent/test/agent-session-magic-keywords.test.ts @@ -2,7 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; -import { Agent } from "@oh-my-pi/pi-agent-core"; +import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; import { Effort } from "@oh-my-pi/pi-ai"; import { getBundledModel } from "@oh-my-pi/pi-catalog/models"; import * as autoThinkingClassifier from "@oh-my-pi/pi-coding-agent/auto-thinking/classifier"; @@ -13,8 +13,20 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import { AUTO_THINKING } from "@oh-my-pi/pi-coding-agent/thinking"; import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import { type } from "arktype"; -async function createMagicKeywordSession(root: string): Promise<{ +const mockTaskTool: AgentTool = { + name: "task", + label: "Task", + description: "Mock task tool", + parameters: type({}), + execute: async () => ({ content: [{ type: "text" as const, text: "ok" }] }), +}; + +async function createMagicKeywordSession( + root: string, + tools: AgentTool[] = [mockTaskTool], +): Promise<{ session: AgentSession; settings: Settings; authStorage: AuthStorage; @@ -25,7 +37,7 @@ async function createMagicKeywordSession(root: string): Promise<{ initialState: { model, systemPrompt: ["Test"], - tools: [], + tools, messages: [], thinkingLevel: Effort.High, }, @@ -103,6 +115,34 @@ describe("AgentSession magic keyword settings", () => { ]); }); + it("renders workflowz notice for the active task schema", async () => { + const created = await createMagicKeywordSession(root); + session = created.session; + authStorage = created.authStorage; + created.settings.set("task.batch", false); + const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined); + + await session.prompt("please workflowz this"); + + const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ content?: string; customType?: string }>; + const notice = promptMessages.find(message => message.customType === "workflow-notice")?.content ?? ""; + expect(notice).toContain("once per independent subagent"); + expect(notice).toContain("Do not pass `context` or `tasks[]`"); + expect(notice).not.toContain("Call `task` once per independent fan-out batch"); + }); + + it("skips workflowz notice when the task tool is inactive", async () => { + const created = await createMagicKeywordSession(root, []); + session = created.session; + authStorage = created.authStorage; + const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined); + + await session.prompt("please workflowz this"); + + const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ customType?: string }>; + expect(promptMessages.map(message => message.customType).filter(Boolean)).toEqual([]); + }); + it("does not use a disabled ultrathink keyword to force auto thinking", async () => { const created = await createMagicKeywordSession(root); session = created.session; diff --git a/packages/coding-agent/test/agent-session-retry-cap.test.ts b/packages/coding-agent/test/agent-session-retry-cap.test.ts index d9b8d1333..1efa5c1bd 100644 --- a/packages/coding-agent/test/agent-session-retry-cap.test.ts +++ b/packages/coding-agent/test/agent-session-retry-cap.test.ts @@ -282,6 +282,79 @@ describe("AgentSession retry delay cap", () => { expect(last.content).toContainEqual({ type: "text", text: "recovered after credential switch" }); }); + it("switches same-provider credentials before model fallback on ChatGPT usage limits", async () => { + const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + const fallbackModel = getBundledModel("openai", "gpt-5.5"); + if (!primaryModel || !fallbackModel) { + throw new Error("Expected bundled primary and fallback test models to exist"); + } + + authStorage.removeRuntimeApiKey("anthropic"); + authStorage.setRuntimeApiKey("openai", "openai-fallback-key"); + await authStorage.set("anthropic", [ + { type: "api_key", key: "anthropic-key-1" }, + { type: "api_key", key: "anthropic-key-2" }, + ]); + + const usageLimitError = "Error: You have hit your ChatGPT usage limit (k12 plan). Try again in ~231 min."; + const mock = createMockModel(); + const requestedModels: string[] = []; + const requestedKeys: string[] = []; + let agent!: Agent; + agent = new Agent({ + getApiKey: model => modelRegistry.resolver(model, agent.sessionId), + initialState: { + model: primaryModel, + systemPrompt: ["Test"], + tools: [], + messages: [], + }, + streamFn: (requestedModel, context, options) => { + requestedModels.push(`${requestedModel.provider}/${requestedModel.id}`); + const apiKey = resolveInitialApiKey(options?.apiKey); + requestedKeys.push(apiKey); + if (requestedKeys.length === 1) { + mock.push({ throw: usageLimitError }); + } else { + mock.push({ content: ["recovered after sibling account"] }); + } + return mock.stream(requestedModel, context, options); + }, + }); + + const settings = Settings.isolated({ + "compaction.enabled": false, + "retry.baseDelayMs": 5, + "retry.maxDelayMs": 100, + "retry.maxRetries": 1, + "retry.modelFallback": true, + "retry.fallbackChains": { + default: [`${fallbackModel.provider}/${fallbackModel.id}`], + }, + }); + settings.setModelRole("default", `${primaryModel.provider}/${primaryModel.id}`); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + }); + + vi.spyOn(scheduler, "wait").mockResolvedValue(undefined); + await session.prompt("Trigger k12 usage limit"); + await session.waitForIdle(); + + expect(requestedModels).toEqual([ + `${primaryModel.provider}/${primaryModel.id}`, + `${primaryModel.provider}/${primaryModel.id}`, + ]); + expect([...requestedKeys].sort()).toEqual(["anthropic-key-1", "anthropic-key-2"]); + const last = lastAssistant(session); + expect(last.stopReason).toBe("stop"); + expect(last.content).toContainEqual({ type: "text", text: "recovered after sibling account" }); + }); + it("waits for the earliest sibling unblock instead of failing the delay cap", async () => { // Regression: with every sibling credential momentarily blocked (e.g. a // short post-401 or usage-probe block), a usage-limit 429 with a diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 4ae3b5c1d..345d42df3 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -219,6 +219,59 @@ describe("AgentSession retry fallback", () => { ]); }); + it("uses the active initial model as the default fallback primary when other role fallback chains are configured", async () => { + const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); + const fallbackModel = getBundledModel("openai", "gpt-4o-mini"); + const otherRoleFallbackModel = getBundledModel("openai", "gpt-4o"); + if (!primaryModel || !fallbackModel || !otherRoleFallbackModel) { + throw new Error("Expected bundled test models to exist"); + } + + const requestedModels: string[] = []; + const fallbackAppliedEvents: Array> = []; + const agent = createFallbackAgent(primaryModel, requestedModels); + + const settings = Settings.isolated({ + "compaction.enabled": false, + "retry.maxRetries": 1, + "retry.fallbackChains": { + default: [`${fallbackModel.provider}/${fallbackModel.id}`], + smol: [`${otherRoleFallbackModel.provider}/${otherRoleFallbackModel.id}`], + }, + }); + + session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings, + modelRegistry, + }); + + session.subscribe(event => { + if (event.type === "retry_fallback_applied") { + fallbackAppliedEvents.push(event); + } + }); + + await session.prompt("Recover using implicit default primary"); + await session.waitForIdle(); + + expect(requestedModels).toEqual([ + `${primaryModel.provider}/${primaryModel.id}`, + `${fallbackModel.provider}/${fallbackModel.id}`, + ]); + expect(session.model?.provider).toBe(fallbackModel.provider); + expect(session.model?.id).toBe(fallbackModel.id); + expect(fallbackAppliedEvents).toEqual([ + { + type: "retry_fallback_applied", + from: `${primaryModel.provider}/${primaryModel.id}`, + to: `${fallbackModel.provider}/${fallbackModel.id}`, + role: "default", + }, + ]); + }); + it("falls back on structured classifier refusals and pins the fallback", async () => { const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); const fallbackModel = getBundledModel("openai", "gpt-4o-mini"); diff --git a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts index 6cefd9c5b..00b909753 100644 --- a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts +++ b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts @@ -1,4 +1,4 @@ -import { afterEach, describe, expect, it, setSystemTime } from "bun:test"; +import { afterEach, describe, expect, it } from "bun:test"; import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; import type { Model } from "@oh-my-pi/pi-ai"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; @@ -68,6 +68,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { interface NewSessionOptions { mcpDiscoveryEnabled?: boolean; getMcpServerInstructions?: () => Map | undefined; + getLocalCalendarDate?: () => string; } function newSession( @@ -101,6 +102,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { }), mcpDiscoveryEnabled: options.mcpDiscoveryEnabled, getMcpServerInstructions: options.getMcpServerInstructions, + getLocalCalendarDate: options.getLocalCalendarDate, }); sessions.push(session); return { session }; @@ -176,6 +178,52 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { expect(rebuildCount).toBe(baseline + 2); }); + it("updates live active-tool predicates before rebuilding the prompt", async () => { + const activeToolNames = new Set(["read", "bash", "grep"]); + const readTool = createBasicTool("read", "Read"); + const bashTool = createBasicTool("bash", "Bash"); + const grepTool = createBasicTool("grep", "Grep"); + Object.defineProperty(bashTool, "description", { + get: () => (activeToolNames.has("grep") ? "bash sees grep" : "bash hides grep"), + enumerable: true, + configurable: true, + }); + const toolRegistry = new Map([ + [readTool.name, readTool], + [bashTool.name, bashTool], + [grepTool.name, grepTool], + ]); + const agent = new Agent({ + initialState: { + model: createModel(), + systemPrompt: ["initial"], + tools: [readTool, bashTool, grepTool], + messages: [], + }, + }); + const session = new AgentSession({ + agent, + sessionManager: SessionManager.inMemory(), + settings: Settings.isolated({ "compaction.enabled": false }), + modelRegistry: {} as never, + toolRegistry, + setActiveToolNames: names => { + activeToolNames.clear(); + for (const name of names) { + activeToolNames.add(name); + } + }, + rebuildSystemPrompt: async (_toolNames, tools) => ({ + systemPrompt: [tools.get("bash")?.description ?? "missing bash"], + }), + }); + sessions.push(session); + + await session.setActiveToolsByName(["read", "bash"]); + + expect(agent.state.systemPrompt).toEqual(["bash hides grep"]); + }); + it("does not skip when refreshBaseSystemPrompt is called explicitly", async () => { let rebuildCount = 0; const { session } = newSession(async toolNames => { @@ -388,40 +436,38 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { await session.refreshMCPTools([dynamicTool]); expect(rebuildCount).toBe(baseline + 1); }); - it("rebuilds when the calendar date rolls over between tool-stable MCP refreshes", async () => { - // `buildSystemPrompt` injects today's date into the prompt body. - // A session spanning midnight must not serve yesterday's date after an MCP - // reconnect that happens to bring an identical tool set. - setSystemTime(new Date("2025-01-01T23:59:58Z")); - try { - let rebuildCount = 0; - const { session } = newSession(async toolNames => { + it("rebuilds when the local calendar date rolls over between tool-stable MCP refreshes", async () => { + // `buildSystemPrompt` injects today's local date into the prompt body. The + // signature reads the same date provider so a session spanning local midnight + // must rebuild after an MCP reconnect with an otherwise identical tool set. + let currentDate = "2026-06-30"; + let rebuildCount = 0; + const { session } = newSession( + async toolNames => { rebuildCount++; return `tools:${toolNames.join(",")}`; - }); - const tool = createMcpCustomTool("mcp__nucleus_search", "nucleus", "search", "Search"); + }, + { getLocalCalendarDate: () => currentDate }, + ); + const tool = createMcpCustomTool("mcp__nucleus_search", "nucleus", "search", "Search"); - // First refresh: no signature yet, must rebuild. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(1); + // First refresh: no signature yet, must rebuild. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(1); - // Same tools, same day: signature matches, skip. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(1); + // Same tools, same local day: signature matches, skip. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(1); - // Advance past midnight. - setSystemTime(new Date("2025-01-02T00:00:01Z")); + currentDate = "2026-07-01"; - // Same tools, new calendar day: date segment changed, must rebuild. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(2); + // Same tools, new local calendar day: date segment changed, must rebuild. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(2); - // Same tools, same new day: skip again. - await session.refreshMCPTools([tool]); - expect(rebuildCount).toBe(2); - } finally { - setSystemTime(); // restore real time - } + // Same tools, same new local day: skip again. + await session.refreshMCPTools([tool]); + expect(rebuildCount).toBe(2); }); it("does not rebuild when MCP server instructions change only beyond the 4000-char truncation boundary", async () => { // `rebuildSystemPrompt` (sdk.ts) truncates each server instruction to 4000 chars diff --git a/packages/coding-agent/test/auth-storage-rotation.test.ts b/packages/coding-agent/test/auth-storage-rotation.test.ts index c1278e9b3..bf7e3e6c5 100644 --- a/packages/coding-agent/test/auth-storage-rotation.test.ts +++ b/packages/coding-agent/test/auth-storage-rotation.test.ts @@ -95,4 +95,54 @@ describe("AuthStorage account rotation", () => { const exhaustedFallbackKey = await authStorage.getApiKey("openai-codex", sessionId); expect(exhaustedFallbackKey).toMatch(/^api-acct-/); }); + + test("usage-limit rotation can match the failed bearer when session stickiness is missing", async () => { + await authStorage.set("openai-codex", [ + { + type: "oauth", + access: "access-1", + refresh: "refresh-1", + expires: Date.now() + 60_000, + accountId: "acct-1", + }, + { + type: "oauth", + access: "access-2", + refresh: "refresh-2", + expires: Date.now() + 60_000, + accountId: "acct-2", + }, + ]); + + const sessionId = "missing-sticky-session"; + const result = await authStorage.markUsageLimitReached("openai-codex", sessionId, { apiKey: "access-1" }); + expect(result.switched).toBe(true); + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-2"); + }); + + test("usage-limit rotation trusts the failed bearer over stale session stickiness", async () => { + await authStorage.set("openai-codex", [ + { + type: "oauth", + access: "plus-access", + refresh: "plus-refresh", + expires: Date.now() + 60_000, + accountId: "plus-acct", + }, + { + type: "oauth", + access: "k12-access", + refresh: "k12-refresh", + expires: Date.now() + 60_000, + accountId: "k12-acct", + }, + ]); + + const sessionId = "stale-sticky-session"; + const stickyKey = await authStorage.getApiKey("openai-codex", sessionId); + const failedKey = stickyKey === "api-plus-acct" ? "k12-access" : "plus-access"; + const result = await authStorage.markUsageLimitReached("openai-codex", sessionId, { apiKey: failedKey }); + expect(result.switched).toBe(true); + expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe(stickyKey); + }); }); diff --git a/packages/coding-agent/test/bash-executor.test.ts b/packages/coding-agent/test/bash-executor.test.ts index c8436d7da..b11a15f96 100644 --- a/packages/coding-agent/test/bash-executor.test.ts +++ b/packages/coding-agent/test/bash-executor.test.ts @@ -402,6 +402,15 @@ exit 64 expect(result.output).not.toContain("done"); }); + it("does not arm a deadline when timeout is zero", async () => { + if (process.platform === "win32") { + return; + } + const result = await executeBash("sleep 1.2; echo done", { cwd: tempDir, timeout: 0 }); + expect(result.cancelled).toBe(false); + expect(result.output.trim()).toBe("done"); + }); + it("aborts commands", async () => { if (process.platform === "win32") { return; diff --git a/packages/coding-agent/test/discovery/builtin-rules-md.test.ts b/packages/coding-agent/test/discovery/builtin-rules-md.test.ts index ff39625f5..3d7721fef 100644 --- a/packages/coding-agent/test/discovery/builtin-rules-md.test.ts +++ b/packages/coding-agent/test/discovery/builtin-rules-md.test.ts @@ -1,11 +1,9 @@ /** - * Regression test for #1266: + * Regression tests for top-level `RULES.md` sticky rules. + * * `RULES.md` (singular, top-level) MUST be loaded as a sticky always-apply rule * from both `~/.omp/agent/RULES.md` (user) and the nearest `.omp/RULES.md` * (project, walked up from cwd to repoRoot). - * - * Calls the native provider's `load` directly with the agent dir pointed at a - * tempdir (via setAgentDir) so the user scope can be staged in isolation. */ import { afterEach, beforeEach, expect, test } from "bun:test"; import * as fs from "node:fs"; @@ -15,8 +13,8 @@ import { getCapability } from "@oh-my-pi/pi-coding-agent/capability"; import { clearCache } from "@oh-my-pi/pi-coding-agent/capability/fs"; import { type Rule, ruleCapability } from "@oh-my-pi/pi-coding-agent/capability/rule"; import type { LoadContext } from "@oh-my-pi/pi-coding-agent/capability/types"; -// Register all discovery providers as a side effect. -import "@oh-my-pi/pi-coding-agent/discovery"; +// Importing discovery registers all providers as a side effect. +import { loadCapability } from "@oh-my-pi/pi-coding-agent/discovery"; import { getConfigRootDir, removeSyncWithRetries, setAgentDir } from "@oh-my-pi/pi-utils"; let tempDir: string; @@ -40,6 +38,11 @@ async function loadNativeRules(ctx: LoadContext): Promise { return result.items; } +async function loadRulesCapability(cwd: string): Promise { + const result = await loadCapability(ruleCapability.id, { cwd, providers: ["native"] }); + return result.items; +} + beforeEach(() => { clearCache(); tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-rules-md-")); @@ -81,7 +84,7 @@ test("project .omp/RULES.md becomes an alwaysApply rule", async () => { const rules = await loadNativeRules({ cwd: project, home, repoRoot: project }); - const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES"); + const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES@project"); expect(projectRule).toBeDefined(); expect(projectRule?.alwaysApply).toBe(true); expect(projectRule?.content).toContain("Always say hi."); @@ -94,12 +97,48 @@ test("project RULES.md is found walking up from a sub-package cwd", async () => const rules = await loadNativeRules({ cwd: subPkg, home, repoRoot: project }); - const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES"); + const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES@project"); expect(projectRule).toBeDefined(); expect(projectRule?.alwaysApply).toBe(true); expect(projectRule?.path).toBe(path.join(project, ".omp", "RULES.md")); }); +test("user and project sticky RULES.md both survive public capability dedup", async () => { + const userRulesPath = path.join(home, ".omp", "agent", "RULES.md"); + const projectRulesPath = path.join(project, ".omp", "RULES.md"); + const userRuleText = "User sticky rule: keep the personal safety checklist active.\n"; + const projectRuleText = "Project sticky rule: require repo-local release notes.\n"; + writeFile(userRulesPath, userRuleText); + writeFile(projectRulesPath, projectRuleText); + + const rules = await loadRulesCapability(project); + + const stickyRules = rules.filter(rule => rule.path === userRulesPath || rule.path === projectRulesPath); + expect(stickyRules).toHaveLength(2); + + const userRule = stickyRules.find(rule => rule._source.level === "user"); + const projectRule = stickyRules.find(rule => rule._source.level === "project"); + + if (!userRule) throw new Error("user sticky rule missing"); + expect(userRule.name).toBe("RULES"); + expect(userRule.path).toBe(userRulesPath); + expect(userRule._source.path).toBe(userRulesPath); + expect(userRule.alwaysApply).toBe(true); + expect(userRule.content).toContain(userRuleText.trim()); + expect("_shadowed" in userRule).toBe(false); + + if (!projectRule) throw new Error("project sticky rule missing"); + expect(projectRule.name).toBe("RULES@project"); + expect(projectRule.path).toBe(projectRulesPath); + expect(projectRule._source.path).toBe(projectRulesPath); + expect(projectRule.alwaysApply).toBe(true); + expect(projectRule.content).toContain(projectRuleText.trim()); + expect("_shadowed" in projectRule).toBe(false); + + expect(userRule.name).not.toBe(projectRule.name); + expect(userRule.content).not.toBe(projectRule.content); +}); + test("alwaysApply is forced even when frontmatter says false", async () => { writeFile(path.join(home, ".omp", "agent", "RULES.md"), "---\nalwaysApply: false\n---\nStick around anyway.\n"); diff --git a/packages/coding-agent/test/discovery/claude-plugins.test.ts b/packages/coding-agent/test/discovery/claude-plugins.test.ts index ba39fe344..371c2e039 100644 --- a/packages/coding-agent/test/discovery/claude-plugins.test.ts +++ b/packages/coding-agent/test/discovery/claude-plugins.test.ts @@ -645,6 +645,354 @@ describe("listClaudePluginRoots", () => { expect(found).toBeUndefined(); }); + + test("reads slash commands from array-form commands manifest field (Claude plugin path-behavior rules)", async () => { + // Mirrors real-world plugins such as addyosmani/agent-skills whose plugin.json + // declares `"commands": ["./.claude/commands", "./commands"]`. Both directories + // contribute; each command lands under the plugin's namespace. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-array"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude", "commands"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-array@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./.claude/commands", "./commands"] }), + ); + await fs.writeFile(path.join(pluginPath, ".claude", "commands", "spec.md"), "Spec\n"); + await fs.writeFile(path.join(pluginPath, ".claude", "commands", "plan.md"), "Plan\n"); + await fs.writeFile(path.join(pluginPath, "commands", "review.md"), "Review\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + const names = result.all + .filter(command => command.name.startsWith("manifest-commands-array:")) + .map(command => command.name) + .sort(); + expect(names).toEqual([ + "manifest-commands-array:plan", + "manifest-commands-array:review", + "manifest-commands-array:spec", + ]); + }); + + test("reads slash commands from array-form manifest file entries", async () => { + // Claude plugins reference allows command paths to be either flat `.md` + // files or directories. A manifest-declared commands field still replaces + // default `commands/`; plugins that want defaults must list `./commands`. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-files"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "custom"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "ops"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-files@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./custom/deploy.md", "./ops"] }), + ); + await fs.writeFile(path.join(pluginPath, "custom", "deploy.md"), "Deploy\n"); + await fs.writeFile(path.join(pluginPath, "ops", "rollback.md"), "Rollback\n"); + await fs.writeFile(path.join(pluginPath, "commands", "default.md"), "Default\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(c => c.name === "manifest-commands-files:deploy")?.content).toBe("Deploy\n"); + expect(result.all.find(c => c.name === "manifest-commands-files:rollback")?.content).toBe("Rollback\n"); + expect(result.all.find(c => c.name === "manifest-commands-files:default")).toBeUndefined(); + }); + + test("array-form commands warns on out-of-root entries while loading valid ones", async () => { + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-mixed"); + const outsideDir = path.join(tempDir, "outside-commands"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude", "commands"), { recursive: true }); + await fs.mkdir(outsideDir, { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-mixed@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./.claude/commands", "../../outside-commands"] }), + ); + await fs.writeFile(path.join(pluginPath, ".claude", "commands", "spec.md"), "Spec\n"); + await fs.writeFile(path.join(outsideDir, "escape.md"), "Escape\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings.some(w => w.includes("Ignoring commands path outside plugin root"))).toBe(true); + expect(result.all.find(c => c.name === "manifest-commands-mixed:spec")).toBeDefined(); + expect(result.all.find(c => c.name === "manifest-commands-mixed:escape")).toBeUndefined(); + }); + + test("reads skills from array-form skills manifest field", async () => { + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-array"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "extra-skills", "alpha"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "more-skills", "beta"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-array@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./extra-skills", "./more-skills"] }), + ); + await fs.writeFile( + path.join(pluginPath, "extra-skills", "alpha", "SKILL.md"), + "---\nname: alpha\ndescription: Alpha skill\n---\nBody\n", + ); + await fs.writeFile( + path.join(pluginPath, "more-skills", "beta", "SKILL.md"), + "---\nname: beta\ndescription: Beta skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "alpha")).toBeDefined(); + expect(result.all.find(s => s.name === "beta")).toBeDefined(); + }); + + test("manifest skills field merges with default skills/ directory (adds, not replaces)", async () => { + // Per Claude plugins reference "Path behavior rules": + // `skills` adds to the default `skills/` scan; the default is always loaded + // alongside any manifest-declared directories. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-merge"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "skills", "default-skill"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "extra-skills", "extra-skill"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-merge@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./extra-skills"] }), + ); + await fs.writeFile( + path.join(pluginPath, "skills", "default-skill", "SKILL.md"), + "---\nname: default-skill\ndescription: Default skill\n---\nBody\n", + ); + await fs.writeFile( + path.join(pluginPath, "extra-skills", "extra-skill", "SKILL.md"), + "---\nname: extra-skill\ndescription: Extra skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "default-skill")).toBeDefined(); + expect(result.all.find(s => s.name === "extra-skill")).toBeDefined(); + }); + + test("marketplace-root skills manifest field replaces default skills directory", async () => { + // Claude path-behavior rules carve out marketplace entries whose source is the + // marketplace root: their manifest `skills` field selects the published + // subdirectories instead of also loading the root `skills/` directory. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-marketplace-root"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "skills", "unpublished-root-skill"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "plugins", "published", "skills", "published-skill"), { + recursive: true, + }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-marketplace-root@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, "marketplace.json"), + JSON.stringify({ + name: "market", + owner: { name: "Market" }, + plugins: [{ name: "manifest-skills-marketplace-root", source: "./" }], + }), + ); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./plugins/published/skills"] }), + ); + await fs.writeFile( + path.join(pluginPath, "skills", "unpublished-root-skill", "SKILL.md"), + "---\nname: unpublished-root-skill\ndescription: Unpublished root skill\n---\nBody\n", + ); + await fs.writeFile( + path.join(pluginPath, "plugins", "published", "skills", "published-skill", "SKILL.md"), + "---\nname: published-skill\ndescription: Published skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "published-skill")).toBeDefined(); + expect(result.all.find(s => s.name === "unpublished-root-skill")).toBeUndefined(); + }); + + test("array-form skills entry pointing at a directory containing SKILL.md loads the single skill", async () => { + // Per Claude plugins reference: a skills path may point directly at a directory whose + // SKILL.md is the skill (frontmatter name → invocation, directory basename → fallback). + // Real plugins use `"skills": ["./"]` — that entry must not silently drop the skill. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-skills-self"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "single"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-skills-self@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ skills: ["./single"] }), + ); + await fs.writeFile( + path.join(pluginPath, "single", "SKILL.md"), + "---\nname: solo-skill\ndescription: Solo skill\n---\nBody\n", + ); + + const result = await loadCapability("skills", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(s => s.name === "solo-skill")).toBeDefined(); + }); + + test("manifest commands field replaces default commands/ directory (Claude replace semantics)", async () => { + // Per Claude plugins reference "Path behavior rules": + // `commands` REPLACES the default `commands/` scan when the manifest key is set. + // A plugin that wants both must list `./commands` explicitly. + const pluginsDir = path.join(tempDir, ".claude", "plugins"); + const pluginPath = path.join(tempDir, "plugins", "manifest-commands-replace"); + await fs.mkdir(pluginsDir, { recursive: true }); + await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true }); + await fs.mkdir(path.join(pluginPath, "admin-commands"), { recursive: true }); + + const registry = { + version: 2, + plugins: { + "manifest-commands-replace@market": [ + { + scope: "user", + installPath: pluginPath, + version: "1.0.0", + installedAt: "2025-01-01T00:00:00Z", + lastUpdated: "2025-01-01T00:00:00Z", + }, + ], + }, + }; + await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry)); + await fs.writeFile( + path.join(pluginPath, ".claude-plugin", "plugin.json"), + JSON.stringify({ commands: ["./admin-commands"] }), + ); + // This file lives under the default commands/ dir and MUST NOT load once the + // manifest declares `commands` (Claude's documented "replaces default" semantic). + await fs.writeFile(path.join(pluginPath, "commands", "default.md"), "Default\n"); + await fs.writeFile(path.join(pluginPath, "admin-commands", "admin.md"), "Admin\n"); + + const result = await loadCapability("slash-commands", { cwd: tempDir }); + expect(result.warnings).toEqual([]); + expect(result.all.find(c => c.name === "manifest-commands-replace:admin")).toBeDefined(); + expect(result.all.find(c => c.name === "manifest-commands-replace:default")).toBeUndefined(); + }); }); describe("discoverAgents plugin precedence", () => { diff --git a/packages/coding-agent/test/eval/worker-core.test.ts b/packages/coding-agent/test/eval/worker-core.test.ts new file mode 100644 index 000000000..ab1d71846 --- /dev/null +++ b/packages/coding-agent/test/eval/worker-core.test.ts @@ -0,0 +1,114 @@ +import { describe, expect, it } from "bun:test"; +import { WorkerCore } from "@oh-my-pi/pi-coding-agent/eval/js/worker-core"; +import type { + SessionSnapshot, + Transport, + WorkerInbound, + WorkerOutbound, +} from "@oh-my-pi/pi-coding-agent/eval/js/worker-protocol"; + +interface WorkerHarness { + send(message: WorkerInbound): void; + onMessage(handler: (message: WorkerOutbound) => void): () => void; +} + +function createWorkerHarness(): WorkerHarness { + const hostListeners = new Set<(message: WorkerOutbound) => void>(); + const workerListeners = new Set<(message: WorkerInbound) => void>(); + const transport: Transport = { + send: message => { + queueMicrotask(() => { + for (const listener of hostListeners) listener(message); + }); + }, + onMessage: handler => { + workerListeners.add(handler); + return () => workerListeners.delete(handler); + }, + close: () => {}, + }; + new WorkerCore(transport); + return { + send(message) { + queueMicrotask(() => { + for (const listener of workerListeners) listener(message); + }); + }, + onMessage(handler) { + hostListeners.add(handler); + return () => hostListeners.delete(handler); + }, + }; +} + +function waitForMessage( + harness: WorkerHarness, + predicate: (message: WorkerOutbound) => boolean, +): Promise { + const { promise, resolve } = Promise.withResolvers(); + let unsubscribe = (): void => {}; + unsubscribe = harness.onMessage(message => { + if (!predicate(message)) return; + unsubscribe(); + resolve(message); + }); + return promise; +} + +async function initializeWorker(harness: WorkerHarness, snapshot: SessionSnapshot): Promise { + const ready = waitForMessage(harness, message => message.type === "ready"); + harness.send({ type: "init", snapshot }); + expect((await ready).type).toBe("ready"); +} + +describe("WorkerCore", () => { + it("reports same-realm cwd conflicts through the worker protocol", async () => { + const first = createWorkerHarness(); + const second = createWorkerHarness(); + const cwd = process.cwd(); + await initializeWorker(first, { cwd, sessionId: "same-realm-first", localRoots: {} }); + await initializeWorker(second, { cwd, sessionId: "same-realm-second", localRoots: {} }); + + const gate = Promise.withResolvers(); + const entered = Promise.withResolvers(); + (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }).__omp_worker_core_gate = { + entered: () => entered.resolve(), + wait: gate.promise, + }; + try { + first.send({ + type: "run", + runId: "hold-first-runtime", + code: "globalThis.__omp_worker_core_gate.entered(); await globalThis.__omp_worker_core_gate.wait;", + filename: "[same-realm-first].js", + snapshot: { cwd, sessionId: "same-realm-first", localRoots: {} }, + }); + await entered.promise; + + const result = waitForMessage( + second, + message => message.type === "result" && message.runId === "overlap-second-runtime", + ); + second.send({ + type: "run", + runId: "overlap-second-runtime", + code: "1 + 1;", + filename: "[same-realm-second].js", + snapshot: { cwd, sessionId: "same-realm-second", localRoots: {} }, + }); + + expect(await result).toMatchObject({ + type: "result", + runId: "overlap-second-runtime", + ok: false, + error: { message: "Cannot set cwd while another same-realm JS runtime is running" }, + }); + } finally { + gate.resolve(); + delete (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise } }) + .__omp_worker_core_gate; + first.send({ type: "close" }); + second.send({ type: "close" }); + } + }); +}); diff --git a/packages/coding-agent/test/extensibility/legacy-pi-inplace-load.test.ts b/packages/coding-agent/test/extensibility/legacy-pi-inplace-load.test.ts index 03d2baaa0..baa6a8181 100644 --- a/packages/coding-agent/test/extensibility/legacy-pi-inplace-load.test.ts +++ b/packages/coding-agent/test/extensibility/legacy-pi-inplace-load.test.ts @@ -308,6 +308,22 @@ describe("legacy-pi in-place module loading (issue #1674)", () => { expect(mod.css).toBe(".x{color:red}"); }); + it("leaves JSON import-attribute targets on Bun's native loader", async () => { + const dir = await writePackage({ + "package.json": JSON.stringify({ name: "json-import-ext", version: "1.0.0" }), + "prices.json": JSON.stringify({ input: 0.15 }), + "index.ts": [ + 'import prices from "./prices.json" with { type: "json" };', + "export const inputPrice = prices.input;", + "export default function (pi) { void pi; }", + ].join("\n"), + }); + + const mod = (await loadLegacyPiModule(path.join(dir, "index.ts"))) as { inputPrice: number }; + + expect(mod.inputPrice).toBe(0.15); + }); + it("loads the extension's own node_modules deps natively while remapping legacy pi imports", async () => { const dir = await writePackage({ "package.json": JSON.stringify({ name: "dep-ext", version: "1.0.0" }), diff --git a/packages/coding-agent/test/input-controller-keybindings.test.ts b/packages/coding-agent/test/input-controller-keybindings.test.ts index 772dbf9cd..e0f1e9382 100644 --- a/packages/coding-agent/test/input-controller-keybindings.test.ts +++ b/packages/coding-agent/test/input-controller-keybindings.test.ts @@ -251,6 +251,23 @@ describe("InputController keybinding setup", () => { expect(spies.resetDisplay).toHaveBeenCalledTimes(1); }); + it("does not mark pasted shell prompts as Python mode while editing", async () => { + const { InputController, ctx, editor } = await createContext(); + const controller = new InputController(ctx); + + controller.setupKeyHandlers(); + + editor.onChange?.("$ cd ~/project && sudo ./build-and-push.sh o5.7 2>&1 | tail -4"); + + expect(ctx.isPythonMode).toBe(false); + expect(ctx.updateEditorBorderColor).not.toHaveBeenCalled(); + + editor.onChange?.("$ print(1)"); + + expect(ctx.isPythonMode).toBe(true); + expect(ctx.updateEditorBorderColor).toHaveBeenCalledTimes(1); + }); + it("registers retry as an editor action and retries the failed turn", async () => { const { InputController, ctx, editor, spies } = await createContext(); const controller = new InputController(ctx); diff --git a/packages/coding-agent/test/input-controller-python-prefix.test.ts b/packages/coding-agent/test/input-controller-python-prefix.test.ts index 92cd133ca..6e48d1af4 100644 --- a/packages/coding-agent/test/input-controller-python-prefix.test.ts +++ b/packages/coding-agent/test/input-controller-python-prefix.test.ts @@ -109,6 +109,36 @@ describe("InputController Python prompt prefix", () => { ]); }); + it("submits pasted shell-prompt transcripts with OMP chrome as a normal prompt", async () => { + const transcript = + "$ cd ~/project && sudo ./build-and-push.sh o5.7 2>&1 | tail -4\n" + + " |\n" + + " in: 282 out: 152 cache 344K t: 3.3s tok/s: 351.9/s\n" + + " is this command stuck in limbo"; + const { ctx, editor, handlePythonCommand, onInputCallback, startPendingSubmission, submitted } = createContext(); + const controller = new InputController(ctx); + controller.setupEditorSubmitHandler(); + + await editor.onSubmit?.(transcript); + + expect(handlePythonCommand).not.toHaveBeenCalled(); + expect(startPendingSubmission).toHaveBeenCalledWith({ + text: transcript, + images: undefined, + imageLinks: undefined, + streamingBehavior: "steer", + }); + expect(onInputCallback).toHaveBeenCalledTimes(1); + expect(submitted).toEqual([ + { + text: transcript, + images: undefined, + imageLinks: undefined, + streamingBehavior: "steer", + }, + ]); + }); + it("keeps space-separated Python shortcuts available", async () => { const { ctx, editor, handlePythonCommand, onInputCallback } = createContext(); const controller = new InputController(ctx); diff --git a/packages/coding-agent/test/interactive-mode-plan-review.test.ts b/packages/coding-agent/test/interactive-mode-plan-review.test.ts index 7efce822d..6ebf4a59a 100644 --- a/packages/coding-agent/test/interactive-mode-plan-review.test.ts +++ b/packages/coding-agent/test/interactive-mode-plan-review.test.ts @@ -415,6 +415,61 @@ describe("InteractiveMode plan review rendering", () => { expect(await Bun.file(resolvedPlanPath).text()).toContain("edited body"); }); + it("carries pre-approval local artifacts into the fresh approve-and-execute session", async () => { + const planFilePath = "local://handoff-plan.md"; + const localOptions = { + getArtifactsDir: () => session.sessionManager.getArtifactsDir(), + getSessionId: () => session.sessionManager.getSessionId(), + }; + const oldLocalRoot = resolveLocalUrlToPath("local://", localOptions); + const oldPlanPath = resolveLocalUrlToPath(planFilePath, localOptions); + const oldArtifactPath = resolveLocalUrlToPath("local://handoff/nested/context.txt", localOptions); + await fs.mkdir(path.dirname(oldArtifactPath), { recursive: true }); + await Bun.write(oldArtifactPath, "pre-approval handoff"); + await Bun.write(oldPlanPath, "# Plan\n\noriginal body\n"); + + mode.planModeEnabled = true; + mode.planModePlanFilePath = planFilePath; + const planContent = "# Plan\n\nfinal approved body\n"; + vi.spyOn(mode, "showPlanReview").mockImplementation(async (_plan, _title, _options, dialogOptions) => { + dialogOptions?.onPlanEdited?.(planContent); + return "Approve and execute"; + }); + vi.spyOn(mode, "handleClearCommand").mockImplementation(async () => { + await session.sessionManager.newSession(); + }); + let artifactAtPrompt = ""; + let planAtPrompt = ""; + const prompt = vi.spyOn(session, "prompt").mockImplementation(async () => { + const promptArtifactPath = resolveLocalUrlToPath("local://handoff/nested/context.txt", localOptions); + const promptPlanPath = resolveLocalUrlToPath(planFilePath, localOptions); + artifactAtPrompt = (await Bun.file(promptArtifactPath).exists()) + ? await Bun.file(promptArtifactPath).text() + : ""; + planAtPrompt = (await Bun.file(promptPlanPath).exists()) ? await Bun.file(promptPlanPath).text() : ""; + return undefined as never; + }); + + expect(await Bun.file(oldArtifactPath).text()).toBe("pre-approval handoff"); + + await mode.handlePlanApproval({ + planFilePath, + planExists: true, + title: "HANDOFF", + }); + + const newLocalRoot = resolveLocalUrlToPath("local://", localOptions); + const newArtifactPath = resolveLocalUrlToPath("local://handoff/nested/context.txt", localOptions); + const newPlanPath = resolveLocalUrlToPath(planFilePath, localOptions); + expect(newLocalRoot).not.toBe(oldLocalRoot); + expect(await Bun.file(newArtifactPath).text()).toBe("pre-approval handoff"); + expect(await Bun.file(newPlanPath).text()).toBe(planContent); + expect(artifactAtPrompt).toBe("pre-approval handoff"); + expect(planAtPrompt).toBe(planContent); + expect(await Bun.file(oldArtifactPath).text()).toBe("pre-approval handoff"); + expect(prompt).toHaveBeenCalledWith(expect.any(String), { synthetic: true }); + }); + it("offers approve-and-keep-context as a distinct plan approval path", async () => { const planFilePath = "local://PLAN.md"; const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, { diff --git a/packages/coding-agent/test/marketplace/fetcher.test.ts b/packages/coding-agent/test/marketplace/fetcher.test.ts index e72c84f1e..5e73e0517 100644 --- a/packages/coding-agent/test/marketplace/fetcher.test.ts +++ b/packages/coding-agent/test/marketplace/fetcher.test.ts @@ -1,4 +1,4 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it, spyOn } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; @@ -7,6 +7,7 @@ import { fetchMarketplace, parseMarketplaceCatalog, } from "@oh-my-pi/pi-coding-agent/extensibility/plugins/marketplace"; +import * as git from "@oh-my-pi/pi-coding-agent/utils/git"; import { removeSyncWithRetries } from "@oh-my-pi/pi-utils"; // Fixture lives at test/marketplace/fixtures/valid-marketplace/ @@ -231,6 +232,24 @@ describe("fetchMarketplace", () => { ); }); + it("hides temp clone paths in cloned catalog validation errors", async () => { + const cloneSpy = spyOn(git, "clone").mockImplementation(async (_url, targetDir) => { + fs.mkdirSync(path.join(targetDir, ".claude-plugin"), { recursive: true }); + fs.writeFileSync( + path.join(targetDir, ".claude-plugin", "marketplace.json"), + JSON.stringify({ name: "broken-marketplace", plugins: [] }), + ); + }); + + try { + await expect(fetchMarketplace("kubeshark/kubeshark", tmpDir)).rejects.toThrow( + 'Cloned repository https://github.com/kubeshark/kubeshark.git: Missing or invalid field "owner" in catalog: .claude-plugin/marketplace.json (source: kubeshark/kubeshark)', + ); + } finally { + cloneSpy.mockRestore(); + } + }); + // Network-dependent tests — skip in CI / offline environments. // These verify real git clone and HTTP fetch error handling. it.skip("github source throws on nonexistent repo", async () => { diff --git a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts index 3e74dc631..50db05d61 100644 --- a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts +++ b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts @@ -14,6 +14,35 @@ function normalizeRenderedText(text: string): string { return stripVTControlCharacters(text).replace(/\s+/g, " ").trim(); } +const DEFAULT_RETRY_FALLBACK_ACTION_LABEL = "Set as DEFAULT retry fallback"; +const DEFAULT_RETRY_FALLBACK_ACTION = "retryFallback"; + +type ModelSelectorAction = "modelRole" | typeof DEFAULT_RETRY_FALLBACK_ACTION; +type TestRoleSelectArgs = [ + model: Model, + role: string | null, + thinkingLevel?: ConfiguredThinkingLevel, + selector?: string, + action?: ModelSelectorAction, +]; +type TestRoleSelectCallback = (...args: TestRoleSelectArgs) => void; + +function isSelectedMenuLine(line: string): boolean { + const trimmed = line.trimStart(); + return trimmed.startsWith("❯") || trimmed.startsWith("▸") || trimmed.startsWith(">") || trimmed.startsWith("\uf054"); +} + +function selectMenuAction(selector: ModelSelectorComponent, label: string): void { + for (let attempt = 0; attempt < 20; attempt++) { + const selectedTarget = stripVTControlCharacters(selector.render(220).join("\n")) + .split("\n") + .find(line => line.includes(label) && isSelectedMenuLine(line)); + if (selectedTarget) return; + selector.handleInput("\x1b[B"); + } + throw new Error(`Menu action not selectable: ${label}`); +} + function createSelector(model: Model, settings: Settings): ModelSelectorComponent { const modelRegistry = { getAll: () => [model], @@ -66,7 +95,7 @@ function createContextTestModel(id: string, contextWindow: number): Model { function createScopedSelector( models: Model[], settings: Settings, - onSelect: (model: Model, role: string | null, thinkingLevel?: ConfiguredThinkingLevel, selector?: string) => void, + onSelect: TestRoleSelectCallback, options?: { temporaryOnly?: boolean; currentContextTokens?: number }, ): ModelSelectorComponent { const modelRegistry = { @@ -82,7 +111,13 @@ function createScopedSelector( settings, modelRegistry, models.map(model => ({ model })), - (model, role, thinkingLevel, selector) => onSelect(model, role, thinkingLevel, selector), + ( + model: Model, + role: string | null, + thinkingLevel?: ConfiguredThinkingLevel, + selector?: string, + action?: ModelSelectorAction, + ) => onSelect(model, role, thinkingLevel, selector, action), () => {}, options, ); @@ -299,6 +334,32 @@ describe("ModelSelector role badge thinking display", () => { expect(onSelect.mock.calls[0]?.[3]).toBe("test/only-small"); }); + test("assigns selected model as default retry fallback without opening thinking options", () => { + installTestTheme(); + const settings = Settings.isolated({}); + const fallback = createContextTestModel("retry-fallback-model", 128_000); + const onSelect = vi.fn(); + const selector = createScopedSelector([fallback], settings, onSelect); + installTestTheme(); + + selector.handleInput("\n"); + const menuRendered = normalizeRenderedText(selector.render(220).join("\n")); + expect(menuRendered).toContain("Action for: retry-fallback-model"); + expect(menuRendered).toContain(DEFAULT_RETRY_FALLBACK_ACTION_LABEL); + + selectMenuAction(selector, DEFAULT_RETRY_FALLBACK_ACTION_LABEL); + selector.handleInput("\n"); + + const afterEnter = normalizeRenderedText(selector.render(220).join("\n")); + expect(afterEnter).not.toContain("Thinking for:"); + expect(onSelect).toHaveBeenCalledTimes(1); + const call = onSelect.mock.calls[0]; + expect(call?.[0]).toBe(fallback); + expect(call?.[1]).toBe("default"); + expect(call?.[3]).toBe("test/retry-fallback-model"); + expect(call?.[4]).toBe(DEFAULT_RETRY_FALLBACK_ACTION); + }); + test("uses cached models for Enter while offline refresh is still pending", () => { installTestTheme(); const settings = Settings.isolated({}); diff --git a/packages/coding-agent/test/modes/components/settings-layout.test.ts b/packages/coding-agent/test/modes/components/settings-layout.test.ts index 72c357be3..e7c7a1c2f 100644 --- a/packages/coding-agent/test/modes/components/settings-layout.test.ts +++ b/packages/coding-agent/test/modes/components/settings-layout.test.ts @@ -107,4 +107,22 @@ describe("settings layout", () => { group: "Services", }); }); + + it("exposes retry fallback chains as editable JSON in the model settings", () => { + const def = getSettingsForTab("model").find(item => item.path === "retry.fallbackChains"); + + expect(def).toMatchObject({ + path: "retry.fallbackChains", + type: "text", + tab: "model", + group: "Retry & Fallback", + label: "Retry Fallback Chains", + }); + if (!def) throw new Error("retry.fallbackChains setting definition missing"); + + const description = def.description.toLowerCase(); + expect(description).toContain("json"); + expect(description).toContain("fallback"); + expect(description).toContain("selector"); + }); }); diff --git a/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts b/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts index 382d2ec56..ce05a9fb5 100644 --- a/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts +++ b/packages/coding-agent/test/modes/controllers/command-controller-hotkeys.test.ts @@ -41,7 +41,8 @@ describe("buildHotkeysMarkdown", () => { expect(markdown).toContain("| `Ctrl+L` | Reset terminal display |"); expect(markdown).toContain("| `Alt+R` | Retry last failed assistant turn |"); expect(markdown).toContain("| `Alt+Shift+P` | Toggle plan mode |"); - expect(markdown).toContain("| `#` | Open prompt actions |"); + expect(markdown).toContain("| `#` | GitHub issue/PR reference"); + expect(markdown).toContain("| `#` / `#` | Prompt actions"); for (const line of lines) { if (line.length === 0) continue; expect(line.startsWith(" ")).toBe(false); diff --git a/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts b/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts index 7c0bc7a04..019fbacae 100644 --- a/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts +++ b/packages/coding-agent/test/modes/controllers/mcp-authorization-link.test.ts @@ -24,10 +24,12 @@ const LONG_AUTH_URL = const LINEAR_AUTH_URL = `https://mcp.linear.app/authorize?response_type=code&client_id=abcdefghij0123456789ABCDEFGHIJ0123456789&redirect_uri=http%3A%2F%2Flocalhost%3A3000%2Fcallback&scope=read%20write%20mcp%3Aall&state=0123456789abcdef0123456789abcdef&code_challenge=5MlkJfN2GhX9uP0rQ7sT8vB1oCwDeFgHiJkLmNoPqRsTuVwXyZ&code_challenge_method=S256`; /** - * Reassemble the copy-URL rows for `label` into a single string, mirroring what - * a browser would produce when a multi-row selection is pasted into its - * address bar (whitespace stripped between chunks). Returns "" if the label - * row isn't found. + * Reassemble the copy-URL rows for `label` into a single string, mirroring a + * real multi-row terminal selection pasted into an address bar: browsers + * strip the newlines, but any other leading bytes survive (verbatim or + * percent-encoded) — so chunks are concatenated RAW, with no indent-stripping + * that could mask a corrupting prefix. Returns "" if the label row isn't + * found. */ function reassembleUrl(plainLines: string[], label: string): string { const start = plainLines.findIndex(line => line.startsWith(` ${label}`)); @@ -36,15 +38,13 @@ function reassembleUrl(plainLines: string[], label: string): string { // Inline form contains the whole URL on the label row. const inlineMatch = first.match(new RegExp(`^ ${label.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")} (.*)$`)); if (inlineMatch) return inlineMatch[1]!; - // Wrapped form: label alone, then continuation rows with a single-space indent. + // Wrapped form: indented label row, then UNINDENTED continuation chunks. + // Any indented row (the next label) or blank row ends this URL's chunks. let joined = ""; for (let i = start + 1; i < plainLines.length; i++) { const line = plainLines[i]!; - // Continuation rows have exactly one leading space; a different indent - // or label ends this URL's rows. - if (!line.startsWith(" ") || line.startsWith(" ") || line.trim().length === 0) break; - if (line.includes(":") && /^ [A-Z][^:]*:/.test(line)) break; // next label row - joined += line.slice(1); + if (line.startsWith(" ") || line.trim().length === 0) break; + joined += line; } return joined; } @@ -89,10 +89,12 @@ describe("MCPAuthorizationLinkPrompt", () => { expect(visibleWidth(line)).toBeLessThanOrEqual(width); } - // Wrapping puts the label on its own row followed by continuation - // chunks with a single-space indent. + // Wrapping puts the label on its own indented row followed by + // UNINDENTED continuation chunks — zero leading bytes, so a multi-row + // selection pastes back to the exact URL. const labelRow = plainLines.indexOf(` ${COPY_URL_LABEL}`); expect(labelRow).toBeGreaterThanOrEqual(0); + expect(plainLines[labelRow + 1]!.startsWith(" ")).toBe(false); // Chunks reassemble to the URL byte-for-byte — the trailing // `code_challenge_method=S256` MUST be present. @@ -140,7 +142,7 @@ describe("MCPAuthorizationLinkPrompt", () => { it("floors the wrap width so degenerately-narrow viewports still emit every character", () => { // Below 16 cols the terminal is unusable, but the render still emits - // chunks (bounded at 16 - indent = 15 chars). No character is silently + // chunks (bounded at the 16-column floor). No character is silently // dropped; the user can widen and reflow. const lines = new MCPAuthorizationLinkPrompt(LINEAR_AUTH_URL).render(4); const plainLines = lines.map(line => stripVTControlCharacters(line)); diff --git a/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts new file mode 100644 index 000000000..0cd330b42 --- /dev/null +++ b/packages/coding-agent/test/modes/github-ref-autocomplete.test.ts @@ -0,0 +1,156 @@ +import { describe, expect, it } from "bun:test"; +import { KeybindingsManager as AppKeybindingsManager } from "@oh-my-pi/pi-coding-agent/config/keybindings"; +import { getGithubRefContext, getGithubRefSuggestions } from "@oh-my-pi/pi-coding-agent/modes/github-ref-autocomplete"; +import { createPromptActionAutocompleteProvider } from "@oh-my-pi/pi-coding-agent/modes/prompt-action-autocomplete"; + +function makeProvider() { + return createPromptActionAutocompleteProvider({ + commands: [], + basePath: "/tmp", + keybindings: AppKeybindingsManager.inMemory({}), + copyCurrentLine: () => {}, + copyPrompt: () => {}, + undo: () => {}, + moveCursorToMessageEnd: () => {}, + moveCursorToMessageStart: () => {}, + moveCursorToLineStart: () => {}, + moveCursorToLineEnd: () => {}, + }); +} + +describe("github-ref autocomplete — token detection", () => { + it("matches a standalone # ending at the cursor", () => { + expect(getGithubRefContext("#3164")).toEqual({ prefix: "#3164", qualifier: null, number: "3164" }); + expect(getGithubRefContext("look at #3164")).toEqual({ prefix: "#3164", qualifier: null, number: "3164" }); + // only the token ending at the cursor (the $ anchor) wins + expect(getGithubRefContext("see #1 then #3164")).toEqual({ + prefix: "#3164", + qualifier: null, + number: "3164", + }); + }); + + it("requires a token boundary before # so embedded hashes don't match", () => { + // cross-repo reference, mid-word, URL fragment — none should offer candidates + expect(getGithubRefContext("owner/repo#3164")).toBeNull(); + expect(getGithubRefContext("foo#3164")).toBeNull(); + expect(getGithubRefContext("C#12")).toBeNull(); + expect(getGithubRefContext("https://github.com/can1357/oh-my-pi#3164")).toBeNull(); + expect(getGithubRefContext("path/#3164")).toBeNull(); + }); + + it("does not match bare #, text, mixed, zero, or leading zeros", () => { + expect(getGithubRefContext("#")).toBeNull(); + expect(getGithubRefContext("#copy")).toBeNull(); + expect(getGithubRefContext("#3164abc")).toBeNull(); + expect(getGithubRefContext("#3a")).toBeNull(); + // a space after the digits closes the token + expect(getGithubRefContext("#3164 ")).toBeNull(); + expect(getGithubRefContext("#0")).toBeNull(); + expect(getGithubRefContext("#0123")).toBeNull(); + }); + + it("detects a pr/pull/issue qualifier word immediately before the number", () => { + expect(getGithubRefContext("pr #3164")).toEqual({ prefix: "pr #3164", qualifier: "pr", number: "3164" }); + expect(getGithubRefContext("PR #3164")).toEqual({ prefix: "PR #3164", qualifier: "pr", number: "3164" }); + expect(getGithubRefContext("pull #3164")).toEqual({ prefix: "pull #3164", qualifier: "pr", number: "3164" }); + expect(getGithubRefContext("issue #3164")).toEqual({ + prefix: "issue #3164", + qualifier: "issue", + number: "3164", + }); + // the word right before the number is the qualifier, even with leading text + expect(getGithubRefContext("look at the issue #3164")?.qualifier).toBe("issue"); + }); + + it("does not treat arbitrary words, or qualifiers glued to a path, as the type", () => { + expect(getGithubRefContext("review #3164")?.qualifier).toBeNull(); + // "pr" inside "src/pr" is preceded by '/', not a boundary, so it is not a qualifier + expect(getGithubRefContext("src/pr #3164")?.qualifier).toBeNull(); + // a qualifier with no space before the # is not recognized + expect(getGithubRefContext("pr#3164")).toBeNull(); + }); +}); + +describe("github-ref autocomplete — suggestions", () => { + it("offers both candidates when no qualifier is given", () => { + const result = getGithubRefSuggestions("#3164"); + expect(result).not.toBeNull(); + expect(result!.prefix).toBe("#3164"); + expect(result!.items).toEqual([ + { value: "pr://3164", label: "PR #3164", description: "GitHub pull request" }, + { value: "issue://3164", label: "Issue #3164", description: "GitHub issue" }, + ]); + }); + + it("offers only the PR candidate for a pr/pull qualifier", () => { + const result = getGithubRefSuggestions("pr #3164"); + expect(result!.prefix).toBe("pr #3164"); + expect(result!.items).toEqual([{ value: "pr://3164", label: "PR #3164", description: "GitHub pull request" }]); + }); + + it("offers only the Issue candidate for an issue qualifier", () => { + const result = getGithubRefSuggestions("issue #3164"); + expect(result!.items).toEqual([{ value: "issue://3164", label: "Issue #3164", description: "GitHub issue" }]); + }); + + it("returns null for embedded or non-ref text", () => { + expect(getGithubRefSuggestions("owner/repo#3164")).toBeNull(); + expect(getGithubRefSuggestions("#copy")).toBeNull(); + expect(getGithubRefSuggestions("#0")).toBeNull(); + }); +}); + +describe("github-ref autocomplete — provider integration", () => { + it("yields both candidates and rewrites the token to the chosen internal URL", async () => { + const provider = makeProvider(); + const suggestions = await provider.getSuggestions(["review #3164"], 0, 12); + expect(suggestions).not.toBeNull(); + expect(suggestions!.prefix).toBe("#3164"); + expect(suggestions!.items.map(item => item.value)).toEqual(["pr://3164", "issue://3164"]); + + const pr = suggestions!.items[0]!; + const issue = suggestions!.items[1]!; + const prResult = provider.applyCompletion(["review #3164"], 0, 12, pr, suggestions!.prefix); + expect(prResult.lines).toEqual(["review pr://3164 "]); + expect(prResult.cursorCol).toBe("review pr://3164 ".length); + + const issueResult = provider.applyCompletion(["review #3164"], 0, 12, issue, suggestions!.prefix); + expect(issueResult.lines).toEqual(["review issue://3164 "]); + }); + + it("constrains to the named type and consumes the qualifier on accept", async () => { + const provider = makeProvider(); + const suggestions = await provider.getSuggestions(["review pr #3164"], 0, 15); + expect(suggestions).not.toBeNull(); + expect(suggestions!.prefix).toBe("pr #3164"); + expect(suggestions!.items.map(item => item.value)).toEqual(["pr://3164"]); + + const pr = suggestions!.items[0]!; + const result = provider.applyCompletion(["review pr #3164"], 0, 15, pr, suggestions!.prefix); + // the "pr " qualifier is replaced along with the number, not left dangling + expect(result.lines).toEqual(["review pr://3164 "]); + }); + + it("revalidates stale prefixes against the live cursor token before applying", async () => { + const provider = makeProvider(); + const staleSuggestions = await provider.getSuggestions(["review #316"], 0, 11); + expect(staleSuggestions).not.toBeNull(); + const stalePr = staleSuggestions!.items[0]!; + + const updatedNumber = provider.applyCompletion(["review #3164"], 0, 12, stalePr, staleSuggestions!.prefix); + expect(updatedNumber.lines).toEqual(["review pr://3164 "]); + expect(updatedNumber.cursorCol).toBe("review pr://3164 ".length); + + const embeddedHash = provider.applyCompletion(["owner/repo#3164"], 0, 15, stalePr, staleSuggestions!.prefix); + expect(embeddedHash.lines).toEqual(["owner/repo#3164"]); + expect(embeddedHash.cursorCol).toBe(15); + }); + + it("does not offer candidates for embedded hashes (falls through to other providers)", async () => { + const provider = makeProvider(); + const isRef = (value: string) => value.startsWith("pr://") || value.startsWith("issue://"); + const embedded = await provider.getSuggestions(["owner/repo#3164"], 0, 15); + expect(embedded?.items.every(item => !isRef(item.value)) ?? true).toBe(true); + }); +}); diff --git a/packages/coding-agent/test/modes/workflow.test.ts b/packages/coding-agent/test/modes/workflow.test.ts index 3d23c40b3..61a5ae838 100644 --- a/packages/coding-agent/test/modes/workflow.test.ts +++ b/packages/coding-agent/test/modes/workflow.test.ts @@ -1,6 +1,11 @@ import { beforeAll, describe, expect, it } from "bun:test"; import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { containsWorkflow, highlightWorkflow, WORKFLOW_NOTICE } from "@oh-my-pi/pi-coding-agent/modes/workflow"; +import { + containsWorkflow, + highlightWorkflow, + renderWorkflowNotice, + WORKFLOW_NOTICE, +} from "@oh-my-pi/pi-coding-agent/modes/workflow"; beforeAll(() => { // highlightWorkflow reads the global theme's color mode. @@ -48,9 +53,18 @@ describe("workflow keyword highlighting", () => { }); describe("workflow notice", () => { - it("is a non-empty system notice carrying the eval-fan-out contract", () => { + it("is a non-empty system notice carrying the task fan-out contract", () => { expect(WORKFLOW_NOTICE.length).toBeGreaterThan(0); expect(WORKFLOW_NOTICE).toContain("**workflowz** keyword"); - expect(WORKFLOW_NOTICE).toContain("parallel("); + expect(WORKFLOW_NOTICE).toContain("Use the `task` tool for batched fan-out"); + expect(WORKFLOW_NOTICE).toContain("tasks[]"); + }); + + it("renders flat task-call guidance when task.batch is disabled", () => { + const notice = renderWorkflowNotice({ taskBatch: false }); + expect(notice).toContain("once per independent subagent"); + expect(notice).toContain("Do not pass `context` or `tasks[]`"); + expect(notice).toContain("one independent task call per leaf"); + expect(notice).not.toContain("Call `task` once per independent fan-out batch"); }); }); diff --git a/packages/coding-agent/test/non-interactive-env.test.ts b/packages/coding-agent/test/non-interactive-env.test.ts index 57c4ca878..669a0b4db 100644 --- a/packages/coding-agent/test/non-interactive-env.test.ts +++ b/packages/coding-agent/test/non-interactive-env.test.ts @@ -1,4 +1,7 @@ import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; import { buildNonInteractiveEnv } from "@oh-my-pi/pi-coding-agent/exec/non-interactive-env"; describe("buildNonInteractiveEnv", () => { @@ -45,3 +48,54 @@ describe("buildNonInteractiveEnv", () => { expect(env.LC_ALL).toBeUndefined(); }); }); + +it("keeps launch .env.local values out of child shell config", async () => { + const tmp = await fs.mkdtemp(path.join(os.tmpdir(), "omp-env-local-")); + try { + await Bun.write( + path.join(tmp, ".env.local"), + "CONVEX_DEPLOYMENT=anonymous:root-local\nCONVEX_URL=http://127.0.0.1:3210\n", + ); + const procmgrPath = path.resolve(import.meta.dir, "../../utils/src/procmgr.ts"); + const script = [ + `import { getShellConfig } from ${JSON.stringify(procmgrPath)};`, + "const env = getShellConfig().env;", + "console.log(JSON.stringify({", + " deployment: env.CONVEX_DEPLOYMENT ?? null,", + " url: env.CONVEX_URL ?? null,", + " inherited: env.OMP_TEST_INHERITED_MARKER ?? null,", + "}));", + ].join("\n"); + const proc = Bun.spawn([process.execPath, "--no-install", "--eval", script], { + cwd: tmp, + env: { + HOME: process.env.HOME ?? "", + OMP_TEST_INHERITED_MARKER: "keep-me", + PATH: process.env.PATH ?? "", + SHELL: process.env.SHELL ?? "/bin/bash", + }, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(proc.stdout).text(), + new Response(proc.stderr).text(), + proc.exited, + ]); + + expect(stderr).toBe(""); + expect(exitCode).toBe(0); + const payload: { + deployment: string | null; + url: string | null; + inherited: string | null; + } = JSON.parse(stdout); + expect(payload).toEqual({ + deployment: null, + url: null, + inherited: "keep-me", + }); + } finally { + await fs.rm(tmp, { recursive: true, force: true }); + } +}); diff --git a/packages/coding-agent/test/selector-settings-side-effects.test.ts b/packages/coding-agent/test/selector-settings-side-effects.test.ts index 3e54ea139..8a16e03d8 100644 --- a/packages/coding-agent/test/selector-settings-side-effects.test.ts +++ b/packages/coding-agent/test/selector-settings-side-effects.test.ts @@ -1,6 +1,9 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; +import { stripVTControlCharacters } from "node:util"; +import { buildModel } from "@oh-my-pi/pi-catalog/build"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller"; +import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { beginSettingsTest, restoreSettingsTestState, type SettingsTestState } from "./helpers/settings-test-state"; let settingsState: SettingsTestState | undefined; @@ -51,4 +54,74 @@ describe("selector setting side effects", () => { expect(invalidate).toHaveBeenCalledTimes(1); expect(requestRender).toHaveBeenCalledTimes(1); }); + + it("replaces malformed default retry fallback chains from the model selector action", async () => { + const testTheme = await getThemeByName("dark"); + if (!testTheme) throw new Error("Failed to load dark theme for model selector test"); + setThemeInstance(testTheme); + + const settings = Settings.isolated({}); + settings.set("retry.fallbackChains", { default: "not-an-array" } as unknown as Record); + const fallback = buildModel({ + id: "retry-fallback-model", + name: "retry-fallback-model", + api: "ollama-chat", + baseUrl: "https://example.com", + reasoning: false, + provider: "test", + input: ["text"], + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, + contextWindow: 128_000, + maxTokens: 1024, + }); + const showStatus = vi.fn(); + const showError = vi.fn(); + const controller = new SelectorController({ + ui: { requestRender: vi.fn(), setFocus: vi.fn() }, + editorContainer: { clear: vi.fn(), addChild: vi.fn() }, + editor: {}, + settings, + session: { + model: undefined, + modelRegistry: { + getAll: () => [fallback], + getDiscoverableProviders: () => [], + }, + scopedModels: [{ model: fallback }], + getContextUsage: () => undefined, + }, + statusLine: { invalidate: vi.fn() }, + updateEditorBorderColor: vi.fn(), + keybindings: { getKeys: () => [] }, + showStatus, + showError, + } as unknown as ConstructorParameters[0]); + let selector: { handleInput(input: string): void; render(width: number): string[] } | undefined; + controller.showSelector = create => { + const result = create(() => {}); + selector = result.component as typeof selector; + }; + + controller.showModelSelector(); + if (!selector) throw new Error("Expected model selector to be shown"); + selector.handleInput("\n"); + for (let attempt = 0; attempt < 20; attempt++) { + const selectedLine = stripVTControlCharacters(selector.render(220).join("\n")) + .split("\n") + .find(line => { + if (!line.includes("Set as DEFAULT retry fallback")) return false; + const trimmed = line.trimStart(); + return trimmed.startsWith("❯") || trimmed.startsWith("▸") || trimmed.startsWith(">"); + }); + if (selectedLine) break; + selector.handleInput("\x1b[B"); + if (attempt === 19) throw new Error("Default retry fallback action was not selectable"); + } + selector.handleInput("\n"); + await Promise.resolve(); + + expect(showError).not.toHaveBeenCalled(); + expect(settings.get("retry.fallbackChains")).toEqual({ default: ["test/retry-fallback-model"] }); + expect(showStatus).toHaveBeenCalledWith("Default fallback model: test/retry-fallback-model"); + }); }); diff --git a/packages/coding-agent/test/status-line-overflow.test.ts b/packages/coding-agent/test/status-line-overflow.test.ts index fd6307971..b73b9d230 100644 --- a/packages/coding-agent/test/status-line-overflow.test.ts +++ b/packages/coding-agent/test/status-line-overflow.test.ts @@ -77,13 +77,27 @@ function createCtx(overrides?: { pathMaxLength?: number; branch?: string | null }; } -function createStatusLineSession(sessionName: string) { +function createStatusLineSession(sessionName: string, modelName?: string) { + const model = modelName ? { name: modelName, contextWindow: 128000 } : undefined; return { - state: { messages: [] }, + state: { messages: [], model }, + messages: [], + model: model ?? { contextWindow: 128000 }, + contextUsageRevision: 0, + systemPrompt: [], + agent: { state: { tools: [] } }, + skills: [], isStreaming: false, + isAutoThinking: false, + autoResolvedThinkingLevel: () => undefined, + isAdvisorActive: () => false, + isFastModeActive: () => false, getAsyncJobSnapshot: () => ({ running: [] }), getCurrentModel: () => undefined, isFastModeEnabled: () => false, + getContextUsage: () => ({ tokens: 0, contextWindow: 128000 }), + getGoalModeState: () => null, + modelRegistry: { isUsingOAuth: () => false }, sessionManager: { getSessionName: () => sessionName, getUsageStatistics: () => ({ @@ -102,6 +116,10 @@ function createStatusLineSession(sessionName: string) { } as unknown as ConstructorParameters[0]; } +function stripAnsi(value: string): string { + return value.replace(/\x1B\[[0-?]*[ -/]*[@-~]/g, ""); +} + describe("status line session accent", () => { function buildComponent(sessionAccent: boolean) { const component = new StatusLineComponent(createStatusLineSession("Named session")); @@ -244,10 +262,17 @@ describe("overflow: path shrinks before git is dropped", () => { } } - // Left-pop loop (fallback) + // Left-segment fallback loop. + const leftOverflowDropIndex = (): number => { + for (let i = leftSegIds.length - 1; i >= 0; i--) { + if (leftSegIds[i] !== "path") return i; + } + return left.length - 1; + }; while (groupWidth() > width && left.length > 0) { - left.pop(); - leftSegIds.pop(); + const dropIdx = leftOverflowDropIndex(); + left.splice(dropIdx, 1); + leftSegIds.splice(dropIdx, 1); } return { surviving: [...leftSegIds], contents: [...left] }; @@ -285,18 +310,20 @@ describe("overflow: path shrinks before git is dropped", () => { }); it("shrinks a short path when maxLength exceeds actual path length", () => { - // Short dir name — rendered path is well under maxLength=80 + // Short dir name — rendered path is well under the configured maxLength. const shortDir = fs.mkdtempSync(path.join(os.tmpdir(), "omp-short-")); setProjectDir(shortDir); try { - const ctx = createCtx({ pathMaxLength: 80, branch: "feat/long-branch-name" }); + const maxLength = 160; + const ctx = createCtx({ pathMaxLength: maxLength, branch: "feat/long-branch-name" }); const fullPath = renderSegment("path", ctx); const fullGit = renderSegment("git", ctx); const pathVW = visibleWidth(fullPath.content); const gitVW = visibleWidth(fullGit.content); - // Sanity: path is shorter than maxLength — this is the bug scenario - expect(pathVW).toBeLessThan(80); + // Sanity: path is shorter than maxLength — this is the bug scenario. + // macOS temp paths can exceed 80 columns once the path icon is included. + expect(pathVW).toBeLessThan(maxLength); // Width that fits a shrunken path + git but not the full path + git const tightWidth = Math.floor(pathVW * 0.5) + gitVW + 10; @@ -338,3 +365,61 @@ describe("overflow: path shrinks before git is dropped", () => { } }); }); + +describe("overflow: path survives before model", () => { + it("drops the model segment before the cwd path when both cannot fit", () => { + const root = fs.mkdtempSync(path.join(os.tmpdir(), "omp-statusline-overflow-")); + const cwd = path.join(root, "cwdxyz"); + fs.mkdirSync(cwd); + setProjectDir(cwd); + + const modelName = `MODEL_SHOULD_DROP_${"x".repeat(24)}`; + const session = createStatusLineSession("overflow test", modelName); + const component = new StatusLineComponent(session); + const pathOptions = { + abbreviate: false, + maxLength: 32, + stripWorkPrefix: false, + }; + component.updateSettings({ + preset: "custom", + leftSegments: ["pi", "model", "path"], + rightSegments: [], + separator: "none", + sessionAccent: false, + transparent: true, + segmentOptions: { + model: { showThinkingLevel: false }, + path: pathOptions, + }, + }); + + const ctx = { + ...createCtx({ pathMaxLength: pathOptions.maxLength }), + session, + options: { + model: { showThinkingLevel: false }, + path: pathOptions, + }, + } as SegmentContext; + const pi = renderSegment("pi", ctx).content; + const model = renderSegment("model", ctx).content; + const minPath = renderSegment("path", { + ...ctx, + options: { ...ctx.options, path: { ...pathOptions, maxLength: 4 } }, + }).content; + const separatorWidth = visibleWidth(theme.sep.space); + const groupWidth = (parts: string[]) => + parts.reduce((sum, part) => sum + visibleWidth(part), 0) + + Math.max(0, parts.length - 1) * (separatorWidth + 2) + + 2; + const width = groupWidth([pi, model]) + 1; + + expect(groupWidth([pi, model, minPath])).toBeGreaterThan(width); + expect(groupWidth([pi, minPath])).toBeLessThanOrEqual(width); + + const rendered = stripAnsi(component.getTopBorder(width).content); + expect(rendered).toContain("xyz"); + expect(rendered).not.toContain("MODEL_SHOULD_DROP"); + }); +}); diff --git a/packages/coding-agent/test/system-prompt-inventory.test.ts b/packages/coding-agent/test/system-prompt-inventory.test.ts index 8a075f367..a142f978a 100644 --- a/packages/coding-agent/test/system-prompt-inventory.test.ts +++ b/packages/coding-agent/test/system-prompt-inventory.test.ts @@ -2,13 +2,15 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { buildSystemPrompt as buildSdkSystemPrompt } from "@oh-my-pi/pi-coding-agent/sdk"; import { buildSystemPrompt, + buildSystemPromptToolMetadata, DEFAULT_SYSTEM_PROMPT_TOOL_NAMES, type SystemPromptToolMetadata, } from "@oh-my-pi/pi-coding-agent/system-prompt"; -import type { Tool } from "@oh-my-pi/pi-coding-agent/tools"; +import { createTools, type Tool, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { cleanupTempHome } from "./helpers/temp-home-cleanup"; const EMPTY_TREE = { @@ -92,6 +94,16 @@ describe("system prompt tool inventory", () => { return text.slice(inventoryStart, inventoryEnd); } + function makeToolSession(settings: Settings): ToolSession { + return { + cwd: tempDir, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings, + } as ToolSession; + } + it("renders a compact name list only when native tools are active and descriptors stay in schemas", async () => { const text = await render({ nativeTools: true, inlineToolDescriptors: false }); expect(text).toContain("- Read: `read`"); @@ -132,6 +144,46 @@ describe("system prompt tool inventory", () => { } expect(inventory).not.toContain("- `browser`"); expect(inventory).not.toContain("- `task`"); + expect(inventory).not.toContain("- `eval`"); + }); + + it("omits eval prompt guidance when every eval backend is disabled", async () => { + const settings = Settings.isolated({ + "eval.py": false, + "eval.js": false, + "eval.rb": false, + "eval.jl": false, + }); + const session = makeToolSession(settings); + const tools = await createTools(session, ["bash", "eval"]); + const toolNames = tools.map(tool => tool.name); + const bash = tools.find(tool => tool.name === "bash"); + + expect(toolNames).toContain("bash"); + expect(toolNames).not.toContain("eval"); + expect(bash?.description).toContain("purpose-built tool"); + expect(bash?.description).not.toContain("eval` cell"); + expect(bash?.description).not.toContain("use `eval` cells"); + expect(bash?.description).not.toContain("Prefer `eval`"); + expect(bash?.description).not.toContain("`grep` tool"); + expect(bash?.description).not.toContain("`ls` → `read`"); + expect(bash?.description).not.toContain("`find` → the `glob` tool"); + + const { systemPrompt } = await buildSystemPrompt({ + cwd: tempDir, + contextFiles: [], + skills: [], + rules: [], + toolNames, + tools: buildSystemPromptToolMetadata(new Map(tools.map(tool => [tool.name, tool]))), + workspaceTree: { ...EMPTY_TREE, rootPath: tempDir }, + nativeTools: true, + inlineToolDescriptors: true, + }); + const text = systemPrompt.join("\n\n"); + + expect(text).not.toContain("Default for any compute"); + expect(text).not.toContain("use `eval` cells"); }); it("SDK wrapper renders provided tools instead of the fallback inventory", async () => { diff --git a/packages/coding-agent/test/system-prompt-model.test.ts b/packages/coding-agent/test/system-prompt-model.test.ts index 8a0eba4f3..276e103ac 100644 --- a/packages/coding-agent/test/system-prompt-model.test.ts +++ b/packages/coding-agent/test/system-prompt-model.test.ts @@ -21,6 +21,69 @@ const EMPTY_TREE = { agentsMdFiles: [], }; +async function expectPromptDateFromStartupTimezone(options: { + tempDir: string; + tempHomeDir: string; + timeZone: string; + now: string; + expectedDate: string; + rejectedDate: string; +}): Promise { + const scenarioPath = path.join(options.tempDir, "prompt-date-timezone.test.ts"); + await Bun.write( + scenarioPath, + `import { expect, it, setSystemTime } from "bun:test"; +import { buildSystemPrompt } from ${JSON.stringify(path.resolve(import.meta.dir, "../src/system-prompt.ts"))}; + +it("renders the prompt date in the startup timezone", async () => { + setSystemTime(new Date(process.env.OMP_TEST_NOW!)); + try { + const { systemPrompt } = await buildSystemPrompt({ + cwd: process.cwd(), + contextFiles: [], + skills: [], + rules: [], + toolNames: [], + workspaceTree: { + rootPath: process.cwd(), + rendered: "", + truncated: false, + totalLines: 0, + agentsMdFiles: [], + }, + activeRepoContext: null, + }); + const rendered = systemPrompt.join("\\n\\n"); + expect(rendered).toContain(\`Today is \${process.env.OMP_EXPECTED_DATE}\`); + expect(rendered).not.toContain(\`Today is \${process.env.OMP_REJECTED_DATE}\`); + } finally { + setSystemTime(); + } +}); +`, + ); + const child = Bun.spawn([process.execPath, "test", scenarioPath], { + cwd: options.tempDir, + env: { + ...process.env, + HOME: options.tempHomeDir, + TZ: options.timeZone, + OMP_TEST_NOW: options.now, + OMP_EXPECTED_DATE: options.expectedDate, + OMP_REJECTED_DATE: options.rejectedDate, + }, + stdout: "pipe", + stderr: "pipe", + }); + const [stdout, stderr, exitCode] = await Promise.all([ + new Response(child.stdout).text(), + new Response(child.stderr).text(), + child.exited, + ]); + expect(`${stdout}\n${stderr}`).toContain("1 pass"); + expect(exitCode).toBe(0); +} + describe("system prompt model identifier", () => { let tempDir = ""; let tempHomeDir = ""; @@ -49,6 +112,17 @@ describe("system prompt model identifier", () => { expect(systemPrompt.join("\n\n")).toContain("Model: anthropic/claude-opus-4"); }); + it("renders the prompt date from the startup local timezone rather than UTC", async () => { + await expectPromptDateFromStartupTimezone({ + tempDir, + tempHomeDir, + timeZone: "America/Los_Angeles", + now: "2026-07-01T03:15:00Z", + expectedDate: "2026-06-30", + rejectedDate: "2026-07-01", + }); + }); + it("omits the model line when no model is provided", async () => { const { systemPrompt } = await buildSystemPrompt({ cwd: tempDir, diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index a556a4351..a3baf9bbd 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -1492,6 +1492,21 @@ function b() { expect(result.details?.requestedTimeoutSeconds).toBe(7200); }); + it("should disable the command deadline when timeout is zero", async () => { + vi.spyOn(toolTimeouts, "clampTimeout").mockReturnValue(0.05); + + const result = await bashTool.execute("test-call-timeout-disabled", { + command: "printf 'start\\n'; sleep 0.1; printf 'done\\n'", + timeout: 0, + }); + + const output = getTextOutput(result); + expect(output).toContain("start"); + expect(output).toContain("done"); + expect(result.details?.timeoutDisabled).toBe(true); + expect(result.details?.timeoutSeconds).toBeUndefined(); + }); + it("should respect timeout", async () => { // Reduce the effective timeout through the production clamp seam; the // real subprocess kill-on-timeout path is still exercised, just faster. diff --git a/packages/coding-agent/test/tools/bash-skill-urls.test.ts b/packages/coding-agent/test/tools/bash-skill-urls.test.ts index e35fb77d4..153eef880 100644 --- a/packages/coding-agent/test/tools/bash-skill-urls.test.ts +++ b/packages/coding-agent/test/tools/bash-skill-urls.test.ts @@ -178,6 +178,22 @@ describe("expandInternalUrls", () => { ); }); + it("leaves literal internal URLs embedded in quoted text unchanged", async () => { + const router = createInternalRouter({ + "memory://root/summary.md": { sourcePath: "/tmp/memories/summary.md" }, + }); + const command = `printf '%s\\n' 'the literal memory://root/summary.md string'`; + + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); + }); + + it("leaves unresolved quoted literal URLs unchanged", async () => { + const router = createInternalRouter({}); + const command = "grep 'memory://xyz-quoted' file.txt"; + + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); + }); + it("expands agent:// URLs when router is available", async () => { const router = createInternalRouter({ "agent://abc": { sourcePath: "/tmp/session/abc.md" }, @@ -239,34 +255,30 @@ describe("expandInternalUrls", () => { ); }); - it("throws when local:// URL is used without local protocol options", async () => { - await expect(expandInternalUrls("mv foo local://bar", { skills: [] })).rejects.toThrow( - "Cannot resolve local:// URL in bash command: local protocol options are unavailable for this session.", - ); + it("leaves local:// URLs unchanged without local protocol options", async () => { + const command = "mv foo local://bar"; + await expect(expandInternalUrls(command, { skills: [] })).resolves.toBe(command); }); - it("throws when non-skill URL is used without an internal router", async () => { - await expect(expandInternalUrls("cat artifact://1", { skills: [] })).rejects.toThrow( - "Cannot resolve artifact:// URL in bash command", - ); + it("leaves non-skill URLs unchanged without an internal router", async () => { + const command = "cat artifact://1"; + await expect(expandInternalUrls(command, { skills: [] })).resolves.toBe(command); }); - it("throws when internal router resolves URL without sourcePath", async () => { + it("leaves internal URLs unchanged when they resolve without sourcePath", async () => { const router = createInternalRouter({ "rule://my-rule": {}, }); - await expect(expandInternalUrls("cat rule://my-rule", { skills: [], internalRouter: router })).rejects.toThrow( - "rule:// URL resolved without a filesystem path", - ); + const command = "cat rule://my-rule"; + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); }); - it("surfaces resolver errors with actionable context", async () => { + it("leaves internal URLs unchanged when the resolver fails", async () => { const router = createInternalRouter({ "memory://root/missing.md": { error: "Memory file not found" }, }); - await expect( - expandInternalUrls("cat memory://root/missing.md", { skills: [], internalRouter: router }), - ).rejects.toThrow("Failed to resolve memory:// URL in bash command"); + const command = "cat memory://root/missing.md"; + await expect(expandInternalUrls(command, { skills: [], internalRouter: router })).resolves.toBe(command); }); it("does not match local:/ inside filesystem paths (e.g. /repo/local:/PLAN.md)", async () => { diff --git a/packages/coding-agent/test/tools/browser-launch.test.ts b/packages/coding-agent/test/tools/browser-launch.test.ts new file mode 100644 index 000000000..1d677507f --- /dev/null +++ b/packages/coding-agent/test/tools/browser-launch.test.ts @@ -0,0 +1,35 @@ +import { describe, expect, it } from "bun:test"; +import { stealthIgnoreDefaultArgsForTest } from "@oh-my-pi/pi-coding-agent/tools/browser/launch"; + +const AUTOMATION_FLAG = "--enable-automation"; + +const EDGE_EXECUTABLE_PATHS = [ + "C:\\Program Files\\Microsoft\\Edge\\Application\\msedge.exe", + "/Applications/Microsoft Edge.app/Contents/MacOS/Microsoft Edge", + "/usr/bin/microsoft-edge-stable", +] as const; + +const CHROME_EXECUTABLE_PATHS = [ + "C:\\Program Files\\Google\\Chrome\\Application\\chrome.exe", + "/Applications/Google Chrome.app/Contents/MacOS/Google Chrome", + "/usr/bin/chromium", +] as const; + +describe("browser launch stealth defaults", () => { + it("keeps Puppeteer's automation default for Microsoft Edge executables", () => { + for (const executablePath of EDGE_EXECUTABLE_PATHS) { + const ignoreDefaultArgs = stealthIgnoreDefaultArgsForTest(executablePath); + + expect(ignoreDefaultArgs).not.toContain(AUTOMATION_FLAG); + expect(ignoreDefaultArgs).toContain("--disable-extensions"); + } + }); + + it("continues filtering Puppeteer's automation default for Chrome and Chromium executables", () => { + for (const executablePath of CHROME_EXECUTABLE_PATHS) { + const ignoreDefaultArgs = stealthIgnoreDefaultArgsForTest(executablePath); + + expect(ignoreDefaultArgs).toContain(AUTOMATION_FLAG); + } + }); +}); diff --git a/packages/coding-agent/test/tools/edit-renderer.test.ts b/packages/coding-agent/test/tools/edit-renderer.test.ts index 7962c7353..7b84723cf 100644 --- a/packages/coding-agent/test/tools/edit-renderer.test.ts +++ b/packages/coding-agent/test/tools/edit-renderer.test.ts @@ -7,10 +7,12 @@ import type { AgentTool } from "@oh-my-pi/pi-agent-core"; import { renderGalleryState, resolveFixture } from "@oh-my-pi/pi-coding-agent/cli/gallery-cli"; import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { editToolRenderer } from "@oh-my-pi/pi-coding-agent/edit/renderer"; +import { renderDiff } from "@oh-my-pi/pi-coding-agent/modes/components/diff"; import { ToolExecutionComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tool-execution"; import * as themeModule from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { Text, type TUI, visibleWidth } from "@oh-my-pi/pi-tui"; import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import chalk from "chalk"; beforeAll(async () => { resetSettingsForTest(); @@ -519,3 +521,172 @@ describe("editToolRenderer", () => { expect(text).not.toContain("No changes"); }); }); + +describe("editToolRenderer diff line wrapping", () => { + // Renders a completed single-line replacement (`-N|old` + `+N|new`) through + // the real renderDiff so the result carries its production shapes: a blanked + // dedup gutter on the `+` row (` +│`) and intra-line inverse highlights. + async function renderSingleLineReplacement( + oldLine: string, + newLine: string, + width: number, + ): Promise { + const uiTheme = await getUiTheme(); + const component = editToolRenderer.renderResult( + { + content: [{ type: "text", text: "Updated demo.ts" }], + details: { diff: `-42|${oldLine}\n+42|${newLine}`, op: "update", path: "demo.ts" }, + }, + { expanded: true, isPartial: false, renderContext: { renderDiff } }, + uiTheme, + { file_path: "demo.ts" }, + ); + return component.render(width); + } + + /** Net SGR inverse state after scanning a row; 38/48 extended-color args must not be misread as attribute 7. */ + function inverseActiveAtRowEnd(row: string): boolean { + let inverse = false; + for (const match of row.matchAll(/\x1b\[([0-9;]*)m/g)) { + const params = match[1].split(";"); + for (let i = 0; i < params.length; i++) { + const param = params[i]; + if (param === "38" || param === "48") { + i += params[i + 1] === "2" ? 4 : params[i + 1] === "5" ? 2 : 0; + } else if (param === "" || param === "0") inverse = false; + else if (param === "7") inverse = true; + else if (param === "27") inverse = false; + } + } + return inverse; + } + + it("keeps added-line continuation rows inside the blanked dedup gutter", async () => { + // renderDiff blanks the repeated line number on the `+` row of a + // single-line replacement (` +│`); the wrapper must still recognize that + // gutter instead of falling back to generic wrapping at column 0. + const rows = ( + await renderSingleLineReplacement( + " the previous synopsis paragraph rambled across quarterly reconciliation notes enumerating every provisional ledger amendment the archival committee had deferred pending review by the regional custodians during the extended winter recess of the auditing season", + " the revised synopsis paragraph now catalogues seasonal festival logistics enumerating lantern shipments drum rehearsals and ribbon inventories that the parade stewards confirmed before dawn, closing with the zephyrQuota tally and the marbledFinale banner", + 100, + ) + ).map(row => Bun.stripANSI(row)); + + // The tail of the added line lands on continuation rows, which must carry + // the spaces-only continuation gutter rather than start as bare prose. + const tailRows = rows.filter(row => row.includes("zephyrQuota") || row.includes("marbledFinale")); + expect(tailRows.length).toBeGreaterThanOrEqual(1); + for (const row of tailRows) expect(row).toMatch(/^│\s+│/); + // Every body row stays inside a code-frame gutter (`-42│`, ` +│`, ` │`). + for (const row of rows.slice(1, -1)) expect(row).toMatch(/^│\s*[+-]?\s*\d*│/); + }); + + it("closes inverse video at every wrapped row end so frame padding stays uninverted", async () => { + // A long contiguous rewritten phrase forces the wrap boundary to land + // inside an inverse-highlighted span; the frame pads each row with spaces, + // so any inverse still active at row end paints those cells as gray blocks. + const previousLevel = chalk.level; + chalk.level = 3; + let rows: readonly string[]; + try { + rows = await renderSingleLineReplacement( + " stanza recounts venerable chronicle passages spanning bygone dynasties whose archivists engraved ledgers onto vellum scrolls", + " stanza celebrates luminous festival processions winding through lantern boulevards while drummers herald jubilant choruses beneath cascading ribbons and fireworks", + 100, + ); + } finally { + chalk.level = previousLevel; + } + + // Precondition: some continuation row's content reopens with inverse right + // after its gutter, proving a highlighted span crossed a wrap boundary. If + // diffWords tokenization ever changes so no span crosses, this fails loudly + // instead of letting the row-end assertions pass vacuously. + expect(rows.some(row => /│\x1b\[7m/.test(row))).toBe(true); + for (const row of rows) expect(inverseActiveAtRowEnd(row)).toBe(false); + }); + + // Error results reuse the same body-line wrapper as diff rows; these tests + // pin the boundary between prose that merely looks pipe-ish and real gutters. + async function renderErrorResultRows(errorText: string): Promise { + const uiTheme = await getUiTheme(); + const component = editToolRenderer.renderResult( + { + content: [{ type: "text", text: errorText }], + details: { diff: "", op: "update", path: "demo.ts" }, + isError: true, + }, + { expanded: true, isPartial: false, renderContext: { renderDiff } }, + uiTheme, + { file_path: "demo.ts" }, + ); + return component.render(100).map(row => Bun.stripANSI(row)); + } + + it("does not give pipe-leading error text a phantom diff gutter when wrapping", async () => { + // Error text is not a diff row even when it starts with `|`: an empty + // gutter must wrap generically, not spawn `|` continuation prefixes. + const rows = await renderErrorResultRows( + "| pipe-leading diagnostic output that is quite long and should certainly wrap at the render width because it keeps going on and on with more words than fit in one row of the frame", + ); + const bodyRows = rows.slice(1, -1); + // Precondition: the text actually wrapped, and the `|` lead survived on row one. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│\| /); + for (const row of bodyRows.slice(1)) expect(row).not.toMatch(/^│\s*\|/); + }); + + it("wraps spaces-then-bare-pipe error text generically instead of minting a gutter", async () => { + // A digit-less ASCII "|" gutter never comes out of formatCodeFrameLine or + // canonical diff rows; indented bare-pipe error text must wrap generically. + const rows = await renderErrorResultRows( + " | indented bare-pipe diagnostic output that is quite long and should certainly wrap at the render width because it keeps going on and on with more words than fit in one row of the frame", + ); + const bodyRows = rows.slice(1, -1); + // Precondition: the text actually wrapped, and the pipe lead survived on row one. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│\s+\| /); + for (const row of bodyRows.slice(1)) expect(row).not.toMatch(/^│\s*\|/); + }); + + it("wraps digit-leading pipe error text generically when the marker column is missing", async () => { + // Canonical ASCII-pipe rows always carry a marker column (`-42|`, ` 42|`); + // `123|` prose has a digit there instead, so it is not a diff row. + const rows = await renderErrorResultRows( + "123| numbered pipe-leading diagnostic output that is quite long and should certainly wrap at the render width because it keeps going on and on with more words than fit in one row of the frame", + ); + const bodyRows = rows.slice(1, -1); + // Precondition: the text actually wrapped, and the numbered lead survived on row one. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│123\| /); + for (const row of bodyRows.slice(1)) expect(row).not.toMatch(/^│\s*\|/); + }); + + it("keeps the numbered ASCII-pipe gutter for canonical rows through the plain fallback", async () => { + // Without renderContext the plain fallback passes canonical rows through + // verbatim; a numbered "-42|" row must still take the gutter path and + // carry an " |" continuation gutter, not generic prose wrapping. + const uiTheme = await getUiTheme(); + const component = editToolRenderer.renderResult( + { + content: [{ type: "text", text: "Updated demo.ts" }], + details: { + diff: "-42| the previous synopsis paragraph rambled across quarterly reconciliation notes enumerating every provisional ledger amendment the archival committee had deferred pending review by the regional custodians", + op: "update", + path: "demo.ts", + }, + }, + { expanded: true, isPartial: false }, + uiTheme, + { file_path: "demo.ts" }, + ); + + const rows = component.render(100).map(row => Bun.stripANSI(row)); + const bodyRows = rows.slice(1, -1); + // Precondition: the row actually wrapped past its first visual line. + expect(bodyRows.length).toBeGreaterThanOrEqual(2); + expect(bodyRows[0]).toMatch(/^│-42\|/); + for (const row of bodyRows.slice(1)) expect(row).toMatch(/^│\s+\|/); + }); +}); diff --git a/packages/coding-agent/test/tools/grep-internal-urls.test.ts b/packages/coding-agent/test/tools/grep-internal-urls.test.ts index 417402f7c..b0d2e48ae 100644 --- a/packages/coding-agent/test/tools/grep-internal-urls.test.ts +++ b/packages/coding-agent/test/tools/grep-internal-urls.test.ts @@ -450,6 +450,35 @@ describe("GrepTool internal URL resolution", () => { expect(text).toMatch(/^\*\d+:.*needle/m); }); + it("read local://: honors URL selector even when a sibling literal `:` file exists (issue #4618)", async () => { + const localRoot = path.join(artifactsDir, "local"); + await fs.mkdir(localRoot, { recursive: true }); + // Base file targeted by `local://notes.md`; selector should slice this one. + await Bun.write( + path.join(localRoot, "notes.md"), + `${Array.from({ length: 10 }, (_, i) => `url-target line ${i + 1}`).join("\n")}\n`, + ); + // Sibling literal `notes.md:1-2` under the same local root — must NOT + // shadow the URL selector semantics of `local://notes.md:1-2`. + await Bun.write(path.join(localRoot, "notes.md:1-2"), "sibling literal shadow\n"); + + LocalProtocolHandler.setOverride({ getArtifactsDir: () => artifactsDir, getSessionId: () => "session" }); + + const session = createSession({ hasEditTool: true }); + session.settings.set("read.summarize.enabled", false); + const result = await new ReadTool(session).execute("test-read-local-url-selector", { + path: "local://notes.md:1-2", + }); + + const text = getResultText(result); + // The base file was targeted (URL selector semantics preserved), not the + // sibling literal. Content check is enough — the read tool's context + // expansion around the requested range is unrelated to the shadow bug. + expect(text).toContain("url-target line 1"); + expect(text).toContain("url-target line 2"); + expect(text).not.toContain("sibling literal shadow"); + }); + it("keeps hashlines on mutable files when mixed with immutable artifact:// inputs", async () => { const content = "alpha line\nbeta needle line\ngamma line\n"; await Bun.write(path.join(artifactsDir, "11.bash.log"), content); diff --git a/packages/coding-agent/test/tools/index.test.ts b/packages/coding-agent/test/tools/index.test.ts index eae16b79c..845a285b1 100644 --- a/packages/coding-agent/test/tools/index.test.ts +++ b/packages/coding-agent/test/tools/index.test.ts @@ -264,6 +264,37 @@ describe("createTools", () => { expect(names).toEqual(["read", "goal", "resolve"]); }); + it("records active tools on the original session object", async () => { + const session = createTestSession(); + + await createTools(session, ["bash"]); + + expect(session.isToolActive?.("bash")).toBe(true); + expect(session.isToolActive?.("read")).toBe(false); + }); + + it("renders bash guidance from the live active tool predicate", async () => { + const activeToolNames = new Set(); + const session = createTestSession({ + isToolActive: name => activeToolNames.has(name), + setActiveToolNames: names => { + activeToolNames.clear(); + for (const name of names) { + activeToolNames.add(name); + } + }, + }); + + const tools = await createTools(session, ["bash", "grep", "read", "glob"]); + const bash = tools.find(tool => tool.name === "bash"); + + expect(bash?.description).toContain("`grep` tool"); + session.setActiveToolNames?.(["bash"]); + expect(bash?.description).not.toContain("`grep` tool"); + expect(bash?.description).not.toContain("`ls` → `read`"); + expect(bash?.description).not.toContain("`find` → the `glob` tool"); + }); + it("includes search_tool_bm25 when MCP tool discovery is enabled and executable", async () => { const session = createTestSession({ settings: createSettingsWithOverrides({ diff --git a/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts new file mode 100644 index 000000000..e2a7d2544 --- /dev/null +++ b/packages/coding-agent/test/tools/path-literal-colon-selector.test.ts @@ -0,0 +1,346 @@ +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { + probeLiteralPathExists, + splitPathAndSel, + splitPathAndSelPreferringLiteral, +} from "@oh-my-pi/pi-coding-agent/tools/path-utils"; +import { ReadTool } from "@oh-my-pi/pi-coding-agent/tools/read"; +import { removeWithRetries } from "@oh-my-pi/pi-utils"; +import { GrepTool } from "../../src/tools/grep"; + +function getText(result: { content: Array<{ type: string; text?: string }> }): string { + return result.content + .filter(entry => entry.type === "text") + .map(entry => entry.text ?? "") + .join("\n"); +} + +const EMPTY_ZIP_EOCD = new Uint8Array([0x50, 0x4b, 0x05, 0x06, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]); + +// Regression: filenames whose tail matches the read-tool selector grammar +// (e.g. `test:1-2`, `log:raw`) used to be shredded by `splitPathAndSel` before +// either tool checked the filesystem — see issue #4618. Both `read` and `grep` +// must prefer a real literal file over the selector interpretation. +describe("literal colon filename resolution (issue #4618)", () => { + let tmpDir: string; + + beforeEach(async () => { + tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "literal-colon-")); + }); + + afterEach(async () => { + await removeWithRetries(tmpDir); + }); + + function createSession(overrides: Partial = {}): ToolSession { + return { + cwd: tmpDir, + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => "*", + settings: Settings.isolated({ "grep.contextBefore": 0, "grep.contextAfter": 0 }), + ...overrides, + }; + } + + describe("splitPathAndSelPreferringLiteral", () => { + it("keeps the raw path intact when a literal colon file exists on disk", async () => { + const literal = "test:1-2"; + await Bun.write(path.join(tmpDir, literal), "test\n"); + + // Strict splitter still peels — this documents the contract the + // literal-preferring variant sits on top of. + expect(splitPathAndSel(literal)).toEqual({ path: "test", sel: "1-2" }); + + expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); + }); + + it("keeps a shell-escaped literal path intact when the resolved file exists", async () => { + await fs.mkdir(path.join(tmpDir, "dir"), { recursive: true }); + await Bun.write(path.join(tmpDir, "dir", "a b:1-2"), "escaped literal\n"); + + expect(await splitPathAndSelPreferringLiteral("dir/a\\ b:1-2", tmpDir)).toEqual({ + path: "dir/a\\ b:1-2", + }); + }); + + it("falls back to selector interpretation when the literal path does not exist", async () => { + // No file created — the selector split wins because the raw path + // cannot be stat'd. + expect(await splitPathAndSelPreferringLiteral("test:1-2", tmpDir)).toEqual({ + path: "test", + sel: "1-2", + }); + }); + + it("also protects `:raw`-shaped literal filenames", async () => { + const literal = "log:raw"; + await Bun.write(path.join(tmpDir, literal), "line one\nline two\n"); + expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); + }); + + it("keeps a literal dangling symlink intact (lstat exists even though stat fails)", async () => { + const literal = path.join(tmpDir, "test:1-2"); + await fs.symlink(path.join(tmpDir, "missing-target"), literal); + + expect(await probeLiteralPathExists(literal, tmpDir)).toBe("exists"); + expect(await splitPathAndSelPreferringLiteral(literal, tmpDir)).toEqual({ path: literal }); + }); + + it("returns the strict split unchanged when there is no selector tail", async () => { + expect(await splitPathAndSelPreferringLiteral("plain.txt", tmpDir)).toEqual({ + path: "plain.txt", + }); + }); + }); + + describe("probeLiteralPathExists", () => { + it('returns "missing" for a path that clearly does not exist', async () => { + expect(await probeLiteralPathExists(path.join(tmpDir, "never-here:1-2"), tmpDir)).toBe("missing"); + }); + + it('returns "exists" for a regular file', async () => { + const literal = path.join(tmpDir, "regular:1-2"); + await Bun.write(literal, "hi\n"); + expect(await probeLiteralPathExists(literal, tmpDir)).toBe("exists"); + }); + + it('returns "exists" for a dangling symlink', async () => { + const literal = path.join(tmpDir, "dangling:1-2"); + await fs.symlink(path.join(tmpDir, "nowhere"), literal); + expect(await probeLiteralPathExists(literal, tmpDir)).toBe("exists"); + }); + }); + + describe("read tool", () => { + it("reads a literal file whose name ends in a selector-shaped suffix", async () => { + const literal = "test:1-2"; + const absolute = path.join(tmpDir, literal); + await Bun.write(absolute, "test\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal", { path: absolute }); + const output = getText(result); + + expect(output).toContain("test"); + // The strict split would have opened `test` (which doesn't exist) + // and thrown "Path 'test' not found". + expect(output).not.toMatch(/not found/i); + }); + + it("reads a shell-escaped literal file whose name ends in a selector-shaped suffix", async () => { + await fs.mkdir(path.join(tmpDir, "dir"), { recursive: true }); + await Bun.write(path.join(tmpDir, "dir", "a b:1-2"), "escaped literal read\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-escaped-literal", { path: "dir/a\\ b:1-2" }); + const output = getText(result); + + expect(output).toContain("escaped literal read"); + }); + + it("prefers a real `foo:1-2` file over interpreting `:1-2` as a range on `foo`", async () => { + await Bun.write(path.join(tmpDir, "foo"), "line 1\nline 2\nline 3\n"); + await Bun.write(path.join(tmpDir, "foo:1-2"), "colon file wins\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal-wins", { + path: path.join(tmpDir, "foo:1-2"), + }); + const output = getText(result); + + expect(output).toContain("colon file wins"); + expect(output).not.toContain("line 1"); + }); + + it("still honors the `:5-10` selector when only the base file exists on disk", async () => { + const absolute = path.join(tmpDir, "notes"); + const lines = Array.from({ length: 40 }, (_, i) => `line ${i + 1}`).join("\n"); + await Bun.write(absolute, `${lines}\n`); + + const session = createSession(); + session.settings.set("read.summarize.enabled", false); + const tool = new ReadTool(session); + const result = await tool.execute("read-selector-preserved", { + path: `${absolute}:5-10`, + }); + const output = getText(result); + + expect(output).toContain("line 5"); + expect(output).toContain("line 10"); + // Lines well outside the requested range must not appear — the selector + // still peels because the raw `notes:5-10` path does not exist literally. + expect(output).not.toContain("line 30"); + expect(output).not.toContain("line 40"); + }); + + it("uses explicit `selector` to read lines from a literal selector-shaped filename deterministically", async () => { + const literal = path.join(tmpDir, "test:1-2"); + const longerLiteral = path.join(tmpDir, "test:1-2:5-6"); + const lines = Array.from({ length: 40 }, (_, i) => `literal line ${i + 1}`).join("\n"); + await Bun.write(literal, `${lines}\n`); + await Bun.write(longerLiteral, "wrong longer literal\n"); + + const session = createSession(); + session.settings.set("read.summarize.enabled", false); + const tool = new ReadTool(session); + const result = await tool.execute("read-explicit-selector-literal", { + path: literal, + selector: "5-6", + }); + const output = getText(result); + + expect(output).toContain("literal line 5"); + expect(output).toContain("literal line 6"); + expect(output).not.toContain("literal line 30"); + expect(output).not.toContain("wrong longer literal"); + }); + + it("reads a literal file that looks like an archive selector (`data.zip:1-2`)", async () => { + // A real POSIX file whose name ends in a selector-shaped tail after an + // archive extension. The archive resolver would otherwise open `data.zip` + // alongside it and error on the phantom member. + const baseArchive = path.join(tmpDir, "data.zip"); + // Empty zip bytes — the file just needs to stat as a real archive so + // the archive resolver would happily accept it. + await Bun.write(baseArchive, EMPTY_ZIP_EOCD); + const literal = path.join(tmpDir, "data.zip:1-2"); + await Bun.write(literal, "literal archive-shaped file\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal-zip-selector", { path: literal }); + const output = getText(result); + + expect(output).toContain("literal archive-shaped file"); + }); + + it("reads a literal file that looks like a sqlite selector (`notes.db:1-2`)", async () => { + // A real POSIX file whose base name matches a sqlite-shaped path plus a + // selector-shaped tail. The sqlite resolver would misroute this to + // `notes.db` and try to open a table named `1-2`. + const baseDb = path.join(tmpDir, "notes.db"); + // SQLite database header (16-byte magic string plus zero-padding). + const header = new Uint8Array(4096); + header.set(Buffer.from("SQLite format 3\0", "utf-8"), 0); + await Bun.write(baseDb, header); + const literal = path.join(tmpDir, "notes.db:1-2"); + await Bun.write(literal, "literal db-shaped file\n"); + + const tool = new ReadTool(createSession()); + const result = await tool.execute("read-literal-db-selector", { path: literal }); + const output = getText(result); + + expect(output).toContain("literal db-shaped file"); + }); + }); + + describe("grep tool", () => { + it("searches inside a literal `test:1-2` file", async () => { + const literal = "test:1-2"; + const absolute = path.join(tmpDir, literal); + await Bun.write(absolute, "needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-literal", { + pattern: "needle", + path: absolute, + }); + const output = getText(result); + + expect(output).toContain("needle"); + expect(output).not.toMatch(/not found/i); + }); + + it("uses explicit `selector` to grep a literal selector-shaped filename deterministically", async () => { + const literal = path.join(tmpDir, "test:1-2"); + const longerLiteral = path.join(tmpDir, "test:1-2:2-2"); + await Bun.write(literal, "needle outside\nneedle inside\nneedle outside again\n"); + await Bun.write(longerLiteral, "wrong longer literal needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-explicit-selector-literal", { + pattern: "needle", + path: literal, + selector: "2-2", + }); + const output = getText(result); + + expect(output).toContain("needle inside"); + expect(output).not.toContain("needle outside again"); + expect(output).not.toContain("wrong longer literal"); + }); + + it("searches a shell-escaped literal file whose name ends in a selector-shaped suffix", async () => { + await fs.mkdir(path.join(tmpDir, "dir"), { recursive: true }); + await Bun.write(path.join(tmpDir, "dir", "a b:1-2"), "escaped literal needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-escaped-literal", { + pattern: "needle", + path: "dir/a\\ b:1-2", + }); + const output = getText(result); + + expect(output).toContain("escaped literal needle"); + }); + + it("searches a literal file whose name contains a semicolon and selector-shaped tail (`a;b:1-2`)", async () => { + // Semicolon is the delimited-path separator; without a raw-literal + // probe in `splitDelimitedPathEntry`, expandDelimitedPathEntries would + // split `a;b:1-2` into `["a", "b:1-2"]` before grep saw the literal file. + const literal = path.join(tmpDir, "a;b:1-2"); + await Bun.write(literal, "delimited literal needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-literal-semicolon-selector", { + pattern: "needle", + path: literal, + }); + const output = getText(result); + + expect(output).toContain("delimited literal needle"); + expect(output).not.toMatch(/not found/i); + }); + + it("searches a literal file that looks like an archive selector (`data.zip:1-2`)", async () => { + // The base archive exists too; grep must not rematerialize the raw + // literal path as archive `data.zip` plus phantom member `1-2`. + const baseArchive = path.join(tmpDir, "data.zip"); + await Bun.write(baseArchive, EMPTY_ZIP_EOCD); + const literal = path.join(tmpDir, "data.zip:1-2"); + await Bun.write(literal, "literal archive needle\n"); + + const tool = new GrepTool(createSession()); + const result = await tool.execute("grep-literal-zip-selector", { + pattern: "needle", + path: literal, + }); + const output = getText(result); + + expect(output).toContain("literal archive needle"); + }); + + it("preserves `:N-M` line-range filtering when the literal file does not exist", async () => { + const absolute = path.join(tmpDir, "notes.txt"); + await Bun.write(absolute, "one\ntwo\nthree\nfour\n"); + + const tool = new GrepTool(createSession()); + const rangedResult = await tool.execute("grep-range-filter", { + pattern: ".", + path: `${absolute}:1-2`, + }); + const rangedOutput = getText(rangedResult); + + expect(rangedOutput).toContain("one"); + expect(rangedOutput).toContain("two"); + // Lines outside the range are filtered out. + expect(rangedOutput).not.toContain("three"); + expect(rangedOutput).not.toContain("four"); + }); + }); +}); diff --git a/packages/coding-agent/test/utils/open.test.ts b/packages/coding-agent/test/utils/open.test.ts index 467c28f4e..11d28a46d 100644 --- a/packages/coding-agent/test/utils/open.test.ts +++ b/packages/coding-agent/test/utils/open.test.ts @@ -1,5 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import * as fs from "node:fs"; +import * as path from "node:path"; import { openPath } from "@oh-my-pi/pi-coding-agent/utils/open"; import * as piUtils from "@oh-my-pi/pi-utils"; import type { Subprocess } from "bun"; @@ -91,6 +92,12 @@ describe("openPath", () => { it("opens existing WSL mount files through wslview with a Windows path", () => { setPlatform("linux"); process.env.WSL_DISTRO_NAME = "Ubuntu"; + // Keep the mocked linux platform deterministic on a Windows dev host: + // the real path.resolve would rewrite /mnt/c/… against the drive root. + const realResolve = path.resolve; + vi.spyOn(path, "resolve").mockImplementation((...segments: string[]) => + segments.length === 1 && segments[0] === existingLinuxPath ? existingLinuxPath : realResolve(...segments), + ); vi.spyOn(piUtils, "$which").mockImplementation(command => (command === "wslview" ? "/usr/bin/wslview" : null)); vi.spyOn(fs, "existsSync").mockImplementation(candidate => candidate === existingLinuxPath); @@ -133,50 +140,100 @@ describe("openPath", () => { expect(spawnCalls.map(call => call.cmd)).toEqual([["xdg-open", existingLinuxPath]]); }); - it("resolves rundll32 through %SystemRoot% so a broken machine PATH cannot silence the opener", () => { + it("resolves PowerShell through %SystemRoot% so a broken machine PATH cannot silence the opener", () => { setPlatform("win32"); const originalSystemRoot = process.env.SystemRoot; process.env.SystemRoot = "D:\\CustomWindows"; + const powershellPath = "D:\\CustomWindows\\System32\\WindowsPowerShell\\v1.0\\powershell.exe"; + vi.spyOn(fs, "existsSync").mockImplementation(candidate => candidate === powershellPath); try { const spawnCalls: SpawnCall[] = []; spySpawn(spawnCalls); - openPath("https://mcp.linear.app/authorize?state=xyz&code_challenge_method=S256"); + const url = "https://mcp.linear.app/authorize?state=xyz&code_challenge_method=S256"; + openPath(url); expect(spawnCalls).toHaveLength(1); const [call] = spawnCalls; - // Absolute rundll32 path — bare `rundll32` was the whole bug on Windows - // boxes where the machine PATH no longer references System32. - expect(call?.cmd[0]).toBe("D:\\CustomWindows\\System32\\rundll32.exe"); - // Handler + URL forwarded verbatim as a single argv slot so `&` in the - // query string cannot be interpreted as a shell separator. - expect(call?.cmd.slice(1)).toEqual([ - "url.dll,FileProtocolHandler", - "https://mcp.linear.app/authorize?state=xyz&code_challenge_method=S256", + // Absolute PowerShell path — bare executable names were the whole bug + // on Windows boxes where the machine PATH no longer references + // System32. + expect(call?.cmd[0]).toBe(powershellPath); + expect(call?.cmd.slice(1, -1)).toEqual([ + "-NoProfile", + "-NonInteractive", + "-WindowStyle", + "Hidden", + "-EncodedCommand", ]); + // The target rides inside the UTF-16LE payload: no cmd/PowerShell + // metacharacter parsing ever sees the `&` in the query string, and the + // terminating error preference makes Start-Process failures exit 1 so + // openPath's non-zero-exit telemetry observes them (rundll32 always + // exited 0). + const decoded = Buffer.from(String(call?.cmd.at(-1)), "base64").toString("utf16le"); + expect(decoded).toBe(`$ErrorActionPreference='Stop';Start-Process '${url}'`); } finally { if (originalSystemRoot === undefined) delete process.env.SystemRoot; else process.env.SystemRoot = originalSystemRoot; } }); - it("falls back to C:\\Windows for rundll32 when SystemRoot is unset", () => { + it("doubles embedded single quotes so the target stays one PowerShell literal", () => { + setPlatform("win32"); + vi.spyOn(fs, "existsSync").mockReturnValue(true); + const spawnCalls: SpawnCall[] = []; + spySpawn(spawnCalls); + + openPath("C:\\Users\\o'brien\\report.html"); + + const decoded = Buffer.from(String(spawnCalls[0]?.cmd.at(-1)), "base64").toString("utf16le"); + expect(decoded).toBe("$ErrorActionPreference='Stop';Start-Process 'C:\\Users\\o''brien\\report.html'"); + }); + + it("falls back to C:\\Windows for PowerShell when SystemRoot is unset, and to the bare name when absent", () => { setPlatform("win32"); const originalSystemRoot = process.env.SystemRoot; const originalSystemRootLower = process.env.SYSTEMROOT; delete process.env.SystemRoot; delete process.env.SYSTEMROOT; try { + const defaultPath = "C:\\Windows\\System32\\WindowsPowerShell\\v1.0\\powershell.exe"; + const existsSpy = vi.spyOn(fs, "existsSync").mockImplementation(candidate => candidate === defaultPath); const spawnCalls: SpawnCall[] = []; spySpawn(spawnCalls); openPath("https://example.com"); + expect(spawnCalls[0]?.cmd[0]).toBe(defaultPath); - expect(spawnCalls).toHaveLength(1); - expect(spawnCalls[0]?.cmd[0]).toBe("C:\\Windows\\System32\\rundll32.exe"); + // Exotic layout: resolved path missing → bare name via PATH. + existsSpy.mockReturnValue(false); + openPath("https://example.com"); + expect(spawnCalls[1]?.cmd[0]).toBe("powershell.exe"); } finally { if (originalSystemRoot !== undefined) process.env.SystemRoot = originalSystemRoot; if (originalSystemRootLower !== undefined) process.env.SYSTEMROOT = originalSystemRootLower; } }); + + it("logs when the opener exits non-zero so Start-Process failures are diagnosable", async () => { + setPlatform("win32"); + vi.spyOn(fs, "existsSync").mockReturnValue(true); + const warnSpy = vi.spyOn(piUtils.logger, "warn").mockImplementation((() => {}) as never); + const failing = { + pid: 1, + exited: Promise.resolve(1), + kill: () => true, + } as unknown as Subprocess; + vi.spyOn(Bun, "spawn").mockImplementation(() => failing); + + openPath("https://example.com"); + await failing.exited; + await Promise.resolve(); + + expect(warnSpy).toHaveBeenCalledWith( + "External opener exited with non-zero status", + expect.objectContaining({ exitCode: 1 }), + ); + }); }); diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index c8c4ffef6..cec936333 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -5,6 +5,7 @@ ### Fixed - Fixed the native build script failing to locate the `@napi-rs/cli` `napi` binary on Windows because the `PATH` lookup joined entries with a Unix `:` separator instead of the platform delimiter (`path.delimiter`). +- Fixed a Windows regression where an abnormal `omp` exit or bash cancellation could `TerminateProcess` unrelated `pwsh.exe` / `powershell.exe` sessions (including other Cursor terminal tabs). `SpawnRegistry` stored only the raw pid of each brush-spawned child and re-opened it via `Process::from_pid` at cancellation time; between those two moments Windows could recycle a freed pid onto an unrelated PowerShell, and `signal_tree` then walked the wrong subtree via Toolhelp. The observer now pins a stable `Process` handle at spawn time — on Windows the open handle keeps the pid slot reserved, on Linux the pidfd carries identity, on macOS the `(pid, start_time)` triple detects impersonation — so cancellation can only reach children this run actually launched. The registry sweeps exited entries once the recorded set crosses a small threshold so a long bash loop of short external commands cannot pin one owned OS handle per historical spawn. ([#4605](https://github.com/can1357/oh-my-pi/issues/4605)) ## [16.3.6] - 2026-07-04 diff --git a/packages/tui/CHANGELOG.md b/packages/tui/CHANGELOG.md index c35862b89..df52f0419 100644 --- a/packages/tui/CHANGELOG.md +++ b/packages/tui/CHANGELOG.md @@ -5,6 +5,14 @@ ### Added - Added optional right-border scrollbar to the `Editor` component (`setScrollbarVisible`): shows a thumb glyph on the right border when content overflows `maxHeight`, enabling scrollable multi-line editors (e.g. advisor instructions) without losing the submit hint off-screen. +### Fixed + +- Fixed selector rendering when a legacy theme omits symbol settings by falling back to an ASCII cursor instead of crashing ([#4745](https://github.com/can1357/oh-my-pi/issues/4745)). +- Kept slash command autocomplete rows compact by truncating descriptions instead of wrapping them into multi-line blocks. +- Fixed mid-prompt skill autocomplete so Tab and Enter accept the highlighted `/skill:` suggestion and Backspace dismisses the popup immediately after removing the triggering slash ([#4619](https://github.com/can1357/oh-my-pi/issues/4619)). +- Fixed submitted slash-command arguments treating `@` file-reference tokens as prompt-composer autocomplete triggers when the command does not define argument completions. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600)) +- Fixed box-drawing tree lines (`├── item` — directory layouts, decision trees) in prose shearing apart when they wrap: continuation rows now hang under the node text with ancestor rails carried through (`├` → `│`, `└` → blank) instead of restarting at column 0. Applies to prose paragraphs (including inside blockquotes) only when a line with a branch-connector prefix (`├──`, `└─`, …) actually overflows; fitting lines, non-tree prose, and code blocks render byte-for-byte as before. + ## [16.3.10] - 2026-07-06 ### Fixed diff --git a/packages/tui/src/autocomplete.ts b/packages/tui/src/autocomplete.ts index 99c111aed..df6b34b1e 100644 --- a/packages/tui/src/autocomplete.ts +++ b/packages/tui/src/autocomplete.ts @@ -276,7 +276,7 @@ function commandMatchesNameOrAlias(cmd: CommandEntry, commandName: string): bool return getCommandAliases(cmd).includes(commandName); } -function scoreCommandTextMatch(lowerPrefix: string, lowerTarget: string): number { +export function scoreCommandTextMatch(lowerPrefix: string, lowerTarget: string): number { if (lowerPrefix.length === 0) return 1; if (lowerPrefix === lowerTarget) return 1000; // Flat score for every prefix match so same-prefix commands keep registry diff --git a/packages/tui/src/components/editor.ts b/packages/tui/src/components/editor.ts index 1d187b1d4..ecf0d37e0 100644 --- a/packages/tui/src/components/editor.ts +++ b/packages/tui/src/components/editor.ts @@ -3,6 +3,7 @@ import { type AutocompleteProvider, findLeadingSlashCommandStart, findTrailingSlashCommandStart, + scoreCommandTextMatch, } from "../autocomplete"; import { BracketedPasteHandler, decodeReencodedPasteControls } from "../bracketed-paste"; import { getKeybindings, type KeybindingsManager } from "../keybindings"; @@ -21,7 +22,7 @@ import { truncateToWidth, visibleWidth, } from "../utils"; -import { SelectList, type SelectListLayoutOptions, type SelectListTheme } from "./select-list"; +import { type SelectItem, SelectList, type SelectListLayoutOptions, type SelectListTheme } from "./select-list"; const AUTOCOMPLETE_SELECT_LIST_LAYOUT: SelectListLayoutOptions = { overflowSearch: false, @@ -31,7 +32,6 @@ const SLASH_COMMAND_SELECT_LIST_LAYOUT: SelectListLayoutOptions = { minPrimaryColumnWidth: 12, maxPrimaryColumnWidth: 32, overflowSearch: false, - wrapDescription: true, }; function sanitizeLoadedText(text: string): string { @@ -1148,16 +1148,16 @@ export class Editor implements Component, Focusable { // If Tab was pressed, always apply the selection if (kb.matches(data, "tui.input.tab")) { + const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to buffer edits since last refresh // (destructive keys or paste can outrun the debounced update). const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; const currentTextBeforeCursor = currentLine.slice(0, this.#state.cursorCol); - if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor)) { + if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor, selected)) { // Autocomplete is stale - silently cancel; Tab has no fallback action here. this.#cancelAutocomplete(); return; } - const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const shouldChainSlashCommandAutocomplete = this.#isSlashCommandNameAutocompleteSelection(); const result = this.#autocompleteProvider.applyCompletion( @@ -1195,14 +1195,14 @@ export class Editor implements Component, Focusable { findLeadingSlashCommandStart(this.#autocompletePrefix) !== null && !this.#selectedCompletionIsPath() ) { + const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to debounce const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; const currentTextBeforeCursor = currentLine.slice(0, this.#state.cursorCol); - if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor)) { + if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor, selected)) { // Autocomplete is stale - cancel and fall through to normal submission this.#cancelAutocomplete(); } else { - const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const result = this.#autocompleteProvider.applyCompletion( this.#state.lines, @@ -1223,14 +1223,14 @@ export class Editor implements Component, Focusable { } // If Enter was pressed on a file path, apply completion else if (kb.matches(data, "tui.input.submit") || data === "\n") { + const selected = this.#autocompleteList.getSelectedItem(); // Check for stale autocomplete state due to buffer edits since last refresh. const currentLine = this.#state.lines[this.#state.cursorLine] ?? ""; const currentTextBeforeCursor = currentLine.slice(0, this.#state.cursorCol); - if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor)) { + if (!this.#autocompletePrefixMatchesCursorText(currentTextBeforeCursor, selected)) { // Autocomplete is stale - cancel and fall through to normal submission this.#cancelAutocomplete(); } else { - const selected = this.#autocompleteList.getSelectedItem(); if (selected && this.#autocompleteProvider) { const result = this.#autocompleteProvider.applyCompletion( this.#state.lines, @@ -2099,8 +2099,15 @@ export class Editor implements Component, Focusable { this.#resetKillSequence(); this.#recordUndoState(); + let removedMidPromptSlashTrigger = false; + if (this.#state.cursorCol > 0) { const line = this.#state.lines[this.#state.cursorLine] || ""; + const textBeforeCursor = line.slice(0, this.#state.cursorCol); + const trailingSlashStart = findTrailingSlashCommandStart(textBeforeCursor); + removedMidPromptSlashTrigger = + trailingSlashStart === this.#state.cursorCol - 1 && + (!this.#hasOnlyWhitespaceBeforeCursorLine() || textBeforeCursor.slice(0, trailingSlashStart).trim() !== ""); // An atomic placeholder token (image/paste marker) deletes as a unit, so a single // backspace never leaves a half-eaten `[Paste #1, +30 lines` behind as stray text. const token = this.#atomicTokenAt(line, this.#state.cursorCol - 1); @@ -2140,7 +2147,12 @@ export class Editor implements Component, Focusable { // Update or re-trigger autocomplete after backspace if (this.#autocompleteState) { - this.#debouncedUpdateAutocomplete(); + if (removedMidPromptSlashTrigger) { + this.#cancelAutocomplete(); + this.onAutocompleteUpdate?.(); + } else { + this.#debouncedUpdateAutocomplete(); + } } else { // If autocomplete was cancelled (no matches), re-trigger if we're in a completable context const currentLine = this.#state.lines[this.#state.cursorLine] || ""; @@ -2910,13 +2922,36 @@ export class Editor implements Component, Focusable { * engages for command-shaped selections: absolute-path completions (`/tmp/fo` * via the no-command-match fall-through) share the leading-slash prefix shape * but must use the live-suffix path rule so the apply slice stays anchored. + * - Mid-prompt skill branch re-anchors when the popup item is a skill and the + * current text still ends in a trailing slash token, matching the provider's + * mid-prompt replacement branch. * - `@`-file branch re-anchors via `#extractAtPrefix`; safe when the current text * still ends in a whitespace-anchored `@`. * - Everything else is stale — accepting it would corrupt the buffer (issue #4295). */ - #autocompletePrefixMatchesCursorText(currentTextBeforeCursor: string): boolean { + #autocompletePrefixMatchesCursorText(currentTextBeforeCursor: string, item?: SelectItem | null): boolean { if (currentTextBeforeCursor === this.#autocompletePrefix) return true; + if (item?.value.startsWith("skill:") && findTrailingSlashCommandStart(this.#autocompletePrefix) !== null) { + const currentTrailingStart = findTrailingSlashCommandStart(currentTextBeforeCursor); + if (currentTrailingStart !== null) { + const token = currentTextBeforeCursor.slice(currentTrailingStart); + if (!token.includes(" ") && !token.slice(1).includes("/")) { + // Guard the timing window where the popup was built for an earlier + // query (e.g. bare `/`) and the user typed further characters before + // the 100 ms debounced refresh fired: accept the stale skill only + // when the current query would still surface it. `tmp` after a bare + // slash therefore falls through to file completion instead of + // rewriting the user's `/tmp` to `/skill:…`. + const lowerToken = token.slice(1).toLowerCase(); + if (scoreCommandTextMatch(lowerToken, item.value.toLowerCase()) > 0) return true; + if (item.description && scoreCommandTextMatch(lowerToken, item.description.toLowerCase()) > 0) + return true; + } + } + return false; + } + if (findLeadingSlashCommandStart(this.#autocompletePrefix) !== null && !this.#selectedCompletionIsPath()) { const currentLeadingStart = findLeadingSlashCommandStart(currentTextBeforeCursor); if (currentLeadingStart !== null) { diff --git a/packages/tui/src/components/markdown.ts b/packages/tui/src/components/markdown.ts index b6eee348c..e81cd52aa 100644 --- a/packages/tui/src/components/markdown.ts +++ b/packages/tui/src/components/markdown.ts @@ -251,6 +251,155 @@ function splitTerminalLines(text: string): string[] { return lines; } +// --------------------------------------------------------------------------- +// Tree-guide hanging wrap +// +// Models routinely emit box-drawing trees ("├── item") inside plain +// paragraphs — directory layouts, decision trees. The lexer sees those lines +// as ordinary prose, so the generic wrap pass restarts wrapped continuations +// at column 0 and visually shears the tree apart (doubly fast for CJK text, +// where every glyph is two cells wide). Mirror the guide semantics of +// `tree(1)` / rich.tree instead: wrap the node text within the cells that +// remain after the guide prefix, and indent every continuation row under the +// node text — branch glyphs swap to their pass-through form (`├` → `│`, +// `└` → blank) so the rails of still-open ancestors stay visually joined. +// --------------------------------------------------------------------------- + +/** Continuation glyph for each guide character a tree prefix may contain. */ +const TREE_GUIDE_CONTINUATION: Record = { + "│": "│", + "┃": "┃", + "║": "║", + "├": "│", + "┣": "┃", + "╠": "║", + "└": " ", + "┗": " ", + "╚": " ", + "╰": " ", + "─": " ", + "━": " ", + "═": " ", + " ": " ", +}; + +/** Cheap pre-gate: any guide glyph at all. The structural test is TREE_BRANCH_CONNECTOR_RE. */ +const TREE_GUIDE_ANCHOR_RE = /[│┃║├┣╠└┗╚╰]/; + +/** + * A prefix qualifies as tree-shaped only when a branch/corner glyph is + * immediately followed by a horizontal connector (`├──`, `└─`, `╰──`, …). + * A lone rail or branch glyph used as prose ("│ is the Unicode vertical box + * drawing glyph…") never qualifies, so such paragraphs keep the plain wrap. + */ +const TREE_BRANCH_CONNECTOR_RE = /[├┣╠└┗╚╰][─━═]/; + +/** Below this many content cells a hanging wrap degenerates; keep the plain wrap. */ +const MIN_TREE_CONTENT_WIDTH = 8; + +const SGR_SEQUENCE_STICKY = /\x1b\[[0-9;:]*m/y; +const SGR_SEQUENCE_GLOBAL = /\x1b\[[0-9;:]*m/g; + +/** + * Everything before the last full SGR reset is dead state — drop it so the + * re-played `carry` stays bounded by the paragraph's live style run instead + * of its whole code history. + */ +function compactSgrCarry(carry: string): string { + const shortReset = carry.lastIndexOf("\x1b[m"); + const longReset = carry.lastIndexOf("\x1b[0m"); + const cut = Math.max(shortReset === -1 ? -1 : shortReset + 3, longReset === -1 ? -1 : longReset + 4); + return cut === -1 ? carry : carry.slice(cut); +} + +interface TreeGuidePrefix { + /** Index of the first char past the guide run (start of the node text). */ + end: number; + /** SGR sequences interleaved with the guides, in order (zero visible width). */ + codes: string; + /** Guide characters with SGR stripped, exactly as they appear on screen. */ + guides: string; +} + +/** + * Match the leading box-drawing guide run of a rendered line (e.g. `│ ├── `), + * tolerating interleaved SGR styling. Returns undefined unless the run + * contains a branch glyph joined to a horizontal connector and node text + * follows, so dash art, indented prose, and lone glyphs used as prose are + * never treated as a tree. + */ +function matchTreeGuidePrefix(line: string): TreeGuidePrefix | undefined { + let codes = ""; + let guides = ""; + let i = 0; + while (i < line.length) { + if (line.charCodeAt(i) === 0x1b) { + SGR_SEQUENCE_STICKY.lastIndex = i; + const match = SGR_SEQUENCE_STICKY.exec(line); + if (!match) break; + codes += match[0]; + i = SGR_SEQUENCE_STICKY.lastIndex; + continue; + } + const char = line[i]!; + if (!(char in TREE_GUIDE_CONTINUATION)) break; + guides += char; + i++; + } + if (i >= line.length || !TREE_BRANCH_CONNECTOR_RE.test(guides)) return undefined; + return { end: i, codes, guides }; +} + +/** + * Hanging wrap for box-drawing tree lines inside prose block text. + * + * Returns undefined when no line needs the treatment, so paragraphs without + * overflowing tree lines keep their exact current render. When a paragraph + * does hang, its lines are returned pre-split and style-self-contained: the + * SGR state open at each line start is re-played onto that line (`carry`), + * because the caller's wrap pass — which normally carries SGR state across + * the newlines of a single entry — no longer sees them as one entry. + */ +function hangWrapTreeGuideLines(text: string, width: number): string[] | undefined { + if (width < MIN_TREE_CONTENT_WIDTH || !TREE_GUIDE_ANCHOR_RE.test(text)) return undefined; + + const sourceLines = text.split("\n"); + const hangs = (line: string): TreeGuidePrefix | undefined => { + if (visibleWidth(line) <= width) return undefined; + const prefix = matchTreeGuidePrefix(line); + if (!prefix) return undefined; + if (width - visibleWidth(prefix.guides) < MIN_TREE_CONTENT_WIDTH) return undefined; + return prefix; + }; + if (!sourceLines.some(line => hangs(line) !== undefined)) return undefined; + + const out: string[] = []; + let carry = ""; + for (const line of sourceLines) { + const prefix = hangs(line); + if (!prefix) { + out.push(carry ? carry + line : line); + carry = compactSgrCarry(carry + (line.match(SGR_SEQUENCE_GLOBAL)?.join("") ?? "")); + continue; + } + // Re-play the SGR state ahead of the node text so the wrapper carries + // it onto every continuation row; the codes are zero-width, so measured + // row widths are unaffected. + const activeCodes = carry + prefix.codes; + const rows = wrapTextWithAnsi(activeCodes + line.slice(prefix.end), width - visibleWidth(prefix.guides)); + let hang = ""; + for (const guide of prefix.guides) hang += TREE_GUIDE_CONTINUATION[guide] ?? " "; + const hangShortfall = visibleWidth(prefix.guides) - visibleWidth(hang); + if (hangShortfall > 0) hang += padding(hangShortfall); + out.push(carry + line.slice(0, prefix.end) + rows[0]!.slice(activeCodes.length)); + for (let i = 1; i < rows.length; i++) { + out.push(activeCodes + hang + rows[i]!); + } + carry = compactSgrCarry(carry + (line.match(SGR_SEQUENCE_GLOBAL)?.join("") ?? "")); + } + return out; +} + class StrictStrikethroughTokenizer extends Tokenizer { override del(src: string): Tokens.Del | undefined { const match = STRICT_STRIKETHROUGH_REGEX.exec(src); @@ -1383,7 +1532,7 @@ export class Markdown implements Component { break; } const paragraphText = this.#renderInlineTokens(token.tokens || [], styleContext); - lines.push(paragraphText); + lines.push(...(hangWrapTreeGuideLines(paragraphText, width) ?? [paragraphText])); // Don't add spacing if next token is space or list if (nextTokenType && nextTokenType !== "list" && nextTokenType !== "space") { lines.push(""); diff --git a/packages/tui/src/components/select-list.ts b/packages/tui/src/components/select-list.ts index 06b151a42..651de51b2 100644 --- a/packages/tui/src/components/select-list.ts +++ b/packages/tui/src/components/select-list.ts @@ -12,6 +12,8 @@ const DEFAULT_PRIMARY_COLUMN_WIDTH = 32; const PRIMARY_COLUMN_GAP = 2; const MIN_DESCRIPTION_WIDTH = 10; +const DEFAULT_CURSOR_SYMBOL = ">"; + function sanitizeSingleLine(text: string): string { return replaceTabs(text) .replace(/[\r\n]+/g, " ") @@ -372,9 +374,8 @@ export class SelectList implements Component, MouseRoutable { width: number, primaryColumnWidth: number, ): SelectItemLayout { - const prefix = isSelected - ? `${this.theme.symbols.cursor} ` - : padding(visibleWidth(this.theme.symbols.cursor) + 1); + const cursor = this.theme.symbols?.cursor ?? DEFAULT_CURSOR_SYMBOL; + const prefix = isSelected ? `${cursor} ` : padding(visibleWidth(cursor) + 1); const prefixWidth = visibleWidth(prefix); const descriptionSingleLine = item.description ? sanitizeSingleLine(item.description) : undefined; diff --git a/packages/tui/src/tui.ts b/packages/tui/src/tui.ts index bcf72f3b1..0d44beb4a 100644 --- a/packages/tui/src/tui.ts +++ b/packages/tui/src/tui.ts @@ -503,6 +503,12 @@ export class Container implements Component { this.#memoLines = undefined; } + /** Dispose every child, then detach it from this container. */ + disposeChildren(): void { + this.dispose(); + this.clear(); + } + invalidate(): void { this.#memoLines = undefined; for (const child of this.children) { diff --git a/packages/tui/test/container-dispose.test.ts b/packages/tui/test/container-dispose.test.ts index 7c497a2f6..a5d1da224 100644 --- a/packages/tui/test/container-dispose.test.ts +++ b/packages/tui/test/container-dispose.test.ts @@ -27,4 +27,15 @@ describe("Container.dispose", () => { outer.dispose(); expect(leafDisposed).toBe(1); }); + + it("disposeChildren disposes children and detaches them", () => { + let disposed = 0; + const container = new Container(); + container.addChild(inert(() => disposed++)); + + container.disposeChildren(); + + expect(disposed).toBe(1); + expect(container.children).toEqual([]); + }); }); diff --git a/packages/tui/test/editor-autocomplete-actions.test.ts b/packages/tui/test/editor-autocomplete-actions.test.ts index 626d82acf..35b5becfa 100644 --- a/packages/tui/test/editor-autocomplete-actions.test.ts +++ b/packages/tui/test/editor-autocomplete-actions.test.ts @@ -205,6 +205,94 @@ class SyncSlashProvider implements AutocompleteProvider { } describe("Editor Enter handler sync slash completion", () => { + const skillCommands = [ + { name: "skill:security-scan", description: "Security scan" }, + { name: "model", description: "Switch model" }, + ]; + + function createSkillEditor(): Editor { + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider(new CombinedAutocompleteProvider(skillCommands, "/tmp")); + return editor; + } + + async function openMidPromptSkillAutocomplete(editor: Editor, prose: string): Promise { + editor.handleInput(prose); + editor.handleInput("/"); + await Promise.resolve(); + + expect(editor.getText()).toBe(`${prose}/`); + expect(editor.isShowingAutocomplete()).toBe(true); + } + + it("accepts a bare mid-prompt skill slash with Tab without replacing prose", async () => { + const editor = createSkillEditor(); + + await openMidPromptSkillAutocomplete(editor, "run a "); + editor.handleInput("\t"); + + expect(editor.getText()).toBe("run a /skill:security-scan "); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + + it("accepts a bare mid-prompt skill slash with Enter and submits the completed prompt", async () => { + const editor = createSkillEditor(); + let submitted = ""; + editor.onSubmit = text => { + submitted = text; + }; + + await openMidPromptSkillAutocomplete(editor, "run a "); + editor.handleInput("\r"); + + expect(submitted).toBe("run a /skill:security-scan"); + expect(editor.getText()).toBe(""); + }); + + it("hides mid-prompt skill autocomplete immediately when Backspace removes the slash", async () => { + const editor = createSkillEditor(); + + await openMidPromptSkillAutocomplete(editor, "run a "); + editor.handleInput("\x7f"); + + expect(editor.getText()).toBe("run a "); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + + it("does not apply a stale mid-prompt skill suggestion when the live token stops matching", async () => { + const editor = createSkillEditor(); + + await openMidPromptSkillAutocomplete(editor, "see "); + // Race the 100 ms debounce: type a non-skill token before the popup refreshes. + editor.handleInput("tmp"); + editor.handleInput("\t"); + + // The stale `skill:security-scan` popup must not rewrite `/tmp` to `/skill:…`. + expect(editor.getText()).toBe("see /tmp"); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + + it("accepts a stale mid-prompt skill suggestion when the live token still matches its description", async () => { + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider( + [ + { name: "skill:hardening", description: "Security scan" }, + { name: "model", description: "Switch model" }, + ], + "/tmp", + ), + ); + + await openMidPromptSkillAutocomplete(editor, "run a "); + // Race the 100 ms debounce: type a query that matches only the skill description. + editor.handleInput("scan"); + editor.handleInput("\t"); + + expect(editor.getText()).toBe("run a /skill:hardening "); + expect(editor.isShowingAutocomplete()).toBe(false); + }); + it("opens mid-prompt skill autocomplete and inserts the skill token without wiping the draft on Tab", async () => { const editor = new Editor(defaultEditorTheme); editor.setAutocompleteProvider( diff --git a/packages/tui/test/editor.test.ts b/packages/tui/test/editor.test.ts index d451c758e..1e0b27dfb 100644 --- a/packages/tui/test/editor.test.ts +++ b/packages/tui/test/editor.test.ts @@ -309,6 +309,34 @@ describe("Editor component", () => { await expect(promise).resolves.toBe("/"); }); + it("renders slash-command suggestions as compact item rows", async () => { + const editor = new Editor(defaultEditorTheme); + editor.setAutocompleteMaxVisible(10); + const longDescription = + "Plan and execute non-trivial architectural improvements to the codebase without turning each slash command into a multi-line block."; + editor.setAutocompleteProvider( + new CombinedAutocompleteProvider( + Array.from({ length: 12 }, (_, i) => ({ + name: `cmd${i}`, + description: longDescription, + })), + "/tmp", + ), + ); + + const { promise: autocompleteUpdated, resolve: resolveAutocompleteUpdated } = Promise.withResolvers(); + editor.onAutocompleteUpdate = resolveAutocompleteUpdated; + + editor.handleInput("/"); + await autocompleteUpdated; + + const rendered = editor.render(80).map(line => stripVTControlCharacters(line)); + for (let i = 0; i < 10; i += 1) { + expect(rendered.some(line => line.includes(`cmd${i}`))).toBe(true); + } + expect(rendered.some(line => line.includes("cmd10"))).toBe(false); + }); + it("triggers file-reference autocomplete when typing at-sign", async () => { const editor = new Editor(defaultEditorTheme); const { promise, resolve } = Promise.withResolvers(); diff --git a/packages/tui/test/loader.test.ts b/packages/tui/test/loader.test.ts index e200e1b24..e5e7967e8 100644 --- a/packages/tui/test/loader.test.ts +++ b/packages/tui/test/loader.test.ts @@ -1,5 +1,5 @@ import { afterEach, describe, expect, it, setSystemTime, spyOn, vi } from "bun:test"; -import { TUI } from "@oh-my-pi/pi-tui"; +import { Container, TUI } from "@oh-my-pi/pi-tui"; import { Loader, type LoaderMessageColorFn } from "@oh-my-pi/pi-tui/components/loader"; import { visibleWidth } from "@oh-my-pi/pi-tui/utils"; import { VirtualTerminal } from "./virtual-terminal"; @@ -144,4 +144,28 @@ describe("Loader component", () => { expect(() => loader.dispose()).not.toThrow(); // idempotent tui.stop(); }); + + it("container disposeChildren stops detached loader repaints", () => { + vi.useFakeTimers(); + const term = new VirtualTerminal(20, 4); + const tui = new TUI(term); + const spy = spyOn(tui, "requestComponentRender"); + const container = new Container(); + const loader = new Loader( + tui, + text => text, + text => text, + "Checking", + ["0", "1"], + ); + container.addChild(loader); + const afterMount = spy.mock.calls.length; + + container.disposeChildren(); + vi.advanceTimersByTime(200); + + expect(spy.mock.calls.length).toBe(afterMount); + expect(container.children).toEqual([]); + tui.stop(); + }); }); diff --git a/packages/tui/test/markdown-tree-wrap.test.ts b/packages/tui/test/markdown-tree-wrap.test.ts new file mode 100644 index 000000000..16097efcc --- /dev/null +++ b/packages/tui/test/markdown-tree-wrap.test.ts @@ -0,0 +1,318 @@ +import { describe, expect, it } from "bun:test"; +import { stripVTControlCharacters } from "node:util"; +import { Markdown } from "@oh-my-pi/pi-tui/components/markdown"; +import { visibleWidth, wrapTextWithAnsi } from "@oh-my-pi/pi-tui/utils"; +import { Chalk } from "chalk"; +import { defaultMarkdownTheme } from "./test-themes.js"; + +const WIDTH = 40; + +function renderRaw(text: string, width = WIDTH): readonly string[] { + return new Markdown(text, 0, 0, defaultMarkdownTheme).render(width); +} + +/** Rendered rows as plain text, right-padding stripped (rows are padded to full width). */ +function renderPlain(text: string, width = WIDTH): string[] { + return renderRaw(text, width).map(line => stripVTControlCharacters(line).trimEnd()); +} + +describe("Markdown tree-guide hanging wrap", () => { + it("hangs an overflowing '├── ' node under the node-text column with double-width Korean text", () => { + const node = "가나다라 마바사아 자차카타 파하가나 다라마바 사자차카"; // 6 words x 8 cells + const raw = renderRaw(`├── ${node}`); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("├── 가나다라")).toBeTruthy(); + + for (const line of plain.slice(1)) { + // Exactly `│` + 3 spaces: the node text column is cell 4, so the + // continuation text must begin right there — not a cell earlier or later. + expect(line.startsWith("│ ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + } + for (const line of raw) { + expect(visibleWidth(line)).toBeLessThanOrEqual(WIDTH); + } + + // No glyph may be lost or duplicated by the wrap (spaces are consumed at + // break points, so compare with spaces removed). + const rejoined = [plain[0]!.slice("├── ".length), ...plain.slice(1).map(line => line.slice("│ ".length))] + .join("") + .replace(/ /g, ""); + expect(rejoined).toBe(node.replace(/ /g, "")); + }); + + it("keeps the outer rail and releases the corner for a nested '│ └── ' node", () => { + const plain = renderPlain("│ └── delta echo foxtrot golf hotel india juliet"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("│ └── delta")).toBeTruthy(); + for (const line of plain.slice(1)) { + // `│` stays (outer level still open), `└──` releases to spaces. + expect(line.startsWith("│ ")).toBeTruthy(); + expect(line[8]).not.toBe(" "); + } + }); + + it("releases a last-child '└── ' node to pure spaces with no rail on continuations", () => { + const plain = renderPlain("└── alpha bravo charlie delta echo foxtrot golf hotel"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("└── alpha")).toBeTruthy(); + for (const line of plain.slice(1)) { + expect(line.startsWith(" ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + expect(line.includes("│")).toBeFalsy(); + } + }); + + it("treats the rounded corner '╰── ' like '└── '", () => { + const plain = renderPlain("╰── alpha bravo charlie delta echo foxtrot golf hotel"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[0]!.startsWith("╰── alpha")).toBeTruthy(); + for (const line of plain.slice(1)) { + expect(line.startsWith(" ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + expect(line.includes("│")).toBeFalsy(); + } + }); + + it("leaves a non-tree paragraph byte-identical to the plain wrap", () => { + const text = "alpha bravo charlie delta echo foxtrot golf hotel india juliet kilo lima"; + const plain = renderPlain(text); + + // Differential against the generic wrapper: the tree feature must not + // have touched this paragraph at all. + const expected = wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd()); + expect(plain).toEqual(expected); + + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("does not treat a dash-only '── ' start as a tree", () => { + const text = "── alpha bravo charlie delta echo foxtrot golf hotel india"; + const plain = renderPlain(text); + + const expected = wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd()); + expect(plain).toEqual(expected); + + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + // Flush at column 0: no injected hang, no rail. + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("renders a fitting tree line-for-line unchanged", () => { + const plain = renderPlain("├── alpha\n│ └── beta\n└── gamma"); + + expect(plain).toEqual(["├── alpha", "│ └── beta", "└── gamma"]); + }); + + it("keeps the old column-0 wrap for '├── ' lines inside fenced code blocks", () => { + const codeLine = "├── alpha bravo charlie delta echo foxtrot golf hotel india"; + const raw = renderRaw(`\`\`\`\n${codeLine}\n\`\`\``); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain[0]).toBe("```"); + expect(plain[plain.length - 1]).toBe("```"); + + const treeRow = plain.findIndex(line => line.includes("├──")); + expect(treeRow).toBeGreaterThan(0); + // The code line overflows, so a continuation row exists before the + // closing fence — and it starts flush at column 0, no hanging prefix. + const continuation = plain[treeRow + 1]!; + expect(treeRow + 1).toBeLessThan(plain.length - 1); + expect(continuation.length).toBeGreaterThan(0); + expect(continuation[0]).not.toBe(" "); + expect(continuation[0]).not.toBe("│"); + + for (const line of raw) { + expect(visibleWidth(line)).toBeLessThanOrEqual(WIDTH); + } + }); + + it("carries an open bold span onto the continuation row", () => { + const raw = renderRaw("├── aaaa bbbb **cccc dddd eeee ffff gggg hhhh**"); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + // Row 0 holds 36 node cells ("aaaa bbbb cccc dddd eeee ffff gggg"), + // so "hhhh" — inside the bold span — lands on the continuation row. + expect(plain.length).toBeGreaterThanOrEqual(2); + expect(plain[1]!.startsWith("│ hhhh")).toBeTruthy(); + + const continuation = raw[1]!; + const boldOpen = continuation.indexOf("\x1b[1m"); + expect(boldOpen).toBeGreaterThanOrEqual(0); + expect(boldOpen).toBeLessThan(continuation.indexOf("hhhh")); + }); + + it("hangs inside a blockquote, after the quote border", () => { + const plain = renderPlain("> ├── alpha bravo charlie delta echo foxtrot golf hotel india"); + + expect(plain.length).toBeGreaterThanOrEqual(2); + // Quote border symbol, border gap, then the tree prefix. + expect(plain[0]!.startsWith("│ ├── alpha")).toBeTruthy(); + const continuations = plain.slice(1).filter(line => line !== ""); // drop trailing spacing rows + expect(continuations.length).toBeGreaterThanOrEqual(1); + for (const line of continuations) { + expect(line.startsWith("│ │ ")).toBeTruthy(); + expect(line[6]).not.toBe(" "); + } + }); + + describe("detection strictness and carry edge cases", () => { + const SGR_RE = /\x1b\[[0-9;:]*m/g; + + it("keeps plain wrap for prose starting with a lone '│ ' rail glyph", () => { + const text = "│ is the Unicode vertical box drawing glyph used for rails in terminal trees"; + const raw = renderRaw(text); + + // Byte-identical to the generic wrapper: no branch+connector pair, + // so the tree feature must not inject rails or indent. + expect(raw.map(line => line.trimEnd())).toEqual(wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd())); + + const plain = renderPlain(text); + expect(plain.length).toBeGreaterThanOrEqual(2); // the paragraph really overflowed + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("keeps plain wrap for prose starting with '├ ' without a horizontal connector", () => { + const text = "├ marks a branch point in a tree diagram and has no horizontal connector here"; + const raw = renderRaw(text); + + expect(raw.map(line => line.trimEnd())).toEqual(wrapTextWithAnsi(text, WIDTH).map(line => line.trimEnd())); + + const plain = renderPlain(text); + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + }); + + it("falls back to plain wrap when fewer than 8 content cells remain after the prefix", () => { + const text = "├── alpha bravo charlie delta echo foxtrot golf hotel india"; + + // Width 10 leaves 10 - 4 = 6 content cells after the '├── ' prefix: + // below the minimum, so the hang degenerates and plain wrap wins. + const raw = renderRaw(text, 10); + expect(raw.map(line => line.trimEnd())).toEqual(wrapTextWithAnsi(text, 10).map(line => line.trimEnd())); + const plain = renderPlain(text, 10); + expect(plain.length).toBeGreaterThanOrEqual(2); + for (const line of plain.slice(1)) { + expect(line[0]).not.toBe(" "); + expect(line[0]).not.toBe("│"); + } + + // Width 12 leaves exactly 8 content cells — the boundary where the + // hanging wrap applies again. + const hung = renderPlain(text, 12); + expect(hung[0]).toBe("├── alpha"); + expect(hung.length).toBeGreaterThanOrEqual(2); + for (const line of hung.slice(1)) { + expect(line.startsWith("│ ")).toBeTruthy(); + expect(line[4]).not.toBe(" "); + } + }); + + it("replays SGR state opened on an earlier line onto a later hung line's continuation rows", () => { + // With a default text style, the renderer re-opens the default color + // after `**bold**` (the style prefix) and the soft break leaves that + // re-open unclosed — a style opened on line 1 that is still active + // when line 2 hangs. Line 1's bold open/close pair exists nowhere on + // line 2, so finding it ahead of the continuation text proves the + // carry was re-played rather than line 2's own codes. + const chalk = new Chalk({ level: 3 }); + const raw = new Markdown( + "aaa **bold**\n├── alpha bravo charlie delta echo foxtrot golf hotel india", + 0, + 0, + defaultMarkdownTheme, + { color: text => chalk.red(text) }, + ).render(WIDTH); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain.length).toBe(3); + expect(plain[1]!.startsWith("├── alpha")).toBeTruthy(); + expect(plain[2]!.startsWith("│ foxtrot")).toBeTruthy(); + + // Line 1 genuinely ends with an unclosed style: its last SGR is the + // default-color re-open, not a close. + const row0Codes = raw[0]!.trimEnd().match(SGR_RE)!; + expect(row0Codes[row0Codes.length - 1]).toBe("\x1b[31m"); + + // The continuation row starts with a zero-width SGR run that replays + // line 1's history (the carried bold pair) and nets out to the + // default color being open ahead of the visible text. + const continuation = raw[2]!; + const hangAt = continuation.indexOf("│ "); + expect(hangAt).toBeGreaterThan(0); + const replayed = continuation.slice(0, hangAt); + expect(replayed.replace(SGR_RE, "")).toBe(""); + expect(replayed).toContain("\x1b[1m"); + const replayedCodes = replayed.match(SGR_RE)!; + expect(replayedCodes[replayedCodes.length - 1]).toBe("\x1b[31m"); + }); + + it("re-renders byte-identically after a width round-trip (44 → 80 → 44)", () => { + const doc = "├── alpha bravo charlie delta echo foxtrot golf\n└── hotel india juliet kilo lima mike november"; + const md = new Markdown(doc, 0, 0, defaultMarkdownTheme); + + const first = [...md.render(44)]; + // Narrow render actually hung — the round-trip below is not vacuous. + expect(first.map(line => stripVTControlCharacters(line).trimEnd())).toEqual([ + "├── alpha bravo charlie delta echo foxtrot", + "│ golf", + "└── hotel india juliet kilo lima mike", + " november", + ]); + + // Wide render fits line-for-line: a genuinely different layout. + const wide = md.render(80).map(line => stripVTControlCharacters(line).trimEnd()); + expect(wide).toEqual([ + "├── alpha bravo charlie delta echo foxtrot golf", + "└── hotel india juliet kilo lima mike november", + ]); + + expect([...md.render(44)]).toEqual(first); + expect([...new Markdown(doc, 0, 0, defaultMarkdownTheme).render(44)]).toEqual(first); + }); + + it("does not leak styles opened before a full SGR reset onto later hung rows", () => { + // Raw ANSI in component input passes through marked byte-exact: line 1 + // opens bold, fully resets, then opens italic. Only the italic — the + // live style after the reset — may carry onto the hung line. + const raw = renderRaw( + "aaa \x1b[1mbold\x1b[0m\x1b[3mrest and filler\n├── alpha bravo charlie delta echo foxtrot golf hotel india", + ); + const plain = raw.map(line => stripVTControlCharacters(line).trimEnd()); + + expect(plain.length).toBe(3); + expect(plain[1]!.startsWith("├── alpha")).toBeTruthy(); + expect(plain[2]!.startsWith("│ foxtrot")).toBeTruthy(); + + for (const row of raw.slice(1)) { + expect(row).not.toContain("\x1b[1m"); // dead pre-reset style must not re-play + expect(row).not.toContain("\x1b[0m"); + } + // The live post-reset style carries onto the hung line and is the + // entire replayed run ahead of the continuation's hang glyphs. + expect(raw[1]!.startsWith("\x1b[3m├── ")).toBeTruthy(); + const continuation = raw[2]!; + const hangAt = continuation.indexOf("│ "); + expect(hangAt).toBeGreaterThan(0); + expect(continuation.slice(0, hangAt)).toBe("\x1b[3m"); + }); + }); +}); diff --git a/packages/tui/test/select-list.test.ts b/packages/tui/test/select-list.test.ts index f9b2128ed..4d5eefe78 100644 --- a/packages/tui/test/select-list.test.ts +++ b/packages/tui/test/select-list.test.ts @@ -1,5 +1,5 @@ import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import { SelectList } from "@oh-my-pi/pi-tui/components/select-list"; +import { SelectList, type SelectListTheme } from "@oh-my-pi/pi-tui/components/select-list"; import { KeybindingsManager, setKeybindings, TUI_KEYBINDINGS } from "@oh-my-pi/pi-tui/keybindings"; import type { SgrMouseEvent } from "@oh-my-pi/pi-tui/mouse"; import { visibleWidth } from "@oh-my-pi/pi-tui/utils"; @@ -78,6 +78,14 @@ describe("SelectList", () => { expect(rendered[0]).toContain("Line one Line two Line three"); }); + it("falls back to an ASCII cursor when a legacy theme omits symbols", () => { + const legacyTheme: SelectListTheme = { ...testTheme }; + Reflect.deleteProperty(legacyTheme, "symbols"); + const list = new SelectList([{ value: "run", label: "run" }], 1, legacyTheme); + + expect(list.render(40)).toEqual(["> run"]); + }); + it("keeps descriptions aligned when the primary text is truncated", () => { const items = [ { value: "short", label: "short", description: "short description" }, diff --git a/packages/utils/CHANGELOG.md b/packages/utils/CHANGELOG.md index 80ebafb3a..a6f89cf70 100644 --- a/packages/utils/CHANGELOG.md +++ b/packages/utils/CHANGELOG.md @@ -2,6 +2,14 @@ ## [Unreleased] +### Added + +- Added `postmortem.interceptUnhandledRejections()` to register interceptors consulted before an unhandled rejection tears the process down; a consuming interceptor (e.g. the JS eval runtime claiming rejections floated by user cell code) keeps the process alive and owns reporting. + +### Fixed + +- Fixed child shell environment filtering to drop launch-directory `.env.local` values that Bun auto-loaded before OMP starts command shells. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723)) + ## [16.3.10] - 2026-07-06 ### Added diff --git a/packages/utils/src/env.ts b/packages/utils/src/env.ts index b3337f92e..21619ae2d 100644 --- a/packages/utils/src/env.ts +++ b/packages/utils/src/env.ts @@ -53,6 +53,19 @@ export function filterProcessEnv(env: Record): Recor return result; } +/** Filters process env for child shells without launch-cwd `.env.local` values. */ +export function filterChildShellEnv( + env: Record, + cwd: string = process.cwd(), +): Record { + const result = filterProcessEnv(env); + const launchLocalEnv = parseEnvFile(path.join(cwd, ".env.local")); + for (const key in launchLocalEnv) { + if (result[key] === launchLocalEnv[key]) delete result[key]; + } + return result; +} + /** * Parses a .env file synchronously and extracts key-value string pairs. * Ignores lines that are empty or start with '#'. Trims whitespace. diff --git a/packages/utils/src/postmortem.ts b/packages/utils/src/postmortem.ts index 01d208e9e..42c37b42e 100644 --- a/packages/utils/src/postmortem.ts +++ b/packages/utils/src/postmortem.ts @@ -121,6 +121,24 @@ export function isExpectedCleanupError(reason: unknown): boolean { return false; } +/** + * Interceptors consulted by the global `unhandledRejection` handler before the + * fatal path. See {@link interceptUnhandledRejections}. + */ +const rejectionInterceptors = new Set<(reason: unknown) => boolean>(); + +/** + * Register an interceptor consulted before an unhandled rejection tears the + * process down. Return `true` to consume the rejection — the interceptor owns + * reporting and the process continues. Used by embedded script runtimes (JS + * eval cells) whose user code can float rejections the host must not die for. + * Returns an unregister function. + */ +export function interceptUnhandledRejections(interceptor: (reason: unknown) => boolean): () => void { + rejectionInterceptors.add(interceptor); + return () => rejectionInterceptors.delete(interceptor); +} + function formatFatalError(label: string, err: Error): string { const name = err.name || "Error"; const message = err.message || "(no message)"; @@ -172,6 +190,15 @@ if (isMainThread) { logger.warn("Ignoring expected cleanup rejection", { err }); return; } + for (const interceptor of rejectionInterceptors) { + try { + if (interceptor(reason)) return; + } catch (interceptorErr) { + logger.warn("Unhandled-rejection interceptor threw; continuing with fatal path", { + err: interceptorErr, + }); + } + } process.stderr.write(formatFatalError("Unhandled Rejection", err)); logger.error("Unhandled rejection", { err }); await runCleanup(Reason.UNHANDLED_REJECTION); diff --git a/packages/utils/src/procmgr.ts b/packages/utils/src/procmgr.ts index 3097900b1..5a4cb30b9 100644 --- a/packages/utils/src/procmgr.ts +++ b/packages/utils/src/procmgr.ts @@ -2,7 +2,7 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { Process, ProcessStatus } from "@oh-my-pi/pi-natives"; import type { Subprocess } from "bun"; -import { $env, filterProcessEnv } from "./env"; +import { $env, filterChildShellEnv } from "./env"; import { $which } from "./which"; export interface ShellConfig { @@ -31,7 +31,7 @@ export function isExecutable(path: string): boolean { function buildSpawnEnv(shell: string): Record { const noCI = $env.PI_BASH_NO_CI || $env.CLAUDE_BASH_NO_CI; return { - ...filterProcessEnv(Bun.env), + ...filterChildShellEnv(Bun.env), SHELL: shell, GIT_EDITOR: "true", GPG_TTY: "not a tty",