Merge remote-tracking branch 'upstream/main' into feat/error-notify
This commit is contained in:
+5
-3
@@ -184,9 +184,11 @@ RUN bun install --frozen-lockfile --ignore-scripts
|
||||
# hoisted node_modules that `bun install` just produced.
|
||||
COPY . /pi/
|
||||
|
||||
# Regenerate the docs index that `--ignore-scripts` skipped above. The root
|
||||
# package.json's `prepare` script normally handles this on a vanilla install.
|
||||
RUN bun --cwd=packages/coding-agent run gen:docs
|
||||
# Regenerate the docs index and tool views that `--ignore-scripts` skipped
|
||||
# above. The root package.json's `prepare` script normally handles these on a
|
||||
# vanilla install.
|
||||
RUN bun --cwd=packages/coding-agent run gen:docs \
|
||||
&& bun --cwd=/pi/packages/coding-agent run gen:tool-views
|
||||
|
||||
ENTRYPOINT ["/usr/bin/tini", "--", "/usr/local/bin/omp"]
|
||||
CMD ["--help"]
|
||||
|
||||
+344
-12
@@ -1571,6 +1571,11 @@ impl TerminationTargets {
|
||||
/// Record a pid. Duplicates are ignored. If the pid is alive, opens
|
||||
/// a stable [`Process`] reference so the descendant tree can be
|
||||
/// killed even if the original pid is reused later.
|
||||
///
|
||||
/// Prefer [`add_process`](Self::add_process) when the caller already holds a
|
||||
/// [`Process`] captured at spawn time: opening by pid here loses the
|
||||
/// original identity if the pid was recycled between the child exiting
|
||||
/// and this call.
|
||||
pub fn add_pid(&mut self, pid: i32) {
|
||||
if self.seen_pids.insert(pid)
|
||||
&& let Some(process) = Process::from_pid(pid)
|
||||
@@ -1579,6 +1584,17 @@ impl TerminationTargets {
|
||||
}
|
||||
}
|
||||
|
||||
/// Record a pre-pinned [`Process`] handle. Duplicates (by pid) are ignored.
|
||||
///
|
||||
/// This is the correct entry point when the caller captured the handle at
|
||||
/// spawn time — the handle already pins OS-level identity, so no `from_pid`
|
||||
/// re-open (and its PID-reuse race) is needed at cancellation time.
|
||||
pub fn add_process(&mut self, process: Process) {
|
||||
if self.seen_pids.insert(process.pid()) {
|
||||
self.processes.push(process);
|
||||
}
|
||||
}
|
||||
|
||||
/// True when no targets have been recorded.
|
||||
#[must_use]
|
||||
pub const fn is_empty(&self) -> bool {
|
||||
@@ -1599,10 +1615,19 @@ impl TerminationTargets {
|
||||
}
|
||||
|
||||
/// A single external child reported by the shell's spawn-observer hook.
|
||||
#[derive(Debug, Clone, Copy)]
|
||||
///
|
||||
/// `process` is captured *at spawn time* so its OS-level identity is pinned
|
||||
/// before the pid can be recycled. On Windows an open process handle keeps
|
||||
/// the pid reserved for the lifetime of the reference; on Linux the pidfd
|
||||
/// pins identity; on macOS the recorded `(pid, start_time)` triple detects
|
||||
/// impersonation. Storing only the raw pid and re-opening at cancellation
|
||||
/// time — as previous versions did — leaked kills onto unrelated processes
|
||||
/// that happened to acquire the recycled pid between the child exiting and
|
||||
/// the run being cancelled (issue #4605).
|
||||
#[derive(Clone)]
|
||||
struct SpawnedProcess {
|
||||
pid: i32,
|
||||
pgid: Option<i32>,
|
||||
process: Option<Process>,
|
||||
pgid: Option<i32>,
|
||||
}
|
||||
|
||||
/// Per-run record of the OS processes a single shell command launched,
|
||||
@@ -1613,12 +1638,43 @@ struct SpawnedProcess {
|
||||
/// host process: a run that cancelled would signal *any* descendant spawned
|
||||
/// after its baseline, including another run's children. Ownership is now
|
||||
/// explicit — only processes this run actually spawned are ever signalled.
|
||||
#[derive(Default)]
|
||||
struct RegistryState {
|
||||
spawned: Vec<SpawnedProcess>,
|
||||
/// The next `spawned.len()` at which `record` runs a sweep. Bounds sweep
|
||||
/// frequency when the live set stabilizes above the initial threshold:
|
||||
/// without this watermark, every subsequent `record` would find
|
||||
/// `len >= PRUNE_THRESHOLD` true and sweep on every spawn (O(n²) in a
|
||||
/// large-fan-out run like `for i in {1..1000}; do sleep 60 & done`). With
|
||||
/// it, the next sweep only fires once the vec has grown by another
|
||||
/// `PRUNE_THRESHOLD` entries since the previous sweep — restoring true
|
||||
/// amortized O(1) per spawn regardless of how many entries survive each
|
||||
/// sweep.
|
||||
next_sweep_at: usize,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
pub struct SpawnRegistry {
|
||||
spawned: Mutex<Vec<SpawnedProcess>>,
|
||||
state: Mutex<RegistryState>,
|
||||
}
|
||||
|
||||
impl SpawnRegistry {
|
||||
/// Amortized-cost threshold for opportunistic pruning of exited entries.
|
||||
///
|
||||
/// A shell run that spawns many short-lived external commands (e.g. a bash
|
||||
/// loop invoking a binary per iteration) would otherwise retain one owned
|
||||
/// process handle per spawn — a pidfd on Linux, a `HANDLE` on Windows — for
|
||||
/// the lifetime of the run, exhausting per-process FD/handle limits.
|
||||
///
|
||||
/// Each sweep costs `O(N)` (one non-blocking status probe per entry, plus
|
||||
/// a Toolhelp descendant walk on Windows for exited roots). The next sweep
|
||||
/// is scheduled `PRUNE_THRESHOLD` further records away — via the
|
||||
/// `next_sweep_at` watermark — so a run that keeps many concurrent
|
||||
/// long-lived children (`for i in {1..1000}; do sleep 60 & done`) does not
|
||||
/// sweep on every spawn just because the vec is already above threshold.
|
||||
/// Amortized cost per spawn stays `O(1)` regardless of the live-set size.
|
||||
const PRUNE_THRESHOLD: usize = 64;
|
||||
|
||||
/// Create an empty registry.
|
||||
#[must_use]
|
||||
pub fn new() -> Self {
|
||||
@@ -1626,24 +1682,64 @@ impl SpawnRegistry {
|
||||
}
|
||||
|
||||
/// Record a freshly spawned child. Called from the spawn-observer hook.
|
||||
pub fn record(&self, pid: i32, pgid: Option<i32>) {
|
||||
self.spawned.lock().push(SpawnedProcess { pid, pgid });
|
||||
///
|
||||
/// The `Process` handle MUST be opened by the caller *immediately* after
|
||||
/// the child's pid becomes visible, so identity is pinned before any race
|
||||
/// with pid recycling can start. When the pin fails (child already exited
|
||||
/// before we could `Process::from_pid`) the entry becomes a no-op at
|
||||
/// termination time — there is nothing left to signal.
|
||||
///
|
||||
/// Exited entries are swept opportunistically once the recorded vec
|
||||
/// crosses the next-sweep watermark, so long-running loops of short
|
||||
/// external commands cannot exhaust the process' FD/handle limit by
|
||||
/// retaining one owned handle per historical spawn.
|
||||
pub fn record(&self, pgid: Option<i32>, process: Option<Process>) {
|
||||
let mut state = self.state.lock();
|
||||
state.spawned.push(SpawnedProcess { process, pgid });
|
||||
if state.spawned.len() >= state.next_sweep_at.max(Self::PRUNE_THRESHOLD) {
|
||||
prune_exited(&mut state.spawned);
|
||||
// Schedule the next sweep `PRUNE_THRESHOLD` further records away.
|
||||
// Comparing against the post-sweep live-set size (not the pre-sweep
|
||||
// length) bounds the sweep frequency when many entries survive:
|
||||
// each sweep costs O(N) but now runs at most once per
|
||||
// `PRUNE_THRESHOLD` records, so amortized per-record cost is O(1)
|
||||
// even if the live set stays large.
|
||||
state.next_sweep_at = state.spawned.len() + Self::PRUNE_THRESHOLD;
|
||||
}
|
||||
}
|
||||
|
||||
/// Build the kill set from the processes recorded so far. Re-read on every
|
||||
/// signal wave so a child spawned during a grace window — between the
|
||||
/// cancel firing and the next wave — is still reaped.
|
||||
///
|
||||
/// A recorded pid contributes only while alive (`add_pid` opens a stable
|
||||
/// handle, skipping the dead); a recorded pgid contributes only while the
|
||||
/// group still has members, so once the run's whole tree exits the targets
|
||||
/// are empty and the wave loop can stop early.
|
||||
/// A recorded process contributes only while alive; a recorded pgid
|
||||
/// contributes only while the group still has members, so once the run's
|
||||
/// whole tree exits the targets are empty and the wave loop can stop early.
|
||||
///
|
||||
/// Pruning also runs here so a cancellation cycle sees a compact target
|
||||
/// set even when the record-time threshold hasn't fired yet.
|
||||
#[must_use]
|
||||
pub fn build_targets(&self) -> TerminationTargets {
|
||||
let mut targets = TerminationTargets::new();
|
||||
let spawned = self.spawned.lock().clone();
|
||||
let spawned = {
|
||||
let mut state = self.state.lock();
|
||||
prune_exited(&mut state.spawned);
|
||||
// Reset the watermark to the current live-set size + threshold;
|
||||
// leaving a stale pre-sweep value would misgate the next
|
||||
// record-time sweep.
|
||||
state.next_sweep_at = state.spawned.len() + Self::PRUNE_THRESHOLD;
|
||||
state.spawned.clone()
|
||||
};
|
||||
for entry in spawned {
|
||||
targets.add_pid(entry.pid);
|
||||
if let Some(process) = entry.process {
|
||||
targets.add_process(process);
|
||||
}
|
||||
// If the observer failed to pin a handle at spawn time (the child
|
||||
// exited before `Process::from_pid` could open it), the child is
|
||||
// already gone — signalling anything for that pid would either
|
||||
// no-op or, worse, race a recycled pid onto an unrelated process.
|
||||
// Drop the entry entirely rather than reintroduce the pid-reuse
|
||||
// window this whole change exists to close (#4605).
|
||||
if let Some(pgid) = entry.pgid
|
||||
&& pgid > 0
|
||||
&& process_group_alive(pgid)
|
||||
@@ -1655,6 +1751,40 @@ impl SpawnRegistry {
|
||||
}
|
||||
}
|
||||
|
||||
/// Drop registry entries whose pinned process, process group, and — on
|
||||
/// Windows — descendant tree are all gone. With nothing still-live the entry
|
||||
/// contributes nothing to the next termination wave and only pins an owned OS
|
||||
/// handle for no reason.
|
||||
///
|
||||
/// The platform split matters because Windows has no process groups. On Unix
|
||||
/// a child reparented onto init keeps its pgid, so a live pgid still catches
|
||||
/// grandchildren whose immediate parent exited. On Windows there is no
|
||||
/// reparenting and no pgid, so we probe the descendant tree directly through
|
||||
/// the still-open pinned handle — dropping that handle would release the pid
|
||||
/// slot, letting a recycled pid make future Toolhelp walks unsafe (issue
|
||||
/// #4605) and orphaning any leftover child from the next cancellation wave.
|
||||
fn prune_exited(spawned: &mut Vec<SpawnedProcess>) {
|
||||
spawned.retain(|entry| {
|
||||
if let Some(process) = &entry.process {
|
||||
if process.status() == ProcessStatus::Running {
|
||||
return true;
|
||||
}
|
||||
// Windows-only: root exited but the pinned handle still keeps its
|
||||
// pid reserved, so `live_descendants` walks the *original* subtree
|
||||
// via Toolhelp. If any child is still running we must keep the
|
||||
// entry — closing the handle would both release the pid (racing
|
||||
// pid reuse) and strand the surviving child.
|
||||
#[cfg(target_os = "windows")]
|
||||
if !process.live_descendants().is_empty() {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
entry
|
||||
.pgid
|
||||
.is_some_and(|pgid| pgid > 0 && process_group_alive(pgid))
|
||||
});
|
||||
}
|
||||
|
||||
/// True when process group `pgid` still has at least one member. `kill(2)`
|
||||
/// with signal 0 performs permission/existence checks without delivering a
|
||||
/// signal; `EPERM` means the group exists but is not ours to signal, which
|
||||
@@ -1752,4 +1882,206 @@ mod tests {
|
||||
broken `proc_listchildpids`",
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression test for issue #4605: `SpawnRegistry` MUST pin a stable
|
||||
/// [`Process`] reference at spawn time rather than defer re-opening the
|
||||
/// pid until termination.
|
||||
///
|
||||
/// Before the fix, `SpawnRegistry` stored only the raw pid; `build_targets`
|
||||
/// called `Process::from_pid` at cancellation time. On Windows pids recycle
|
||||
/// aggressively, so a bash-spawned `pwsh.exe` that had already exited could
|
||||
/// see its pid reassigned to an unrelated PowerShell session (e.g. the
|
||||
/// user's other Cursor terminal). `Process::from_pid` at cancel time would
|
||||
/// happily open that unrelated process, and `signal_tree` would then
|
||||
/// enumerate — and `TerminateProcess` — the entire foreign subtree.
|
||||
///
|
||||
/// This test cannot literally trigger Windows pid recycling from a
|
||||
/// cross-platform Rust test, but it can prove the observable defense: a
|
||||
/// recorded process reference survives the original pid's death (so no
|
||||
/// "look it up again" step exists to be raced), and the registry never
|
||||
/// consults `Process::from_pid` when a handle was pinned at record time.
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn spawn_registry_pins_identity_at_record_time() {
|
||||
use std::{process::Command, thread, time::Duration};
|
||||
|
||||
// Phase 1: while the child is alive, the pinned handle carries identity
|
||||
// forward into `build_targets` without any `Process::from_pid` re-open
|
||||
// step existing to be raced against pid reuse.
|
||||
let mut long = Command::new("sleep")
|
||||
.arg("30")
|
||||
.spawn()
|
||||
.expect("spawn sleep");
|
||||
let long_pid = i32::try_from(long.id()).expect("child pid fits in i32");
|
||||
|
||||
let registry = SpawnRegistry::new();
|
||||
let pinned = Process::from_pid(long_pid).expect("pin child at record time");
|
||||
registry.record(None, Some(pinned));
|
||||
|
||||
let live_targets = registry.build_targets();
|
||||
assert!(
|
||||
!live_targets.is_empty(),
|
||||
"a still-live pinned child must appear in the target set — otherwise the \
|
||||
cancellation cleanup would silently miss it"
|
||||
);
|
||||
let live_pids: Vec<i32> = live_targets.processes.iter().map(Process::pid).collect();
|
||||
assert_eq!(
|
||||
live_pids,
|
||||
vec![long_pid],
|
||||
"target set must come from the pinned handle recorded at spawn time, not a \
|
||||
re-lookup by pid (which would race pid reuse — issue #4605)"
|
||||
);
|
||||
|
||||
let _ = long.kill();
|
||||
let _ = long.wait();
|
||||
|
||||
// Phase 2: once the child exits, the registry MUST drop the entry
|
||||
// rather than reintroduce a `Process::from_pid` re-open at kill time.
|
||||
// Poll until pruning sees the pidfd as Exited (kernel-visible within
|
||||
// milliseconds in practice).
|
||||
let mut empty_after_exit = false;
|
||||
for _ in 0..40 {
|
||||
if registry.build_targets().is_empty() {
|
||||
empty_after_exit = true;
|
||||
break;
|
||||
}
|
||||
thread::sleep(Duration::from_millis(25));
|
||||
}
|
||||
assert!(
|
||||
empty_after_exit,
|
||||
"once the pinned child exits the registry must drop it — re-opening by pid at \
|
||||
termination time is exactly the pid-reuse race #4605 closes"
|
||||
);
|
||||
}
|
||||
|
||||
/// `TerminationTargets::add_process` must accept a pre-pinned handle
|
||||
/// without going through `Process::from_pid`. This is the API contract
|
||||
/// `SpawnRegistry` relies on to avoid the PID-reuse race.
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn add_process_bypasses_from_pid_lookup() {
|
||||
let self_pid = i32::try_from(std::process::id()).expect("self pid fits in i32");
|
||||
let pinned = Process::from_pid(self_pid).expect("pin self");
|
||||
|
||||
let mut targets = TerminationTargets::new();
|
||||
targets.add_process(pinned.clone());
|
||||
assert!(!targets.is_empty(), "add_process must record the pinned handle");
|
||||
|
||||
// Adding the same pid again through either entry point must dedupe:
|
||||
// otherwise every wave in `terminate_run` would re-signal the same
|
||||
// tree N times.
|
||||
targets.add_process(pinned);
|
||||
targets.add_pid(self_pid);
|
||||
assert_eq!(targets.processes.len(), 1, "duplicate pids must be deduped");
|
||||
}
|
||||
|
||||
/// Regression test for the review on PR #4606: a long-running shell
|
||||
/// command that spawns many short-lived external processes must not
|
||||
/// retain one owned handle per historical spawn — that would exhaust
|
||||
/// per-process FD/handle limits (pidfd on Linux, `HANDLE` on Windows).
|
||||
/// The registry MUST prune dead entries once the recorded vec crosses
|
||||
/// the sweep threshold.
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn spawn_registry_prunes_exited_entries() {
|
||||
use std::{thread, time::Duration};
|
||||
|
||||
let registry = SpawnRegistry::new();
|
||||
|
||||
// Fabricate many recorded-then-exited children by pinning ourselves,
|
||||
// pushing the entry, then immediately treating it as "dead" from the
|
||||
// registry's perspective. To simulate the exit without actually
|
||||
// killing the harness, use `Process::from_pid(1)` for a pid that
|
||||
// (on Linux) is init and never exits — but wrap the recording in a
|
||||
// pattern that guarantees `status()` returns Exited for the pruner:
|
||||
// spawn a tiny child, pin it, wait for exit, then record.
|
||||
for _ in 0..(SpawnRegistry::PRUNE_THRESHOLD * 2) {
|
||||
let mut child = std::process::Command::new("true")
|
||||
.spawn()
|
||||
.expect("spawn true");
|
||||
let pid = i32::try_from(child.id()).expect("child pid fits in i32");
|
||||
let pinned = Process::from_pid(pid);
|
||||
let _ = child.wait();
|
||||
// Give the kernel a moment to mark the pidfd readable so `status()`
|
||||
// reports Exited when the pruner probes.
|
||||
for _ in 0..20 {
|
||||
if pinned
|
||||
.as_ref()
|
||||
.is_some_and(|process| process.status() == ProcessStatus::Exited)
|
||||
{
|
||||
break;
|
||||
}
|
||||
thread::sleep(Duration::from_millis(5));
|
||||
}
|
||||
registry.record(None, pinned);
|
||||
}
|
||||
|
||||
let retained = registry.state.lock().spawned.len();
|
||||
assert!(
|
||||
retained < SpawnRegistry::PRUNE_THRESHOLD,
|
||||
"pruning must bound retained entries below the sweep threshold once the pinned \
|
||||
processes have exited; got {retained} retained (threshold {})",
|
||||
SpawnRegistry::PRUNE_THRESHOLD
|
||||
);
|
||||
|
||||
// build_targets sees no live handles → empty target set, matching the
|
||||
// contract that fully-exited registries stop the wave loop early.
|
||||
let targets = registry.build_targets();
|
||||
assert!(
|
||||
targets.is_empty(),
|
||||
"registry of only-dead entries must produce an empty target set"
|
||||
);
|
||||
}
|
||||
|
||||
/// Regression test for the third review on PR #4606: once the recorded
|
||||
/// vec crosses `PRUNE_THRESHOLD`, subsequent `record` calls must NOT
|
||||
/// sweep on every spawn. Without the `next_sweep_at` watermark, a large
|
||||
/// fan-out run whose live children exceed the threshold turned every
|
||||
/// spawn into an O(N) status probe of the whole retained set.
|
||||
///
|
||||
/// The check reasons about the observable side effect: after N records
|
||||
/// past threshold with entries that CANNOT be pruned (all still live),
|
||||
/// the retained size grows monotonically by exactly N — no sweep runs
|
||||
/// have modified the vec in between. The direct signal of "did a sweep
|
||||
/// happen" is a stable pinned handle count across records.
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn spawn_registry_watermark_bounds_sweep_frequency() {
|
||||
let self_pid = i32::try_from(std::process::id()).expect("self pid fits in i32");
|
||||
let registry = SpawnRegistry::new();
|
||||
|
||||
// Fill past threshold with entries that are permanently alive
|
||||
// (pinning ourselves) so the pruner has nothing to remove.
|
||||
let fill = SpawnRegistry::PRUNE_THRESHOLD + 10;
|
||||
for _ in 0..fill {
|
||||
registry.record(None, Process::from_pid(self_pid));
|
||||
}
|
||||
let after_fill = registry.state.lock().spawned.len();
|
||||
assert_eq!(
|
||||
after_fill, fill,
|
||||
"live-only entries must not be pruned during warm-up"
|
||||
);
|
||||
let watermark_after_fill = registry.state.lock().next_sweep_at;
|
||||
|
||||
// Every additional record with a live entry must land in the vec
|
||||
// verbatim and — critically — NOT re-enter `prune_exited` until the
|
||||
// vec crosses the freshly scheduled watermark. If the guard were
|
||||
// still `len >= PRUNE_THRESHOLD` (pre-fix), a sweep would fire on
|
||||
// every one of these records.
|
||||
let extra = 20;
|
||||
for _ in 0..extra {
|
||||
registry.record(None, Process::from_pid(self_pid));
|
||||
}
|
||||
let after_extra = registry.state.lock().spawned.len();
|
||||
assert_eq!(
|
||||
after_extra,
|
||||
after_fill + extra,
|
||||
"records with live entries must accumulate without triggering per-spawn sweeps"
|
||||
);
|
||||
assert_eq!(
|
||||
registry.state.lock().next_sweep_at,
|
||||
watermark_after_fill,
|
||||
"watermark must not advance while the vec stays below it — otherwise a sweep ran"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1372,7 +1372,14 @@ async fn read_output_bytes(
|
||||
|
||||
impl SpawnObserver for process::SpawnRegistry {
|
||||
fn on_spawn(&self, pid: i32, pgid: Option<i32>) {
|
||||
self.record(pid, pgid);
|
||||
// Pin a stable process reference *now*, before the pid can be recycled.
|
||||
// On Windows an open handle keeps the pid slot reserved for the lifetime
|
||||
// of the handle; on Linux the pidfd carries identity; on macOS the
|
||||
// recorded start-time triple detects impersonation. Deferring the open
|
||||
// to `build_targets` (as the old code did) let a recycled pid resolve
|
||||
// to an unrelated process — issue #4605.
|
||||
let process = process::Process::from_pid(pid);
|
||||
self.record(pgid, process);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+3
-3
@@ -240,9 +240,9 @@ Prompt selection:
|
||||
|
||||
Remote summarization modes:
|
||||
|
||||
- If `compaction.remoteEndpoint` is set and remote compaction is enabled, local summary generation POSTs:
|
||||
- `{ systemPrompt, prompt }`
|
||||
- Expects JSON containing at least `{ summary }`.
|
||||
- If `compaction.remoteEndpoint` is set and remote compaction is enabled, local summary generation POSTs one of two wire formats:
|
||||
- custom omp summarizer endpoints receive `{ systemPrompt, prompt }` and must return JSON containing at least `{ summary }`.
|
||||
- OpenAI-compatible endpoints whose path ends in `/chat/completions` receive `{ model, messages, stream: false }`, where `messages` contains one system prompt and one user prompt. The summary is read from `choices[0].message.content`, which lets self-hosted servers such as llama.cpp and vLLM act as remote compactors without a separate summarizer shim.
|
||||
- For OpenAI/OpenAI Codex models, compaction first tries the provider-native `/responses/compact` endpoint when remote compaction is enabled. It preserves provider replacement history in `preserveData.openaiRemoteCompaction` and falls back to local summarization if that native request fails.
|
||||
|
||||
### Handoff generation
|
||||
|
||||
@@ -2,6 +2,15 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added per-tool abort metadata so stream-wide aborts can label matching tool-call placeholders separately from unaffected sibling calls ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed handoff generation retrying with `toolChoice: "auto"` when custom OpenAI-compatible providers reject `toolChoice: "none"` with an auto-only 400. ([#4715](https://github.com/can1357/oh-my-pi/issues/4715))
|
||||
- Fixed generic remote compaction against OpenAI-compatible `/chat/completions` endpoints (for example llama.cpp `openai-completions`) by sending chat messages instead of the custom `{ systemPrompt, prompt }` summarizer payload. ([#4630](https://github.com/can1357/oh-my-pi/issues/4630))
|
||||
|
||||
## [16.3.7] - 2026-07-05
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -112,6 +112,26 @@ function hardToolChoiceBlocks(choice: ToolChoice | undefined, requiredTool: stri
|
||||
* tool's own window elapses. A cheap synchronous queue check; latency-bounded
|
||||
* at one tick.
|
||||
*/
|
||||
/**
|
||||
* Abort reason for a turn-wide interruption where only some tool calls caused
|
||||
* the abort and sibling placeholders need neutral messages.
|
||||
*/
|
||||
export interface ToolScopedAbortReason {
|
||||
readonly kind: "tool-scoped-abort";
|
||||
readonly message: string;
|
||||
readonly toolCallMessages: Record<string, string>;
|
||||
readonly defaultToolCallMessage: string;
|
||||
}
|
||||
|
||||
/** Creates an abort reason that labels matching tool calls separately from siblings. */
|
||||
export function createToolScopedAbortReason(
|
||||
message: string,
|
||||
toolCallMessages: Record<string, string>,
|
||||
defaultToolCallMessage: string,
|
||||
): ToolScopedAbortReason {
|
||||
return { kind: "tool-scoped-abort", message, toolCallMessages, defaultToolCallMessage };
|
||||
}
|
||||
|
||||
const STEERING_INTERRUPT_POLL_MS = 250;
|
||||
|
||||
class HarmonyLeakInterruption extends Error {
|
||||
@@ -173,6 +193,7 @@ function snapshotAssistantMessage(message: AssistantMessage): AssistantMessage {
|
||||
cost: { ...message.usage.cost },
|
||||
},
|
||||
disabledFeatures: message.disabledFeatures ? [...message.disabledFeatures] : undefined,
|
||||
toolCallAbortMessages: message.toolCallAbortMessages ? { ...message.toolCallAbortMessages } : undefined,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -918,9 +939,17 @@ async function runLoopBody(
|
||||
(c): c is ToolCallContent =>
|
||||
c.type === "toolCall" && (c as CursorExecResolvedCarrier)[kCursorExecResolved] !== true,
|
||||
);
|
||||
// Provider-built aborted messages (stream error events) carry no
|
||||
// per-tool labels; derive them from a tool-scoped abort signal so
|
||||
// only the matching call is blamed and siblings stay neutral.
|
||||
const scopedAbort = toolScopedAbortReason(signal);
|
||||
const toolCallAbortMessages =
|
||||
message.toolCallAbortMessages ??
|
||||
(scopedAbort ? buildToolCallAbortMessages(message, scopedAbort) : undefined);
|
||||
const toolResults: ToolResultMessage[] = [];
|
||||
for (const toolCall of toolCalls) {
|
||||
const result = createAbortedToolResult(toolCall, stream, message.stopReason, message.errorMessage);
|
||||
const errorMessage = toolCallAbortMessages?.[toolCall.id] ?? message.errorMessage;
|
||||
const result = createAbortedToolResult(toolCall, stream, message.stopReason, errorMessage);
|
||||
currentContext.messages.push(result);
|
||||
newMessages.push(result);
|
||||
toolResults.push(result);
|
||||
@@ -1610,6 +1639,34 @@ function emitDiscardedHarmonyPartial(
|
||||
});
|
||||
}
|
||||
|
||||
function isStringRecord(value: unknown): value is Record<string, string> {
|
||||
if (!value || typeof value !== "object" || Array.isArray(value)) return false;
|
||||
return Object.values(value).every(child => typeof child === "string");
|
||||
}
|
||||
|
||||
function toolScopedAbortReason(signal: AbortSignal | undefined): ToolScopedAbortReason | undefined {
|
||||
const reason = signal?.reason;
|
||||
if (!reason || typeof reason !== "object") return undefined;
|
||||
if (Reflect.get(reason, "kind") !== "tool-scoped-abort") return undefined;
|
||||
if (typeof Reflect.get(reason, "message") !== "string") return undefined;
|
||||
if (typeof Reflect.get(reason, "defaultToolCallMessage") !== "string") return undefined;
|
||||
return isStringRecord(Reflect.get(reason, "toolCallMessages")) ? reason : undefined;
|
||||
}
|
||||
|
||||
function buildToolCallAbortMessages(
|
||||
message: AssistantMessage,
|
||||
reason: ToolScopedAbortReason,
|
||||
): Record<string, string> | undefined {
|
||||
let hasToolCall = false;
|
||||
const messages: Record<string, string> = {};
|
||||
for (const block of message.content) {
|
||||
if (block.type !== "toolCall") continue;
|
||||
hasToolCall = true;
|
||||
messages[block.id] = reason.toolCallMessages[block.id] ?? reason.defaultToolCallMessage;
|
||||
}
|
||||
return hasToolCall ? messages : undefined;
|
||||
}
|
||||
|
||||
/** Resolve the human-readable reason an abort carried. A caller that aborts via
|
||||
* `AbortController.abort(reason)` with a string or a non-`AbortError` `Error`
|
||||
* (e.g. the coding agent's user-interrupt label) gets that text surfaced on the
|
||||
@@ -1617,6 +1674,8 @@ function emitDiscardedHarmonyPartial(
|
||||
* `signal.reason` is the default `AbortError` `DOMException`) falls back to the
|
||||
* generic sentinel that downstream renderers treat as "no specific reason". */
|
||||
export function abortReasonText(signal: AbortSignal | undefined): string {
|
||||
const scopedReason = toolScopedAbortReason(signal);
|
||||
if (scopedReason) return scopedReason.message;
|
||||
const reason = signal?.reason;
|
||||
if (typeof reason === "string" && reason.trim().length > 0) return reason;
|
||||
if (reason instanceof Error && reason.name !== "AbortError" && reason.message.trim().length > 0) {
|
||||
@@ -1665,6 +1724,11 @@ function emitAbortedAssistantMessage(
|
||||
// labeled user interrupt still surfaces through `errorMessage`, but partial
|
||||
// tool arguments are unsafe to keep and can carry incomplete provider IDs.
|
||||
const retained = retainCompletedToolCalls(base, completedToolCallIds);
|
||||
const scopedAbort = toolScopedAbortReason(requestSignal);
|
||||
const toolCallAbortMessages = scopedAbort ? buildToolCallAbortMessages(retained, scopedAbort) : undefined;
|
||||
if (toolCallAbortMessages) {
|
||||
retained.toolCallAbortMessages = toolCallAbortMessages;
|
||||
}
|
||||
const abortedMessage = snapshotAssistantMessage(retained);
|
||||
if (addedPartial) {
|
||||
context.messages[context.messages.length - 1] = abortedMessage;
|
||||
|
||||
@@ -698,6 +698,12 @@ function createSummarizationError(prefix: string, response: AssistantMessage): E
|
||||
return response.errorStatus === undefined ? new Error(text) : new ProviderHttpError(text, response.errorStatus);
|
||||
}
|
||||
|
||||
function shouldRetryHandoffWithAutoToolChoice(response: AssistantMessage): boolean {
|
||||
if (response.errorStatus !== 400) return false;
|
||||
const message = response.errorMessage ?? "";
|
||||
return /\btool_choice\b/i.test(message) && /\bauto\b/i.test(message) && /\bsupported\b/i.test(message);
|
||||
}
|
||||
|
||||
/**
|
||||
* Generate a summary of the conversation using the LLM.
|
||||
* If previousSummary is provided, uses the update prompt to merge.
|
||||
@@ -813,14 +819,17 @@ export async function generateSummary(
|
||||
];
|
||||
|
||||
if (options?.remoteEndpoint) {
|
||||
const remote = await requestRemoteCompaction(
|
||||
options.remoteEndpoint,
|
||||
{
|
||||
systemPrompt: SUMMARIZATION_SYSTEM_PROMPT,
|
||||
prompt: promptText,
|
||||
},
|
||||
signal,
|
||||
{ fetch: options.fetch },
|
||||
const endpoint = options.remoteEndpoint;
|
||||
const remote = await withAuth(
|
||||
apiKey,
|
||||
key =>
|
||||
requestRemoteCompaction(
|
||||
endpoint,
|
||||
{ systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, prompt: promptText },
|
||||
signal,
|
||||
{ fetch: options.fetch, model, apiKey: key },
|
||||
),
|
||||
{ signal, missingKeyMessage: "Remote compaction credentials unavailable" },
|
||||
);
|
||||
return remote.summary;
|
||||
}
|
||||
@@ -917,24 +926,33 @@ export interface HandoffFromContextOptions {
|
||||
* `streamOptions` that mirror the live turn's cache routing. That keeps the
|
||||
* cache-preserving context construction in the host (which owns the transform
|
||||
* pipeline) while this function centralizes the handoff request contract:
|
||||
* `toolChoice: "none"`, clamped reasoning effort, oneshot telemetry, text-only
|
||||
* extraction, and provider-error mapping.
|
||||
* cache-first `toolChoice: "none"`, clamped reasoning effort, one retry for
|
||||
* auto-only `tool_choice` providers, oneshot telemetry, text-only extraction,
|
||||
* and provider-error mapping.
|
||||
*/
|
||||
export async function generateHandoffFromContext(
|
||||
context: Context,
|
||||
model: Model,
|
||||
options: HandoffFromContextOptions,
|
||||
): Promise<string> {
|
||||
const response = await instrumentedCompleteSimple(
|
||||
model,
|
||||
context,
|
||||
{
|
||||
...options.streamOptions,
|
||||
reasoning: resolveCompactionEffort(model, options.thinkingLevel),
|
||||
toolChoice: "none",
|
||||
},
|
||||
{ telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl },
|
||||
);
|
||||
const requestOptions = {
|
||||
...options.streamOptions,
|
||||
reasoning: resolveCompactionEffort(model, options.thinkingLevel),
|
||||
toolChoice: "none" as const,
|
||||
};
|
||||
let response = await instrumentedCompleteSimple(model, context, requestOptions, {
|
||||
telemetry: options.telemetry,
|
||||
oneshotKind: "handoff",
|
||||
completeImpl: options.completeImpl,
|
||||
});
|
||||
if (response.stopReason === "error" && shouldRetryHandoffWithAutoToolChoice(response)) {
|
||||
response = await instrumentedCompleteSimple(
|
||||
model,
|
||||
context,
|
||||
{ ...requestOptions, toolChoice: "auto" },
|
||||
{ telemetry: options.telemetry, oneshotKind: "handoff", completeImpl: options.completeImpl },
|
||||
);
|
||||
}
|
||||
|
||||
if (response.stopReason === "error") {
|
||||
throw createSummarizationError("Handoff generation failed", response);
|
||||
@@ -1001,14 +1019,17 @@ async function generateShortSummary(
|
||||
promptText += SHORT_SUMMARY_PROMPT;
|
||||
|
||||
if (options?.remoteEndpoint) {
|
||||
const remote = await requestRemoteCompaction(
|
||||
options.remoteEndpoint,
|
||||
{
|
||||
systemPrompt: SUMMARIZATION_SYSTEM_PROMPT,
|
||||
prompt: promptText,
|
||||
},
|
||||
signal,
|
||||
{ fetch: options?.fetch },
|
||||
const endpoint = options.remoteEndpoint;
|
||||
const remote = await withAuth(
|
||||
apiKey,
|
||||
key =>
|
||||
requestRemoteCompaction(
|
||||
endpoint,
|
||||
{ systemPrompt: SUMMARIZATION_SYSTEM_PROMPT, prompt: promptText },
|
||||
signal,
|
||||
{ fetch: options?.fetch, model, apiKey: key },
|
||||
),
|
||||
{ signal, missingKeyMessage: "Remote compaction credentials unavailable" },
|
||||
);
|
||||
return remote.summary;
|
||||
}
|
||||
|
||||
@@ -547,16 +547,55 @@ export async function requestOpenAiRemoteCompaction(
|
||||
return { provider: model.provider, replacementHistory, compactionItem };
|
||||
}
|
||||
|
||||
/**
|
||||
* Generic remote-compaction POST. Two wire shapes are auto-selected by
|
||||
* endpoint suffix so a single `compaction.remoteEndpoint` setting can point at
|
||||
* either a purpose-built omp summarizer (`{systemPrompt, prompt}` → `{summary}`)
|
||||
* or any OpenAI-compatible chat-completions server (`/chat/completions`,
|
||||
* `/v1/chat/completions`, …) as reported for llama.cpp / vLLM / etc. in
|
||||
* issue #4630: without this, the omp payload was rejected with
|
||||
* HTTP 400 `"'messages' is required"`, compaction silently fell back to
|
||||
* local summarization, and context grew unbounded.
|
||||
*
|
||||
* When `context.model` is provided the chat-completions body is tagged with
|
||||
* that model's wire id (llama.cpp requires the field) and `context.apiKey` is
|
||||
* forwarded as `Authorization: Bearer`. Callers wrap this in `withAuth` so
|
||||
* 401s force-refresh through the standard credential rotation policy.
|
||||
*/
|
||||
export async function requestRemoteCompaction(
|
||||
endpoint: string,
|
||||
request: RemoteCompactionRequest,
|
||||
signal?: AbortSignal,
|
||||
opts?: { fetch?: FetchImpl; timeoutMs?: number },
|
||||
opts?: { fetch?: FetchImpl; timeoutMs?: number; model?: Model; apiKey?: string },
|
||||
): Promise<RemoteCompactionResponse> {
|
||||
let endpointPath = endpoint;
|
||||
try {
|
||||
endpointPath = new URL(endpoint).pathname;
|
||||
} catch {
|
||||
// Keep the raw endpoint for relative/custom fetch implementations.
|
||||
}
|
||||
const isChatCompletions = /\/chat\/completions\/?$/.test(endpointPath);
|
||||
const headers: Record<string, string> = { "content-type": "application/json" };
|
||||
if (isChatCompletions) {
|
||||
if (opts?.apiKey) headers.Authorization = `Bearer ${opts.apiKey}`;
|
||||
if (opts?.model?.headers) Object.assign(headers, opts.model.headers);
|
||||
}
|
||||
|
||||
const body: Record<string, unknown> = isChatCompletions
|
||||
? {
|
||||
model: opts?.model ? resolveOpenAiCompactModel(opts.model) : undefined,
|
||||
messages: [
|
||||
{ role: "system", content: request.systemPrompt },
|
||||
{ role: "user", content: request.prompt },
|
||||
],
|
||||
stream: false,
|
||||
}
|
||||
: { systemPrompt: request.systemPrompt, prompt: request.prompt };
|
||||
|
||||
const response = await (opts?.fetch ?? fetch)(endpoint, {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify(request),
|
||||
headers,
|
||||
body: JSON.stringify(body),
|
||||
signal: withRequestTimeout(signal, opts?.timeoutMs ?? REMOTE_COMPACTION_TIMEOUT_MS),
|
||||
});
|
||||
|
||||
@@ -577,6 +616,31 @@ export async function requestRemoteCompaction(
|
||||
);
|
||||
}
|
||||
|
||||
if (isChatCompletions) {
|
||||
type ChatCompletionsResponse = {
|
||||
choices?: Array<{
|
||||
message?: {
|
||||
content?: string | Array<{ type?: string; text?: string }> | null;
|
||||
};
|
||||
}>;
|
||||
};
|
||||
const data = (await response.json()) as ChatCompletionsResponse | undefined;
|
||||
const choice = data?.choices?.[0]?.message?.content;
|
||||
let summary: string | undefined;
|
||||
if (typeof choice === "string") {
|
||||
summary = choice;
|
||||
} else if (Array.isArray(choice)) {
|
||||
summary = choice
|
||||
.filter((part): part is { type?: string; text: string } => typeof part?.text === "string")
|
||||
.map(part => part.text)
|
||||
.join("");
|
||||
}
|
||||
if (typeof summary !== "string" || summary.length === 0) {
|
||||
throw new Error("Remote compaction response missing choices[0].message.content");
|
||||
}
|
||||
return { summary };
|
||||
}
|
||||
|
||||
const data = (await response.json()) as RemoteCompactionResponse | undefined;
|
||||
if (!data || typeof data.summary !== "string") {
|
||||
throw new Error("Remote compaction response missing summary");
|
||||
|
||||
@@ -9,7 +9,7 @@ import {
|
||||
import { ThinkingLevel } from "@oh-my-pi/pi-agent-core/thinking";
|
||||
import type { AssistantMessage, Model, ToolCall } from "@oh-my-pi/pi-ai";
|
||||
import * as ai from "@oh-my-pi/pi-ai";
|
||||
import { Effort } from "@oh-my-pi/pi-ai";
|
||||
import { Effort, z } from "@oh-my-pi/pi-ai";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
|
||||
function createAssistantMessage(content: AssistantMessage["content"]): AssistantMessage {
|
||||
@@ -32,6 +32,28 @@ function createAssistantMessage(content: AssistantMessage["content"]): Assistant
|
||||
};
|
||||
}
|
||||
|
||||
function createAssistantError(errorStatus: number, errorMessage: string): AssistantMessage {
|
||||
return {
|
||||
...createAssistantMessage([]),
|
||||
stopReason: "error",
|
||||
errorStatus,
|
||||
errorMessage,
|
||||
};
|
||||
}
|
||||
|
||||
const handoffToolSchema = z.object({ note: z.string().optional() });
|
||||
|
||||
function createHandoffTool(): AgentTool<typeof handoffToolSchema> {
|
||||
return {
|
||||
name: "handoff_probe",
|
||||
label: "Handoff Probe",
|
||||
description: "Confirms handoff requests keep live tools available.",
|
||||
parameters: handoffToolSchema,
|
||||
intent: "omit",
|
||||
execute: async () => ({ content: [{ type: "text", text: "ok" }], details: {} }),
|
||||
};
|
||||
}
|
||||
|
||||
function getTestModel(): Model {
|
||||
const model = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
if (!model) {
|
||||
@@ -155,4 +177,99 @@ describe("handoff helpers", () => {
|
||||
reasoning: Effort.Medium,
|
||||
});
|
||||
});
|
||||
|
||||
test("generateHandoffFromContext retries auto-only tool_choice rejection with live tools", async () => {
|
||||
const completeSimpleSpy = vi
|
||||
.spyOn(ai, "completeSimple")
|
||||
.mockResolvedValueOnce(
|
||||
createAssistantError(
|
||||
400,
|
||||
"400 Bad Request: Only a tool_choice of 'auto' is supported for this model; param=tool_choice",
|
||||
),
|
||||
)
|
||||
.mockResolvedValueOnce(createAssistantMessage([{ type: "text", text: "## Goal\nRecovered on retry" }]));
|
||||
const model = getTestModel();
|
||||
const tools = [createHandoffTool()];
|
||||
const context = {
|
||||
systemPrompt: ["Live system prompt"],
|
||||
tools,
|
||||
messages: [{ role: "user" as const, content: "prepare handoff", timestamp: 1 }],
|
||||
};
|
||||
|
||||
const document = await generateHandoffFromContext(context, model, {
|
||||
streamOptions: {
|
||||
apiKey: "test-key",
|
||||
sessionId: "sess-auto-only:side:42",
|
||||
promptCacheKey: "sess-auto-only",
|
||||
},
|
||||
thinkingLevel: ThinkingLevel.Medium,
|
||||
});
|
||||
|
||||
expect(document).toBe("## Goal\nRecovered on retry");
|
||||
expect(completeSimpleSpy).toHaveBeenCalledTimes(2);
|
||||
const firstCall = completeSimpleSpy.mock.calls[0];
|
||||
const secondCall = completeSimpleSpy.mock.calls[1];
|
||||
if (!firstCall) throw new Error("Expected initial completeSimple call");
|
||||
if (!secondCall) throw new Error("Expected retry completeSimple call");
|
||||
const [firstModel, firstContext, firstOptions] = firstCall;
|
||||
const [secondModel, secondContext, secondOptions] = secondCall;
|
||||
expect(firstModel).toBe(model);
|
||||
expect(secondModel).toBe(model);
|
||||
expect(firstContext).toBe(context);
|
||||
expect(secondContext).toBe(context);
|
||||
expect(firstContext.tools).toBe(tools);
|
||||
expect(secondContext.tools).toBe(tools);
|
||||
expect(firstOptions).toMatchObject({
|
||||
apiKey: "test-key",
|
||||
sessionId: "sess-auto-only:side:42",
|
||||
promptCacheKey: "sess-auto-only",
|
||||
toolChoice: "none",
|
||||
reasoning: Effort.Medium,
|
||||
});
|
||||
expect(secondOptions).toMatchObject({
|
||||
apiKey: "test-key",
|
||||
sessionId: "sess-auto-only:side:42",
|
||||
promptCacheKey: "sess-auto-only",
|
||||
toolChoice: "auto",
|
||||
reasoning: Effort.Medium,
|
||||
});
|
||||
});
|
||||
|
||||
test("generateHandoffFromContext surfaces unrelated provider 400 without retrying", async () => {
|
||||
const completeSimpleSpy = vi
|
||||
.spyOn(ai, "completeSimple")
|
||||
.mockResolvedValueOnce(createAssistantError(400, "400 Bad Request: unsupported max_tokens; param=max_tokens"));
|
||||
const model = getTestModel();
|
||||
const tools = [createHandoffTool()];
|
||||
const context = {
|
||||
systemPrompt: ["Live system prompt"],
|
||||
tools,
|
||||
messages: [{ role: "user" as const, content: "prepare handoff", timestamp: 1 }],
|
||||
};
|
||||
|
||||
const error = await generateHandoffFromContext(context, model, {
|
||||
streamOptions: {
|
||||
apiKey: "test-key",
|
||||
sessionId: "sess-unrelated-400:side:42",
|
||||
promptCacheKey: "sess-unrelated-400",
|
||||
},
|
||||
thinkingLevel: ThinkingLevel.Medium,
|
||||
}).catch((caught: unknown) => caught);
|
||||
|
||||
if (!(error instanceof Error)) throw new Error("Expected handoff generation to reject");
|
||||
expect(error.message).toContain("unsupported max_tokens");
|
||||
expect(completeSimpleSpy).toHaveBeenCalledTimes(1);
|
||||
const call = completeSimpleSpy.mock.calls[0];
|
||||
if (!call) throw new Error("Expected completeSimple call");
|
||||
const [, calledContext, options] = call;
|
||||
expect(calledContext).toBe(context);
|
||||
expect(calledContext.tools).toBe(tools);
|
||||
expect(options).toMatchObject({
|
||||
apiKey: "test-key",
|
||||
sessionId: "sess-unrelated-400:side:42",
|
||||
promptCacheKey: "sess-unrelated-400",
|
||||
toolChoice: "none",
|
||||
reasoning: Effort.Medium,
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -13,6 +13,7 @@ import {
|
||||
getCompactionV2PreserveData,
|
||||
requestCompactionV2Streaming,
|
||||
requestOpenAiRemoteCompaction,
|
||||
requestRemoteCompaction,
|
||||
shouldUseCompactionV2Streaming,
|
||||
shouldUseOpenAiRemoteCompaction,
|
||||
} from "@oh-my-pi/pi-agent-core/compaction/openai";
|
||||
@@ -540,6 +541,76 @@ describe("requestOpenAiRemoteCompaction timeout", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("requestRemoteCompaction wire formats", () => {
|
||||
test("uses OpenAI chat completions format for /chat/completions endpoints", async () => {
|
||||
const model = buildModel({
|
||||
id: "catalog-selection-id",
|
||||
name: "Qwopus 3.6 35B-A3B Coder",
|
||||
requestModelId: "provider-wire-id",
|
||||
remoteCompaction: { model: "provider-compact-wire-id" },
|
||||
api: "openai-completions",
|
||||
provider: "local-llama",
|
||||
baseUrl: "http://127.0.0.1:8001/v1",
|
||||
headers: { "x-local-llama": "1" },
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 131072,
|
||||
maxTokens: 4096,
|
||||
});
|
||||
let sentBody: unknown;
|
||||
const fetchMock: FetchImpl = async (_input, init) => {
|
||||
if (typeof init?.body !== "string") throw new Error("missing remote compaction request body");
|
||||
sentBody = JSON.parse(init.body) as unknown;
|
||||
const headers = new Headers(init.headers);
|
||||
expect(headers.get("authorization")).toBe("Bearer local-key");
|
||||
expect(headers.get("x-local-llama")).toBe("1");
|
||||
return new Response(JSON.stringify({ choices: [{ message: { content: "remote summary" } }] }), {
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
};
|
||||
|
||||
const result = await requestRemoteCompaction(
|
||||
"http://127.0.0.1:8001/v1/chat/completions",
|
||||
{ systemPrompt: "summarize", prompt: "<conversation>hello</conversation>" },
|
||||
undefined,
|
||||
{ fetch: fetchMock, model, apiKey: "local-key" },
|
||||
);
|
||||
|
||||
expect(result).toEqual({ summary: "remote summary" });
|
||||
expect(sentBody).toEqual({
|
||||
model: "provider-compact-wire-id",
|
||||
messages: [
|
||||
{ role: "system", content: "summarize" },
|
||||
{ role: "user", content: "<conversation>hello</conversation>" },
|
||||
],
|
||||
stream: false,
|
||||
});
|
||||
});
|
||||
|
||||
test("keeps the generic omp summarizer format for other endpoints", async () => {
|
||||
let sentBody: unknown;
|
||||
const fetchMock: FetchImpl = async (_input, init) => {
|
||||
if (typeof init?.body !== "string") throw new Error("missing remote compaction request body");
|
||||
sentBody = JSON.parse(init.body) as unknown;
|
||||
expect(new Headers(init.headers).get("authorization")).toBeNull();
|
||||
return new Response(JSON.stringify({ summary: "generic summary", shortSummary: "generic" }), {
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
};
|
||||
|
||||
const result = await requestRemoteCompaction(
|
||||
"https://compaction.example.test/summarize",
|
||||
{ systemPrompt: "summarize", prompt: "<conversation>hello</conversation>" },
|
||||
undefined,
|
||||
{ fetch: fetchMock, apiKey: "unused-for-generic" },
|
||||
);
|
||||
|
||||
expect(result).toEqual({ summary: "generic summary", shortSummary: "generic" });
|
||||
expect(sentBody).toEqual({ systemPrompt: "summarize", prompt: "<conversation>hello</conversation>" });
|
||||
});
|
||||
});
|
||||
|
||||
describe("compact() remote compaction failure handling", () => {
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
@@ -771,6 +842,53 @@ describe("compact() remote compaction failure handling", () => {
|
||||
expect(completeSpy).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
test("uses configured chat completions endpoints for openai-completions remote compaction", async () => {
|
||||
const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local fallback"));
|
||||
const preparation = makePreparation();
|
||||
preparation.settings = {
|
||||
...preparation.settings,
|
||||
remoteEndpoint: "http://127.0.0.1:8001/v1/chat/completions",
|
||||
remoteStreamingV2Enabled: false,
|
||||
};
|
||||
const model = buildModel({
|
||||
id: "catalog-selection-id",
|
||||
name: "Qwopus 3.6 35B-A3B Coder",
|
||||
requestModelId: "provider-wire-id",
|
||||
api: "openai-completions",
|
||||
provider: "local-llama",
|
||||
baseUrl: "http://127.0.0.1:8001/v1",
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 131072,
|
||||
maxTokens: 4096,
|
||||
});
|
||||
const requestBodies: unknown[] = [];
|
||||
const fetchMock: FetchImpl = async (_input, init) => {
|
||||
if (typeof init?.body !== "string") throw new Error("missing remote compaction request body");
|
||||
requestBodies.push(JSON.parse(init.body) as unknown);
|
||||
expect(new Headers(init.headers).get("authorization")).toBe("Bearer local-key");
|
||||
const summary = requestBodies.length === 1 ? "remote history summary" : "remote short summary";
|
||||
return new Response(JSON.stringify({ choices: [{ message: { content: summary } }] }), {
|
||||
headers: { "content-type": "application/json" },
|
||||
});
|
||||
};
|
||||
|
||||
const result = await compact(preparation, model, "local-key", undefined, undefined, {
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
expect(result.summary).toContain("remote history summary");
|
||||
expect(result.shortSummary).toBe("remote short summary");
|
||||
expect(completeSpy).not.toHaveBeenCalled();
|
||||
expect(requestBodies).toHaveLength(2);
|
||||
expect(requestBodies[0]).toMatchObject({
|
||||
model: "provider-wire-id",
|
||||
messages: [{ role: "system" }, { role: "user", content: expect.stringContaining("long history") }],
|
||||
stream: false,
|
||||
});
|
||||
});
|
||||
|
||||
test("remote compact server failure without abort still falls back to local summarization", async () => {
|
||||
const completeSpy = vi.spyOn(ai, "completeSimple").mockResolvedValue(localSummaryMessage("local summary"));
|
||||
const fetchMock: FetchImpl = async () =>
|
||||
|
||||
@@ -2,6 +2,22 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Added
|
||||
|
||||
- Added `AssistantMessage.toolCallAbortMessages` for per-tool placeholder labels on aborted assistant turns ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Anthropic replay 400s (`tool_use ids were found without tool_result blocks immediately after`) when a persisted assistant turn carries content after a completed tool call — such as a mid-turn `server-side-fallback` handoff (fallback block plus continued text/tool calls after the primary model's `tool_use`) or trailing text from cross-provider replays — by stable-partitioning assistant content so all `tool_use` blocks trail the non-`tool_use` chain. ([#4781](https://github.com/can1357/oh-my-pi/issues/4781), [#544](https://github.com/can1357/oh-my-pi/issues/544))
|
||||
- Fixed access-token-only OAuth credentials attempting token refresh with an empty refresh token after expiry.
|
||||
- Fixed gateway usage-limit retries falling through to cross-provider model fallback before trying a sibling credential from the same provider.
|
||||
- Fixed Codex usage-limit rotation treating Plus and K-12 accounts as separate quota groups for shared 5-hour/7-day windows.
|
||||
- Fixed OpenAI Responses streams that end with `response.done` being misclassified as premature stream closures.
|
||||
- Fixed OpenCode Go `/login` credentials being shadowed by an existing `OPENCODE_API_KEY` env fallback after switching accounts. ([#4688](https://github.com/can1357/oh-my-pi/issues/4688))
|
||||
- Fixed OpenAI Codex WebSocket continuations to treat proxy stale-anchor codes such as `codex_previous_response_stale` as an expired `previous_response_id` chain — same recovery class as the OpenAI-standard `previous_response_not_found` — so the turn is retried with full context instead of surfacing the error to the user ([#4624](https://github.com/can1357/oh-my-pi/issues/4624)).
|
||||
- Fixed Azure Foundry Anthropic utility requests to omit the structured-output beta whenever strict tools are disabled, preventing `structured_outputs not supported in your workspace` failures for Sonnet 5 compaction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)).
|
||||
- Fixed OAuth `launchUrl` advertisement for flows whose redirect never returns to the local callback server: custom-scheme redirects (e.g. GitLab Duo's `vscode://` URI, which `new URL` parses without complaint) and fixed non-loopback hosts no longer receive a `http://localhost:<port>/launch` copy target that misrepresents the callback endpoint and resolves nowhere for remote users.
|
||||
|
||||
## [16.3.11] - 2026-07-06
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -233,6 +233,7 @@ async function refreshGatewayApiKeyAfterAuthError(
|
||||
retryAfterMs,
|
||||
baseUrl: model.baseUrl,
|
||||
modelId: model.id,
|
||||
apiKey: oldKey,
|
||||
signal,
|
||||
});
|
||||
logger.debug("auth-gateway retrying provider request after usage-limit block", {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
import type { OAuthAccess } from "./auth-storage";
|
||||
import * as AIError from "./error";
|
||||
import { isAuthRetryableError } from "./error/auth-classify";
|
||||
import { isUsageLimit } from "./error/flags";
|
||||
|
||||
/**
|
||||
* Context passed to an {@link ApiKeyResolver} on each resolution attempt.
|
||||
@@ -23,6 +24,8 @@ export interface ApiKeyResolveContext {
|
||||
lastChance: boolean;
|
||||
/** The auth error that triggered this re-resolution, or `undefined` on the initial resolve. */
|
||||
error: unknown;
|
||||
/** Bearer used by the failed attempt, when the caller can expose it. */
|
||||
previousKey?: string;
|
||||
/** Caller cancel signal, threaded into any credential refresh / rotation work. */
|
||||
signal?: AbortSignal;
|
||||
}
|
||||
@@ -87,9 +90,11 @@ export async function resolveRetryKey(
|
||||
lastChance: boolean,
|
||||
error: unknown,
|
||||
signal?: AbortSignal,
|
||||
previousKey?: string,
|
||||
): Promise<string | undefined> {
|
||||
try {
|
||||
return (await resolver({ lastChance, error, signal })) || undefined;
|
||||
const rotateSibling = lastChance || (!lastChance && isUsageLimit(error));
|
||||
return (await resolver({ lastChance: rotateSibling, error, signal, previousKey })) || undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
@@ -136,7 +141,7 @@ export async function withAuth<T>(
|
||||
}
|
||||
|
||||
for (let i = 0; i < AUTH_RETRY_STEPS.length; i++) {
|
||||
const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal);
|
||||
const nextKey = await resolveRetryKey(resolver, AUTH_RETRY_STEPS[i]!, lastError, signal, lastKey);
|
||||
if (nextKey === undefined || nextKey === lastKey) continue;
|
||||
lastKey = nextKey;
|
||||
try {
|
||||
|
||||
+104
-36
@@ -66,6 +66,7 @@ const USAGE_RANKING_METRIC_EPSILON = 1e-9;
|
||||
export type ApiKeyCredential = {
|
||||
type: "api_key";
|
||||
key: string;
|
||||
source?: "login";
|
||||
};
|
||||
|
||||
export type OAuthCredential = {
|
||||
@@ -1390,12 +1391,14 @@ export class AuthStorage {
|
||||
|
||||
const credentialId = this.#getStoredCredentials(provider)[credentialIndex]?.id;
|
||||
if (credentialId === undefined) return blockedUntil;
|
||||
const persistedGlobalBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, "");
|
||||
if (
|
||||
persistedGlobalBlockedUntil !== undefined &&
|
||||
(blockedUntil === undefined || persistedGlobalBlockedUntil > blockedUntil)
|
||||
) {
|
||||
blockedUntil = persistedGlobalBlockedUntil;
|
||||
if (!blockScope || provider !== "openai-codex") {
|
||||
const persistedGlobalBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, "");
|
||||
if (
|
||||
persistedGlobalBlockedUntil !== undefined &&
|
||||
(blockedUntil === undefined || persistedGlobalBlockedUntil > blockedUntil)
|
||||
) {
|
||||
blockedUntil = persistedGlobalBlockedUntil;
|
||||
}
|
||||
}
|
||||
if (blockScope) {
|
||||
const persistedScopedBlockedUntil = this.#readPersistedCredentialBlock(credentialId, providerKey, blockScope);
|
||||
@@ -1554,13 +1557,14 @@ export class AuthStorage {
|
||||
provider: string,
|
||||
type: T,
|
||||
sessionId?: string,
|
||||
filter?: (credential: AuthCredential) => boolean,
|
||||
): { credential: Extract<AuthCredential, { type: T }>; index: number } | undefined {
|
||||
const credentials = this.#getCredentialsForProvider(provider)
|
||||
.map((credential, index) => ({ credential, index }))
|
||||
.filter(
|
||||
(entry): entry is { credential: Extract<AuthCredential, { type: T }>; index: number } =>
|
||||
entry.credential.type === type,
|
||||
);
|
||||
.filter((entry): entry is { credential: Extract<AuthCredential, { type: T }>; index: number } => {
|
||||
if (entry.credential.type !== type) return false;
|
||||
return filter?.(entry.credential) ?? true;
|
||||
});
|
||||
|
||||
if (credentials.length === 0) return undefined;
|
||||
if (credentials.length === 1) return credentials[0];
|
||||
@@ -1851,8 +1855,8 @@ export class AuthStorage {
|
||||
/**
|
||||
* Classify where a provider's auth comes from, following the same precedence
|
||||
* as {@link AuthStorage.getApiKey}: runtime override → config override →
|
||||
* stored OAuth → env var → stored api_key → fallback resolver. Returns
|
||||
* undefined when no auth is configured.
|
||||
* stored OAuth → login-stored api_key → env var → stored api_key →
|
||||
* fallback resolver. Returns undefined when no auth is configured.
|
||||
*
|
||||
* Compact, structured counterpart to {@link describeCredentialSource}.
|
||||
*/
|
||||
@@ -1861,6 +1865,9 @@ export class AuthStorage {
|
||||
if (this.#configOverrides.has(provider)) return { kind: "config" };
|
||||
const stored = this.#getCredentialsForProvider(provider);
|
||||
if (stored.some(credential => credential.type === "oauth")) return { kind: "oauth" };
|
||||
if (stored.some(credential => credential.type === "api_key" && credential.source === "login")) {
|
||||
return { kind: "api_key" };
|
||||
}
|
||||
if (getEnvApiKey(provider)) return { kind: "env", envVar: getEnvApiKeyName(provider) };
|
||||
if (stored.some(credential => credential.type === "api_key")) return { kind: "api_key" };
|
||||
if (this.#fallbackResolver?.(provider)) return { kind: "fallback" };
|
||||
@@ -2005,7 +2012,7 @@ export class AuthStorage {
|
||||
if (!result) {
|
||||
return;
|
||||
}
|
||||
const newCredential: ApiKeyCredential = { type: "api_key", key: result };
|
||||
const newCredential: ApiKeyCredential = { type: "api_key", key: result, source: "login" };
|
||||
const stored = this.#store.upsertAuthCredentialRemote
|
||||
? await this.#store.upsertAuthCredentialRemote(provider, newCredential)
|
||||
: this.#store.upsertAuthCredentialForProvider(provider, newCredential);
|
||||
@@ -3057,9 +3064,20 @@ export class AuthStorage {
|
||||
async markUsageLimitReached(
|
||||
provider: string,
|
||||
sessionId: string | undefined,
|
||||
options?: { retryAfterMs?: number; baseUrl?: string; modelId?: string; signal?: AbortSignal },
|
||||
options?: { retryAfterMs?: number; baseUrl?: string; modelId?: string; apiKey?: string; signal?: AbortSignal },
|
||||
): Promise<UsageLimitMarkResult> {
|
||||
const sessionCredential = this.#getSessionCredential(provider, sessionId);
|
||||
let sessionCredential: { type: AuthCredential["type"]; index: number } | undefined;
|
||||
if (options?.apiKey) {
|
||||
const stored = this.#getStoredCredentials(provider);
|
||||
for (let index = 0; index < stored.length; index++) {
|
||||
const entry = stored[index];
|
||||
if (entry && (await this.#credentialMatchesApiKey(entry.credential, options.apiKey))) {
|
||||
sessionCredential = { type: entry.credential.type, index };
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
sessionCredential ??= this.#getSessionCredential(provider, sessionId);
|
||||
if (!sessionCredential) return { switched: false };
|
||||
|
||||
const providerKey = this.#getProviderTypeKey(provider, sessionCredential.type);
|
||||
@@ -3387,12 +3405,21 @@ export class AuthStorage {
|
||||
const checkUsage = strategy !== undefined && (credentials.length > 1 || requiresProModel);
|
||||
const sessionCredential = this.#getSessionCredential(provider, sessionId);
|
||||
const sessionPreferredIndex = sessionCredential?.type === "oauth" ? sessionCredential.index : undefined;
|
||||
const sessionPreferredCredential =
|
||||
sessionPreferredIndex !== undefined
|
||||
? credentials.find(entry => entry.index === sessionPreferredIndex)?.credential
|
||||
: undefined;
|
||||
const sessionPreferredCanRefreshOrUse =
|
||||
sessionPreferredCredential !== undefined &&
|
||||
(sessionPreferredCredential.refresh.trim().length > 0 ||
|
||||
Date.now() + OAUTH_REFRESH_SKEW_MS < sessionPreferredCredential.expires);
|
||||
// Skip ranking only when the session already has a working preferred credential — re-ranking
|
||||
// mid-session causes account switches that cold-start the server-side prompt cache. New sessions
|
||||
// (no preference) and sessions whose preferred is blocked still rank, so we pick the account
|
||||
// with the most headroom proactively and fall back intelligently when rate-limited.
|
||||
const sessionPreferredIsAvailable =
|
||||
sessionPreferredIndex !== undefined &&
|
||||
sessionPreferredCanRefreshOrUse &&
|
||||
!this.#isCredentialBlocked(provider, providerKey, sessionPreferredIndex, blockScope);
|
||||
const shouldRank = checkUsage && (!sessionPreferredIsAvailable || requiresProModel);
|
||||
const rankingOrder = shouldRank && sessionId ? credentials.map((_credential, index) => index) : order;
|
||||
@@ -3899,8 +3926,8 @@ export class AuthStorage {
|
||||
return configKey;
|
||||
}
|
||||
|
||||
// Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored
|
||||
// static api_key (which may be a stale broker-migrated copy) as a last resort.
|
||||
// Precedence: a deliberate OAuth/login credential wins, then an explicit env var,
|
||||
// then a stored static api_key (which may be a stale broker-migrated copy) as a last resort.
|
||||
const oauthSelection = this.#selectCredentialByType(provider, "oauth");
|
||||
if (oauthSelection) {
|
||||
const expiresAt = oauthSelection.credential.expires;
|
||||
@@ -3916,6 +3943,16 @@ export class AuthStorage {
|
||||
}
|
||||
}
|
||||
|
||||
const loginApiKeySelection = this.#selectCredentialByType(
|
||||
provider,
|
||||
"api_key",
|
||||
undefined,
|
||||
credential => credential.type === "api_key" && credential.source === "login",
|
||||
);
|
||||
if (loginApiKeySelection) {
|
||||
return this.#configValueResolver(loginApiKeySelection.credential.key);
|
||||
}
|
||||
|
||||
const envKey = getEnvApiKey(provider);
|
||||
if (envKey) return envKey;
|
||||
|
||||
@@ -3933,9 +3970,10 @@ export class AuthStorage {
|
||||
* 1. Runtime override (CLI --api-key)
|
||||
* 2. Config override (models.yml `providers.<name>.apiKey`)
|
||||
* 3. OAuth token from storage (auto-refreshed)
|
||||
* 4. Environment variable
|
||||
* 5. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins
|
||||
* 6. Fallback resolver (models.yml custom providers, last-resort)
|
||||
* 4. API key persisted by a successful `/login`
|
||||
* 5. Environment variable
|
||||
* 6. Stored API key (e.g. a broker-migrated copy) — last resort, so an explicit env var wins
|
||||
* 7. Fallback resolver (models.yml custom providers, last-resort)
|
||||
*/
|
||||
async getApiKey(provider: string, sessionId?: string, options?: AuthApiKeyOptions): Promise<string | undefined> {
|
||||
// Runtime override takes highest priority
|
||||
@@ -3954,13 +3992,24 @@ export class AuthStorage {
|
||||
return configKey;
|
||||
}
|
||||
|
||||
// Precedence: a deliberate OAuth login wins, then an explicit env var, then a stored
|
||||
// static api_key (which may be a stale broker-migrated copy) as a last resort.
|
||||
// Precedence: a deliberate OAuth/login credential wins, then an explicit env var,
|
||||
// then a stored static api_key (which may be a stale broker-migrated copy) as a last resort.
|
||||
const oauthResolved = await this.#resolveOAuthSelection(provider, sessionId, options);
|
||||
if (oauthResolved) {
|
||||
return oauthResolved.apiKey;
|
||||
}
|
||||
|
||||
const loginApiKeySelection = this.#selectCredentialByType(
|
||||
provider,
|
||||
"api_key",
|
||||
sessionId,
|
||||
credential => credential.type === "api_key" && credential.source === "login",
|
||||
);
|
||||
if (loginApiKeySelection) {
|
||||
this.#recordSessionCredential(provider, sessionId, "api_key", loginApiKeySelection.index);
|
||||
return this.#configValueResolver(loginApiKeySelection.credential.key);
|
||||
}
|
||||
|
||||
// Past OAuth: the session sticky (if any) is stale — the request authenticates via
|
||||
// env/api_key/fallback, not OAuth, so clear it now so getOAuthAccountId() correctly
|
||||
// suppresses account_uuid for this session.
|
||||
@@ -3969,7 +4018,12 @@ export class AuthStorage {
|
||||
const envKey = getEnvApiKey(provider);
|
||||
if (envKey) return envKey;
|
||||
|
||||
const apiKeySelection = this.#selectCredentialByType(provider, "api_key", sessionId);
|
||||
const apiKeySelection = this.#selectCredentialByType(
|
||||
provider,
|
||||
"api_key",
|
||||
sessionId,
|
||||
credential => credential.type !== "api_key" || credential.source !== "login",
|
||||
);
|
||||
if (apiKeySelection) {
|
||||
this.#recordSessionCredential(provider, sessionId, "api_key", apiKeySelection.index);
|
||||
return this.#configValueResolver(apiKeySelection.credential.key);
|
||||
@@ -4381,7 +4435,7 @@ export class AuthStorage {
|
||||
async rotateSessionCredential(
|
||||
provider: string,
|
||||
sessionId: string | undefined,
|
||||
options?: { error?: unknown; modelId?: string; signal?: AbortSignal },
|
||||
options?: { error?: unknown; modelId?: string; apiKey?: string; signal?: AbortSignal },
|
||||
): Promise<boolean> {
|
||||
const sessionCredential = this.#getSessionCredential(provider, sessionId);
|
||||
if (!sessionCredential) return false;
|
||||
@@ -4393,6 +4447,7 @@ export class AuthStorage {
|
||||
return (
|
||||
await this.markUsageLimitReached(provider, sessionId, {
|
||||
modelId: options?.modelId,
|
||||
apiKey: options?.apiKey,
|
||||
signal: options?.signal,
|
||||
})
|
||||
).switched;
|
||||
@@ -4674,9 +4729,10 @@ export class AuthStorage {
|
||||
* 1. Runtime override (`--api-key`).
|
||||
* 2. Config override (`models.yml` `providers.<name>.apiKey`).
|
||||
* 3. Stored OAuth credential.
|
||||
* 4. Env var — overrides a stored static api_key (e.g. a stale broker copy).
|
||||
* 5. Stored api_key credential.
|
||||
* 6. Fallback resolver.
|
||||
* 4. API key persisted by a successful `/login`.
|
||||
* 5. Env var — overrides a stored static api_key (e.g. a stale broker copy).
|
||||
* 6. Stored api_key credential.
|
||||
* 7. Fallback resolver.
|
||||
*
|
||||
* The string is purely informational; consumers must not parse it.
|
||||
*/
|
||||
@@ -4691,14 +4747,16 @@ export class AuthStorage {
|
||||
const baseLabel = this.#sourceLabel ?? "local store";
|
||||
const stored = this.#getStoredCredentials(provider);
|
||||
const session = sessionId ? this.#sessionLastCredential.get(provider)?.get(sessionId) : undefined;
|
||||
// Describe the stored credential of a given type, honoring the session sticky index.
|
||||
const describeStored = (type: AuthCredential["type"]): string | undefined => {
|
||||
const describeStored = (
|
||||
type: AuthCredential["type"],
|
||||
filter?: (credential: AuthCredential) => boolean,
|
||||
): string | undefined => {
|
||||
const typed = stored
|
||||
.map((entry, index) => ({ entry, index }))
|
||||
.filter(({ entry }) => entry.credential.type === type);
|
||||
.filter(({ entry }) => entry.credential.type === type && (filter?.(entry.credential) ?? true));
|
||||
if (typed.length === 0) return undefined;
|
||||
const index = session?.type === type ? session.index : typed[0].index;
|
||||
const chosen = stored[index] ?? typed[0].entry;
|
||||
const sticky = session?.type === type ? typed.find(entry => entry.index === session.index) : undefined;
|
||||
const chosen = sticky?.entry ?? typed[0].entry;
|
||||
const credential = chosen.credential;
|
||||
const identity =
|
||||
credential.type === "oauth"
|
||||
@@ -4707,11 +4765,19 @@ export class AuthStorage {
|
||||
return `${baseLabel} · ${type} #${chosen.id} (${identity})`;
|
||||
};
|
||||
|
||||
// A deliberate OAuth login wins; then an explicit env var; then a stored static api_key.
|
||||
// Deliberate login credentials win; then an explicit env var; then a stored static api_key.
|
||||
const oauthSource = describeStored("oauth");
|
||||
if (oauthSource) return oauthSource;
|
||||
const loginApiKeySource = describeStored(
|
||||
"api_key",
|
||||
credential => credential.type === "api_key" && credential.source === "login",
|
||||
);
|
||||
if (loginApiKeySource) return loginApiKeySource;
|
||||
if (getEnvApiKey(provider)) return `env (over ${baseLabel})`;
|
||||
const apiKeySource = describeStored("api_key");
|
||||
const apiKeySource = describeStored(
|
||||
"api_key",
|
||||
credential => credential.type !== "api_key" || credential.source !== "login",
|
||||
);
|
||||
if (apiKeySource) return apiKeySource;
|
||||
if (this.#fallbackResolver?.(provider) !== undefined) return "fallback resolver";
|
||||
return undefined;
|
||||
@@ -4776,9 +4842,10 @@ function normalizeStoredIdentityKey(identityKey: string | null | undefined): str
|
||||
|
||||
function serializeCredential(provider: string, credential: AuthCredential): SerializedCredentialRecord | null {
|
||||
if (credential.type === "api_key") {
|
||||
const data = credential.source === "login" ? { key: credential.key, source: "login" } : { key: credential.key };
|
||||
return {
|
||||
credentialType: "api_key",
|
||||
data: JSON.stringify({ key: credential.key }),
|
||||
data: JSON.stringify(data),
|
||||
identityKey: null,
|
||||
};
|
||||
}
|
||||
@@ -4806,7 +4873,8 @@ function deserializeCredential(row: AuthRow): AuthCredential | null {
|
||||
if (row.credential_type === "api_key") {
|
||||
const data = parsed as Record<string, unknown>;
|
||||
if (typeof data.key === "string") {
|
||||
return { type: "api_key", key: data.key };
|
||||
const source = data.source === "login" ? "login" : undefined;
|
||||
return source ? { type: "api_key", key: data.key, source } : { type: "api_key", key: data.key };
|
||||
}
|
||||
}
|
||||
if (row.credential_type === "oauth") {
|
||||
|
||||
@@ -127,12 +127,13 @@ export function buildBetaHeader(baseBetas: readonly string[], extraBetas: readon
|
||||
|
||||
const midConversationSystemBeta = "mid-conversation-system-2026-04-07";
|
||||
const contextManagementBeta = "context-management-2025-06-27";
|
||||
const structuredOutputsBeta = "structured-outputs-2025-12-15";
|
||||
const claudeCodeUtilityBetaDefaults = [
|
||||
"oauth-2025-04-20",
|
||||
"interleaved-thinking-2025-05-14",
|
||||
contextManagementBeta,
|
||||
"prompt-caching-scope-2026-01-05",
|
||||
"structured-outputs-2025-12-15",
|
||||
structuredOutputsBeta,
|
||||
] as const;
|
||||
const claudeCodeAgentBetaDefaults = [
|
||||
"claude-code-20250219",
|
||||
@@ -159,10 +160,12 @@ function buildClaudeCodeBetas(
|
||||
agentRequest: boolean,
|
||||
thinkingRequest: boolean,
|
||||
redactThinking: boolean,
|
||||
disableStrictTools = false,
|
||||
): readonly string[] {
|
||||
if (!agentRequest && !redactThinking) return claudeCodeUtilityBetaDefaults;
|
||||
if (!agentRequest && !redactThinking && !disableStrictTools) return claudeCodeUtilityBetaDefaults;
|
||||
const betas: string[] = [];
|
||||
for (const beta of agentRequest ? claudeCodeAgentBetaDefaults : claudeCodeUtilityBetaDefaults) {
|
||||
if (disableStrictTools && beta === structuredOutputsBeta) continue;
|
||||
betas.push(beta);
|
||||
// Match CC's header order: redact-thinking immediately follows interleaved-thinking.
|
||||
if (redactThinking && beta === interleavedThinkingBeta) betas.push(redactThinkingBeta);
|
||||
@@ -1098,6 +1101,7 @@ export type AnthropicClientOptionsArgs = {
|
||||
hasTools?: boolean;
|
||||
thinkingEnabled?: boolean;
|
||||
thinkingDisplay?: AnthropicThinkingDisplay;
|
||||
disableStrictTools?: boolean;
|
||||
fetch?: FetchImpl;
|
||||
claudeCodeSessionId?: string;
|
||||
};
|
||||
@@ -1833,6 +1837,7 @@ const streamAnthropicOnce = (
|
||||
thinkingDisplay: options?.thinkingDisplay,
|
||||
fetch: options?.fetch,
|
||||
claudeCodeSessionId: options?.sessionId ?? extractClaudeMetadataSessionId(options?.metadata?.user_id),
|
||||
disableStrictTools,
|
||||
});
|
||||
client = created.client;
|
||||
isOAuthToken = created.isOAuthToken;
|
||||
@@ -2648,8 +2653,10 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
thinkingDisplay,
|
||||
isOAuth,
|
||||
claudeCodeSessionId,
|
||||
disableStrictTools: disableStrictToolsOverride,
|
||||
} = args;
|
||||
const compat = model.compat;
|
||||
const disableStrictTools = disableStrictToolsOverride ?? compat.disableStrictTools;
|
||||
const needsInterleavedBeta = interleavedThinking && !model.thinking?.supportsDisplay;
|
||||
const needsFineGrainedToolStreamingBeta = hasTools && !compat.supportsEagerToolInputStreaming;
|
||||
const oauthToken = isOAuth ?? isAnthropicOAuthToken(apiKey);
|
||||
@@ -2722,7 +2729,12 @@ export function buildAnthropicClientOptions(args: AnthropicClientOptionsArgs): A
|
||||
isCloudflareAiGateway: model.provider === "cloudflare-ai-gateway",
|
||||
claudeCodeSessionId,
|
||||
claudeCodeBetas: oauthToken
|
||||
? buildClaudeCodeBetas(hasTools || thinkingEnabled, thinkingEnabled, thinkingDisplay === "omitted")
|
||||
? buildClaudeCodeBetas(
|
||||
hasTools || thinkingEnabled,
|
||||
thinkingEnabled,
|
||||
thinkingDisplay === "omitted",
|
||||
disableStrictTools,
|
||||
)
|
||||
: [],
|
||||
});
|
||||
|
||||
@@ -3529,6 +3541,40 @@ export function convertAnthropicMessages(
|
||||
});
|
||||
}
|
||||
}
|
||||
// Anthropic's replay validator rejects any non-`tool_use` block that
|
||||
// appears after a `tool_use` inside an assistant turn (400:
|
||||
// "tool_use ids were found without tool_result blocks immediately
|
||||
// after: <id>"). A persisted turn can violate this when a mid-turn
|
||||
// server-side fallback handoff lands after the primary model already
|
||||
// emitted a tool_use — the replayed content is then e.g.
|
||||
// [thinking, text, tool_use, fallback, text, tool_use] — and also for
|
||||
// the older cross-provider [text, tool_use, text] shape (issue #544).
|
||||
// Stable-partition into [...non-tool_use, ...tool_use], preserving each
|
||||
// side's relative order: the non-tool_use chain (thinking → text →
|
||||
// fallback → text) carries thinking signatures and the fallback
|
||||
// boundary marker whose order Anthropic verifies, while tool_use blocks
|
||||
// are unsigned and safe to defer to the tail. Fast-path untouched when
|
||||
// already in order so prompt-cache prefixes stay byte-identical.
|
||||
let sawToolUse = false;
|
||||
let needsPartition = false;
|
||||
for (const block of blocks) {
|
||||
if (block.type === "tool_use") {
|
||||
sawToolUse = true;
|
||||
} else if (sawToolUse) {
|
||||
needsPartition = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (needsPartition) {
|
||||
const nonToolUse: ContentBlockParam[] = [];
|
||||
const toolUse: ContentBlockParam[] = [];
|
||||
for (const block of blocks) {
|
||||
if (block.type === "tool_use") toolUse.push(block);
|
||||
else nonToolUse.push(block);
|
||||
}
|
||||
blocks.length = 0;
|
||||
blocks.push(...nonToolUse, ...toolUse);
|
||||
}
|
||||
if (blocks.length === 0) continue;
|
||||
params.push({
|
||||
role: "assistant",
|
||||
|
||||
@@ -1232,12 +1232,27 @@ function getOutputBlockStartEventType(block: CodexOutputBlock): "thinking_start"
|
||||
return "toolcall_start";
|
||||
}
|
||||
|
||||
const CODEX_STALE_PREVIOUS_RESPONSE_CODES: Record<string, true> = {
|
||||
// OpenAI-standard code for an expired/missing `previous_response_id` chain.
|
||||
previous_response_not_found: true,
|
||||
// Proxy-specific: upstream response anchor expired. Same recovery class —
|
||||
// retry the turn with full context and no `previous_response_id`.
|
||||
codex_previous_response_stale: true,
|
||||
};
|
||||
|
||||
function isCodexStalePreviousResponseError(error: unknown): boolean {
|
||||
if (error instanceof CodexProviderStreamError) return error.code === "previous_response_not_found";
|
||||
if (!(error instanceof Error)) return false;
|
||||
if ((error as { code?: string }).code === "previous_response_not_found") return true;
|
||||
// "unsupported": the backend intermittently rejects the parameter outright
|
||||
// with `{"detail":"Unsupported parameter: previous_response_id"}` (no
|
||||
if (
|
||||
"code" in error &&
|
||||
typeof error.code === "string" &&
|
||||
Object.hasOwn(CODEX_STALE_PREVIOUS_RESPONSE_CODES, error.code)
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
// Message-based fallback for providers/proxies that report the condition
|
||||
// without a canonical code. Also covers "unsupported": the backend
|
||||
// intermittently rejects the parameter outright with
|
||||
// `{"detail":"Unsupported parameter: previous_response_id"}` (no
|
||||
// `error.code`); treat it like a stale chain so the turn replays with full
|
||||
// context instead of surfacing the 400.
|
||||
return (
|
||||
|
||||
@@ -668,7 +668,7 @@ const streamOpenAIResponsesOnce = (
|
||||
}
|
||||
|
||||
// Detect premature stream closure: the HTTP stream ended without the
|
||||
// provider sending `response.completed` or `response.incomplete`.
|
||||
// provider sending a recognized terminal response event.
|
||||
// Custom/proxy providers may drop the connection mid-stream; without
|
||||
// this guard the incomplete output is silently surfaced as a successful
|
||||
// "stop".
|
||||
|
||||
@@ -80,6 +80,7 @@ import {
|
||||
import type { ChatCompletionCreateParamsStreaming } from "./openai-chat-wire";
|
||||
import type { InputItem } from "./openai-codex/request-transformer";
|
||||
import type {
|
||||
Response as OpenAIResponse,
|
||||
ResponseContentPartAddedEvent,
|
||||
ResponseCreateParamsStreaming,
|
||||
ResponseCustomToolCall,
|
||||
@@ -1798,13 +1799,25 @@ export function finalizeCustomToolCallInputDone(block: ResponsesToolCallBlock, i
|
||||
block.arguments = { input };
|
||||
}
|
||||
|
||||
type OpenAIResponsesTerminalStreamEvent =
|
||||
| Extract<ResponseStreamEvent, { type: "response.completed" | "response.incomplete" }>
|
||||
| { type: "response.done"; response?: Partial<OpenAIResponse> };
|
||||
|
||||
function getOpenAIResponsesTerminalEvent(event: ResponseStreamEvent): OpenAIResponsesTerminalStreamEvent | undefined {
|
||||
const type = (event as { type?: unknown }).type;
|
||||
return type === "response.completed" || type === "response.incomplete" || type === "response.done"
|
||||
? (event as OpenAIResponsesTerminalStreamEvent)
|
||||
: undefined;
|
||||
}
|
||||
|
||||
export interface ProcessResponsesStreamOptions {
|
||||
onFirstToken?: () => void;
|
||||
onOutputItemDone?: (item: ResponseOutputItem) => void;
|
||||
/**
|
||||
* Called when a terminal `response.completed` or `response.incomplete` event
|
||||
* is successfully processed. Only invoked on the successful-completion path;
|
||||
* thrown failure (`response.failed`) and cancellation paths never call this.
|
||||
* Called when a terminal `response.completed`, `response.incomplete`, or
|
||||
* `response.done` event is successfully processed. Only invoked on the
|
||||
* successful-completion path; thrown failure (`response.failed`) and
|
||||
* cancellation paths never call this.
|
||||
* Used by callers to detect premature stream closure (i.e. the stream ended
|
||||
* without a recognized terminal event).
|
||||
*/
|
||||
@@ -2039,6 +2052,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
let sawFirstToken = false;
|
||||
|
||||
for await (const event of openaiStream) {
|
||||
const terminalEvent = getOpenAIResponsesTerminalEvent(event);
|
||||
if (event.type === "response.created") {
|
||||
output.responseId = event.response.id;
|
||||
} else if (event.type === "response.output_item.added") {
|
||||
@@ -2297,8 +2311,8 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id, prefixedFunctionCallItemKey(item.call_id));
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
}
|
||||
} else if (event.type === "response.completed" || event.type === "response.incomplete") {
|
||||
const response = event.response;
|
||||
} else if (terminalEvent) {
|
||||
const response = terminalEvent.response;
|
||||
finalizePendingResponsesToolCalls(output);
|
||||
if (response?.id) {
|
||||
output.responseId = response.id;
|
||||
@@ -2336,7 +2350,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
}
|
||||
promoteResponsesToolUseStopReason(output, (response as { end_turn?: boolean } | undefined)?.end_turn);
|
||||
options?.onCompleted?.();
|
||||
// `response.completed`/`response.incomplete` is the last event of a
|
||||
// `response.completed`/`response.incomplete`/`response.done` is the last event of a
|
||||
// Responses stream. Stop pulling instead of waiting for the server to
|
||||
// close the connection: misbehaving providers keep the socket open
|
||||
// after the terminal event, which would park this loop until the idle
|
||||
|
||||
@@ -229,20 +229,32 @@ export abstract class OAuthCallbackFlow {
|
||||
|
||||
/**
|
||||
* Build the `/launch` URL served by the callback server bound to `port`, or
|
||||
* `undefined` when the configured `callbackPath` (or a `redirectUri` whose
|
||||
* pathname resolves to {@link LAUNCH_PATH}) would collide with the launch
|
||||
* route. Kept short (~30 chars) so UIs can advertise it as a
|
||||
* `undefined` when it must not be advertised:
|
||||
* - the configured `callbackPath` (or a `redirectUri` whose pathname
|
||||
* resolves to {@link LAUNCH_PATH}) would collide with the launch route;
|
||||
* - the flow's `redirectUri` never returns to this loopback server: fixed
|
||||
* non-loopback hosts, or custom schemes like GitLab Duo's `vscode://`
|
||||
* URI — which `new URL` parses without complaint, so a scheme/host check
|
||||
* is required, not just the parse failure path. Advertising a localhost
|
||||
* `/launch` target for such flows misrepresents the callback endpoint
|
||||
* and hands remote users a URL that resolves nowhere.
|
||||
* Kept short (~30 chars) so UIs can advertise it as a
|
||||
* viewport-truncation-safe copy target for the full authorization URL.
|
||||
*/
|
||||
#launchUrlIfSafe(port: number): string | undefined {
|
||||
if (this.callbackPath === LAUNCH_PATH) return undefined;
|
||||
if (this.redirectUri) {
|
||||
try {
|
||||
if (new URL(this.redirectUri).pathname === LAUNCH_PATH) return undefined;
|
||||
const parsed = new URL(this.redirectUri);
|
||||
if (parsed.protocol !== "http:" && parsed.protocol !== "https:") return undefined;
|
||||
if (parsed.hostname !== "localhost" && parsed.hostname !== "127.0.0.1" && parsed.hostname !== "[::1]") {
|
||||
return undefined;
|
||||
}
|
||||
if (parsed.pathname === LAUNCH_PATH) return undefined;
|
||||
} catch {
|
||||
// A non-parseable redirectUri (e.g. `vscode://...` handled elsewhere)
|
||||
// can't collide with an HTTP `/launch` route — fall through and
|
||||
// advertise the launch URL against the loopback server.
|
||||
// A redirectUri even WHATWG URL cannot parse certainly does not
|
||||
// return to this server — never advertise a launch URL for it.
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
return `http://${this.callbackHostname}:${port}${LAUNCH_PATH}`;
|
||||
|
||||
@@ -1106,7 +1106,13 @@ export function streamSimple<TApi extends Api>(
|
||||
// Caller aborted between attempts: don't mint a fresh token or fire
|
||||
// another doomed request — emit the captured failure instead.
|
||||
if (signal?.aborted) break;
|
||||
const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal);
|
||||
const nextKey = await resolveRetryKey(
|
||||
apiKeyResolver,
|
||||
AUTH_RETRY_STEPS[step]!,
|
||||
failure.error,
|
||||
signal,
|
||||
lastKey,
|
||||
);
|
||||
if (nextKey === undefined || nextKey === lastKey) continue;
|
||||
lastKey = nextKey;
|
||||
const isLastStep = step === AUTH_RETRY_STEPS.length - 1;
|
||||
|
||||
@@ -715,6 +715,8 @@ export interface AssistantMessage {
|
||||
stopReason: StopReason;
|
||||
stopDetails?: StopDetails | null;
|
||||
errorMessage?: string;
|
||||
/** Per-tool abort messages used when an aborted assistant turn needs different placeholder results per tool call. */
|
||||
toolCallAbortMessages?: Record<string, string>;
|
||||
/** HTTP status surfaced by the provider when the request failed. Populated by every provider's catch block alongside `errorMessage` so consumers (auth retry, telemetry, UI) can branch without regex-scraping the message. */
|
||||
errorStatus?: number;
|
||||
/** Structured machine-readable error classifier; see `utils/error-id.ts` for bit layout and helpers. */
|
||||
|
||||
@@ -285,8 +285,6 @@ function buildUsageLimit(args: {
|
||||
label: usageWindow.label,
|
||||
scope: {
|
||||
provider: "openai-codex",
|
||||
accountId: args.accountId,
|
||||
tier: args.planType,
|
||||
windowId: usageWindow.id,
|
||||
shared: true,
|
||||
},
|
||||
@@ -507,6 +505,9 @@ export const openaiCodexUsageProvider: UsageProvider = {
|
||||
const FIVE_HOUR_MS = 5 * 60 * 60 * 1000;
|
||||
|
||||
export const codexRankingStrategy: CredentialRankingStrategy = {
|
||||
blockScope() {
|
||||
return "shared";
|
||||
},
|
||||
findWindowLimits(report) {
|
||||
const findLimit = (key: "primary" | "secondary"): UsageLimit | undefined => {
|
||||
const direct = report.limits.find(l => l.id === `openai-codex:${key}`);
|
||||
|
||||
@@ -17,7 +17,8 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { convertAnthropicMessages, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import { AnthropicMessages } from "@oh-my-pi/pi-ai/providers/anthropic-client";
|
||||
import type { AssistantMessage, Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import type { MessageParam } from "@oh-my-pi/pi-ai/providers/anthropic-wire";
|
||||
import type { AssistantMessage, Context, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
const fableModel: Model<"anthropic-messages"> = buildModel({
|
||||
@@ -328,3 +329,255 @@ describe("anthropic fallback content-block replay policy", () => {
|
||||
expect(params[0]?.content).toEqual([{ type: "text", text: "continued" }]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("anthropic assistant replay block ordering (tool_use partition)", () => {
|
||||
// Anthropic's replay validator rejects any assistant turn that carries a
|
||||
// non-`tool_use` content block AFTER a `tool_use` block
|
||||
// (`messages.N: tool_use ids were found without tool_result blocks
|
||||
// immediately after`). This bites when a mid-turn server-side fallback
|
||||
// (`server-side-fallback-2026-06-01`) lands AFTER the primary model already
|
||||
// emitted a tool_use, leaving persisted content shaped like
|
||||
// [thinking, text, tool_use, fallback, text, tool_use]. The same validator
|
||||
// rejects the older cross-provider [text, tool_use, text] shape (issue
|
||||
// #544). `convertAnthropicMessages` defends against both by stable-
|
||||
// partitioning every assistant wire message: all non-tool_use blocks first
|
||||
// (original relative order), then all tool_use blocks (original relative
|
||||
// order). Already-valid messages must serialize byte-identically.
|
||||
|
||||
function assistant(content: AssistantMessage["content"]): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content,
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
model: "claude-fable-5",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp: 0,
|
||||
};
|
||||
}
|
||||
|
||||
function toolResult(id: string, name: string): ToolResultMessage {
|
||||
return {
|
||||
role: "toolResult",
|
||||
toolCallId: id,
|
||||
toolName: name,
|
||||
content: [{ type: "text", text: "ok" }],
|
||||
isError: false,
|
||||
timestamp: 0,
|
||||
};
|
||||
}
|
||||
|
||||
function assistantParam(params: MessageParam[]): MessageParam {
|
||||
const asst = params.filter(p => p.role === "assistant");
|
||||
if (asst.length !== 1 || !Array.isArray(asst[0]?.content)) {
|
||||
throw new Error("expected exactly one assistant param with structured content blocks");
|
||||
}
|
||||
return asst[0];
|
||||
}
|
||||
|
||||
it("mid-turn fallback repro: defers trailing tool_use, keeps the fallback marker in place", () => {
|
||||
const params = convertAnthropicMessages(
|
||||
[
|
||||
assistant([
|
||||
{ type: "thinking", thinking: "plan", thinkingSignature: "sig-1" },
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "toolCall", id: "call_a", name: "read", arguments: {} },
|
||||
{ type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } },
|
||||
{ type: "text", text: "after" },
|
||||
{ type: "toolCall", id: "call_b", name: "grep", arguments: {} },
|
||||
{ type: "toolCall", id: "call_c", name: "glob", arguments: {} },
|
||||
]),
|
||||
toolResult("call_a", "read"),
|
||||
toolResult("call_b", "grep"),
|
||||
toolResult("call_c", "glob"),
|
||||
{ role: "user", content: "next", timestamp: 0 },
|
||||
],
|
||||
fableModel,
|
||||
false,
|
||||
{ serverSideFallbackEnabled: true },
|
||||
);
|
||||
expect(assistantParam(params).content).toEqual([
|
||||
{ type: "thinking", thinking: "plan", signature: "sig-1" },
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } },
|
||||
{ type: "text", text: "after" },
|
||||
{ type: "tool_use", id: "call_a", name: "read", input: {} },
|
||||
{ type: "tool_use", id: "call_b", name: "grep", input: {} },
|
||||
{ type: "tool_use", id: "call_c", name: "glob", input: {} },
|
||||
]);
|
||||
});
|
||||
|
||||
it("opt-out drops the fallback marker but still defers tool_use to the tail", () => {
|
||||
const params = convertAnthropicMessages(
|
||||
[
|
||||
assistant([
|
||||
{ type: "thinking", thinking: "plan", thinkingSignature: "sig-1" },
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "toolCall", id: "call_a", name: "read", arguments: {} },
|
||||
{ type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } },
|
||||
{ type: "text", text: "after" },
|
||||
{ type: "toolCall", id: "call_b", name: "grep", arguments: {} },
|
||||
{ type: "toolCall", id: "call_c", name: "glob", arguments: {} },
|
||||
]),
|
||||
toolResult("call_a", "read"),
|
||||
toolResult("call_b", "grep"),
|
||||
toolResult("call_c", "glob"),
|
||||
{ role: "user", content: "next", timestamp: 0 },
|
||||
],
|
||||
fableModel,
|
||||
false,
|
||||
// serverSideFallbackEnabled omitted → fallback block dropped
|
||||
);
|
||||
expect(assistantParam(params).content).toEqual([
|
||||
{ type: "thinking", thinking: "plan", signature: "sig-1" },
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "text", text: "after" },
|
||||
{ type: "tool_use", id: "call_a", name: "read", input: {} },
|
||||
{ type: "tool_use", id: "call_b", name: "grep", input: {} },
|
||||
{ type: "tool_use", id: "call_c", name: "glob", input: {} },
|
||||
]);
|
||||
});
|
||||
|
||||
it("issue-544 family: text after a tool_use is reordered before the tool_use", () => {
|
||||
const params = convertAnthropicMessages(
|
||||
[
|
||||
assistant([
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "toolCall", id: "call_a", name: "read", arguments: {} },
|
||||
{ type: "text", text: "after" },
|
||||
]),
|
||||
toolResult("call_a", "read"),
|
||||
{ role: "user", content: "next", timestamp: 0 },
|
||||
],
|
||||
fableModel,
|
||||
false,
|
||||
{ serverSideFallbackEnabled: true },
|
||||
);
|
||||
expect(assistantParam(params).content).toEqual([
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "text", text: "after" },
|
||||
{ type: "tool_use", id: "call_a", name: "read", input: {} },
|
||||
]);
|
||||
});
|
||||
|
||||
it("identity fast-path: already-valid thinking→text→tool_use serializes in unchanged order", () => {
|
||||
const params = convertAnthropicMessages(
|
||||
[
|
||||
assistant([
|
||||
{ type: "thinking", thinking: "plan", thinkingSignature: "sig-1" },
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "toolCall", id: "call_a", name: "read", arguments: {} },
|
||||
]),
|
||||
toolResult("call_a", "read"),
|
||||
{ role: "user", content: "next", timestamp: 0 },
|
||||
],
|
||||
fableModel,
|
||||
false,
|
||||
{ serverSideFallbackEnabled: true },
|
||||
);
|
||||
expect(assistantParam(params).content).toEqual([
|
||||
{ type: "thinking", thinking: "plan", signature: "sig-1" },
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "tool_use", id: "call_a", name: "read", input: {} },
|
||||
]);
|
||||
});
|
||||
|
||||
it("interleaved signed thinking: signature-chain order preserved, tool_use deferred to the tail", () => {
|
||||
const params = convertAnthropicMessages(
|
||||
[
|
||||
assistant([
|
||||
{ type: "thinking", thinking: "first", thinkingSignature: "sig-1" },
|
||||
{ type: "toolCall", id: "call_a", name: "read", arguments: {} },
|
||||
{ type: "thinking", thinking: "second", thinkingSignature: "sig-2" },
|
||||
{ type: "toolCall", id: "call_b", name: "grep", arguments: {} },
|
||||
]),
|
||||
toolResult("call_a", "read"),
|
||||
toolResult("call_b", "grep"),
|
||||
{ role: "user", content: "next", timestamp: 0 },
|
||||
],
|
||||
fableModel,
|
||||
false,
|
||||
{ serverSideFallbackEnabled: true },
|
||||
);
|
||||
expect(assistantParam(params).content).toEqual([
|
||||
{ type: "thinking", thinking: "first", signature: "sig-1" },
|
||||
{ type: "thinking", thinking: "second", signature: "sig-2" },
|
||||
{ type: "tool_use", id: "call_a", name: "read", input: {} },
|
||||
{ type: "tool_use", id: "call_b", name: "grep", input: {} },
|
||||
]);
|
||||
});
|
||||
|
||||
it("partition is localized: all other wire messages serialize byte-identically", () => {
|
||||
// Prompt-cache contract: partitioning a poisoned assistant turn must be a
|
||||
// LOCAL rewrite of that turn's own content — every OTHER wire message
|
||||
// (the earlier valid assistant turn whose cached prefix must survive, its
|
||||
// tool_result, the poisoned turn's tool_results, and the trailing user
|
||||
// turn) must serialize to the exact same bytes. If the reorder leaked past
|
||||
// the turn boundary (mutated a shared/adjacent message) or the slow path
|
||||
// diverged from an already-ordered fast path, cached prefixes up to the
|
||||
// reordered turn would be invalidated. Two histories, identical except the
|
||||
// poisoned turn is pre-ordered to the partition result in (b): (a) fires
|
||||
// the partition, (b) takes the fast path. Bytes must match everywhere but
|
||||
// the poisoned param, which must converge to the same content either way.
|
||||
const validAssistant = assistant([
|
||||
{ type: "thinking", thinking: "plan", thinkingSignature: "sig-1" },
|
||||
{ type: "text", text: "before" },
|
||||
{ type: "toolCall", id: "call_v", name: "list", arguments: {} },
|
||||
]);
|
||||
const poisonedContent: AssistantMessage["content"] = [
|
||||
{ type: "text", text: "poison-a" },
|
||||
{ type: "toolCall", id: "call_a", name: "read", arguments: {} },
|
||||
{ type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } },
|
||||
{ type: "text", text: "poison-b" },
|
||||
{ type: "toolCall", id: "call_b", name: "grep", arguments: {} },
|
||||
];
|
||||
// Same blocks, hand-ordered to exactly what the stable partition emits:
|
||||
// non-tool_use chain first (order preserved), then the tool_use tail.
|
||||
const preOrderedContent: AssistantMessage["content"] = [
|
||||
{ type: "text", text: "poison-a" },
|
||||
{ type: "fallback", from: { model: "claude-fable-5" }, to: { model: "claude-opus-4-8" } },
|
||||
{ type: "text", text: "poison-b" },
|
||||
{ type: "toolCall", id: "call_a", name: "read", arguments: {} },
|
||||
{ type: "toolCall", id: "call_b", name: "grep", arguments: {} },
|
||||
];
|
||||
const history = (poisoned: AssistantMessage["content"]) => [
|
||||
{ role: "user", content: "start", timestamp: 0 } as const,
|
||||
validAssistant,
|
||||
toolResult("call_v", "list"),
|
||||
assistant(poisoned),
|
||||
toolResult("call_a", "read"),
|
||||
toolResult("call_b", "grep"),
|
||||
{ role: "user", content: "next", timestamp: 0 } as const,
|
||||
];
|
||||
|
||||
const a = convertAnthropicMessages(history(poisonedContent), fableModel, false, {
|
||||
serverSideFallbackEnabled: true,
|
||||
});
|
||||
const b = convertAnthropicMessages(history(preOrderedContent), fableModel, false, {
|
||||
serverSideFallbackEnabled: true,
|
||||
});
|
||||
|
||||
expect(a.length).toBe(b.length);
|
||||
const poisonedIdx = a.findIndex(
|
||||
p =>
|
||||
p.role === "assistant" &&
|
||||
Array.isArray(p.content) &&
|
||||
p.content.some(block => block.type === "tool_use" && block.id === "call_a"),
|
||||
);
|
||||
expect(poisonedIdx).toBeGreaterThanOrEqual(0);
|
||||
for (let i = 0; i < a.length; i++) {
|
||||
if (i === poisonedIdx) continue;
|
||||
expect(JSON.stringify(a[i])).toBe(JSON.stringify(b[i]));
|
||||
}
|
||||
// Convergence: partition (a) and fast path (b) yield identical final content.
|
||||
expect(a[poisonedIdx]).toEqual(b[poisonedIdx]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -99,6 +99,28 @@ describe("withAuth", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("switches accounts before refreshing the same account on usage limits", async () => {
|
||||
const keys: string[] = [];
|
||||
const contexts: ApiKeyResolveContext[] = [];
|
||||
const result = await withAuth(
|
||||
ctx => {
|
||||
contexts.push(ctx);
|
||||
return ctx.error === undefined ? "k0" : ctx.lastChance ? "k2" : "k1";
|
||||
},
|
||||
async key => {
|
||||
keys.push(key);
|
||||
if (key === "k2") return "success";
|
||||
throw usageLimitError();
|
||||
},
|
||||
);
|
||||
expect(result).toBe("success");
|
||||
expect(keys).toEqual(["k0", "k2"]);
|
||||
expect(contexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([
|
||||
{ lastChance: false, hasError: false },
|
||||
{ lastChance: true, hasError: true },
|
||||
]);
|
||||
});
|
||||
|
||||
it("stops retrying when the resolver returns undefined", async () => {
|
||||
const keys: string[] = [];
|
||||
const original = authError();
|
||||
|
||||
@@ -39,9 +39,9 @@ function countCredentialRowsByDisabledState(dbPath: string, provider: string, di
|
||||
}
|
||||
|
||||
describe("AuthStorage api-key login upsert", () => {
|
||||
// A live env var now (correctly) overrides a stored static api_key. These tests verify that a
|
||||
// freshly stored api_key resolves through AuthStorage.getApiKey, so neutralize the env leg
|
||||
// entirely — this ignores every provider's ambient env key, not just the few set locally.
|
||||
// Most tests neutralize the env leg so ambient shell / ~/.env keys cannot
|
||||
// hide the stored credential behavior under test. Login-persisted API keys
|
||||
// have their own precedence coverage below.
|
||||
let tempDir = "";
|
||||
let dbPath = "";
|
||||
let store: SqliteAuthCredentialStore | null = null;
|
||||
@@ -49,9 +49,10 @@ describe("AuthStorage api-key login upsert", () => {
|
||||
let loginDeepSeekSpy: Mock<typeof deepseekModule.loginDeepSeek>;
|
||||
let loginKagiSpy: Mock<typeof kagiModule.loginKagi>;
|
||||
let loginOllamaCloudSpy: Mock<typeof ollamaCloudModule.loginOllamaCloud>;
|
||||
let getEnvApiKeySpy: Mock<typeof aiStream.getEnvApiKey>;
|
||||
|
||||
beforeEach(async () => {
|
||||
vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined);
|
||||
getEnvApiKeySpy = vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined);
|
||||
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "pi-ai-auth-api-key-login-"));
|
||||
dbPath = path.join(tempDir, "agent.db");
|
||||
store = await SqliteAuthCredentialStore.open(dbPath);
|
||||
@@ -118,8 +119,8 @@ describe("AuthStorage api-key login upsert", () => {
|
||||
|
||||
const credentials = store.listAuthCredentials("kagi");
|
||||
expect(credentials.map(entry => entry.credential)).toEqual([
|
||||
{ type: "api_key", key: "first-kagi-key" },
|
||||
{ type: "api_key", key: "second-kagi-key" },
|
||||
{ type: "api_key", key: "first-kagi-key", source: "login" },
|
||||
{ type: "api_key", key: "second-kagi-key", source: "login" },
|
||||
]);
|
||||
const rotatedKeys = [await authStorage.getApiKey("kagi"), await authStorage.getApiKey("kagi")].sort();
|
||||
expect(rotatedKeys).toEqual(["first-kagi-key", "second-kagi-key"]);
|
||||
@@ -188,4 +189,18 @@ describe("AuthStorage api-key login upsert", () => {
|
||||
expect(store.getApiKey("deepseek")).toBe("same-deepseek-key");
|
||||
expect(await authStorage.getApiKey("deepseek", "session-deepseek-relogin")).toBe("same-deepseek-key");
|
||||
});
|
||||
|
||||
it("uses a fresh OpenCode Go login over an existing env fallback", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
|
||||
getEnvApiKeySpy.mockImplementation(provider => (provider === "opencode-go" ? "old-opencode-key" : undefined));
|
||||
|
||||
await authStorage.login("opencode-go", {
|
||||
onAuth: () => {},
|
||||
onPrompt: async () => "new-opencode-key",
|
||||
});
|
||||
|
||||
expect(await authStorage.getApiKey("opencode-go", "session-opencode-go-login")).toBe("new-opencode-key");
|
||||
expect(await authStorage.peekApiKey("opencode-go")).toBe("new-opencode-key");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -98,7 +98,7 @@ function createCredential(accountId: string, email: string): OAuthCredentials {
|
||||
return {
|
||||
access: `access-${accountId}`,
|
||||
refresh: `refresh-${accountId}`,
|
||||
expires: Date.now() + HOUR_MS,
|
||||
expires: Date.now() + WEEK_MS,
|
||||
accountId,
|
||||
email,
|
||||
};
|
||||
@@ -595,6 +595,100 @@ describe("AuthStorage codex oauth ranking", () => {
|
||||
// flaky on loaded CI runners, so maxConcurrent is the authoritative signal.
|
||||
expect(maxConcurrent).toBe(3);
|
||||
});
|
||||
|
||||
test("skips expired access-token-only sticky credential and selects fresh sibling", async () => {
|
||||
if (!authStorage) throw new Error("test setup failed");
|
||||
const sessionId = "sticky-token-only-session";
|
||||
await authStorage.set("openai-codex", [{ type: "oauth", ...createCredential("acct-k12", "k12@example.com") }]);
|
||||
usageByAccount.set(
|
||||
"acct-k12",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-k12",
|
||||
primary: { usedFraction: 0.3, resetInMs: 20 * 60 * 1000 },
|
||||
secondary: { usedFraction: 0.2, resetInMs: 5 * 24 * 60 * 60 * 1000 },
|
||||
}),
|
||||
);
|
||||
expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-k12");
|
||||
usageByAccount.set(
|
||||
"acct-k12",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-k12",
|
||||
primary: { usedFraction: 1, resetInMs: FIVE_HOUR_MS },
|
||||
secondary: { usedFraction: 0.17, resetInMs: WEEK_MS },
|
||||
}),
|
||||
);
|
||||
usageByAccount.set(
|
||||
"acct-plus",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-plus",
|
||||
primary: { usedFraction: 0.2, resetInMs: FIVE_HOUR_MS },
|
||||
secondary: { usedFraction: 0.74, resetInMs: WEEK_MS },
|
||||
}),
|
||||
);
|
||||
|
||||
await authStorage.set("openai-codex", [
|
||||
{
|
||||
type: "oauth",
|
||||
access: "access-acct-k12",
|
||||
refresh: "",
|
||||
expires: Date.now() - 1_000,
|
||||
accountId: "acct-k12",
|
||||
email: "k12@example.com",
|
||||
},
|
||||
{
|
||||
type: "oauth",
|
||||
...createCredential("acct-plus", "plus@example.com"),
|
||||
},
|
||||
]);
|
||||
|
||||
expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-plus");
|
||||
});
|
||||
|
||||
test("ignores legacy global Codex blocks when a scoped quota window has fresh siblings", async () => {
|
||||
if (!authStorage || !store) throw new Error("test setup failed");
|
||||
await authStorage.set("openai-codex", [
|
||||
{ type: "oauth", ...createCredential("acct-k12", "k12@example.com") },
|
||||
{ type: "oauth", ...createCredential("acct-plus", "plus@example.com") },
|
||||
]);
|
||||
usageByAccount.set(
|
||||
"acct-k12",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-k12",
|
||||
primary: { usedFraction: 1, resetInMs: FIVE_HOUR_MS },
|
||||
secondary: { usedFraction: 1, resetInMs: WEEK_MS },
|
||||
}),
|
||||
);
|
||||
usageByAccount.set(
|
||||
"acct-plus",
|
||||
createCodexUsageReport({
|
||||
accountId: "acct-plus",
|
||||
primary: { usedFraction: 0.2, resetInMs: FIVE_HOUR_MS },
|
||||
secondary: { usedFraction: 0.74, resetInMs: WEEK_MS },
|
||||
}),
|
||||
);
|
||||
const plus = store
|
||||
.listAuthCredentials("openai-codex")
|
||||
.find(row => row.credential.type === "oauth" && row.credential.accountId === "acct-plus");
|
||||
if (!plus || !store.upsertCredentialBlock) throw new Error("missing plus credential row");
|
||||
store.upsertCredentialBlock({
|
||||
credentialId: plus.id,
|
||||
providerKey: "openai-codex:oauth",
|
||||
blockScope: "",
|
||||
blockedUntilMs: Date.now() + WEEK_MS,
|
||||
});
|
||||
const k12 = store
|
||||
.listAuthCredentials("openai-codex")
|
||||
.find(row => row.credential.type === "oauth" && row.credential.accountId === "acct-k12");
|
||||
if (!k12 || !store.upsertCredentialBlock) throw new Error("missing k12 credential row");
|
||||
store.upsertCredentialBlock({
|
||||
credentialId: k12.id,
|
||||
providerKey: "openai-codex:oauth",
|
||||
blockScope: "shared",
|
||||
blockedUntilMs: Date.now() + HOUR_MS,
|
||||
});
|
||||
|
||||
expect(await authStorage.getApiKey("openai-codex", "session-with-legacy-global-block")).toBe("api-acct-plus");
|
||||
});
|
||||
});
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -182,4 +182,61 @@ describe("OAuthCallbackFlow /launch route", () => {
|
||||
abort.abort("test done");
|
||||
await login;
|
||||
});
|
||||
|
||||
it("suppresses launchUrl for custom-scheme redirects that never return to the loopback server", async () => {
|
||||
const abort = new AbortController();
|
||||
const authFired = Promise.withResolvers<OAuthAuthInfo>();
|
||||
const flow = new LaunchProbeFlow(
|
||||
{
|
||||
onAuth: info => {
|
||||
authFired.resolve(info);
|
||||
},
|
||||
signal: abort.signal,
|
||||
},
|
||||
{
|
||||
preferredPort: 0,
|
||||
allowPortFallback: true,
|
||||
// GitLab Duo shape: `new URL` parses this happily (pathname
|
||||
// `/authentication`), so the guard must check scheme/host, not
|
||||
// rely on a parse failure. A localhost /launch copy target for
|
||||
// this flow would misrepresent the callback endpoint and point
|
||||
// remote users at a URL that resolves nowhere.
|
||||
redirectUri: "vscode://gitlab.gitlab-workflow/authentication",
|
||||
},
|
||||
);
|
||||
const login = flow.login().catch(() => undefined) as Promise<void>;
|
||||
const info = await authFired.promise;
|
||||
|
||||
expect(info.launchUrl).toBeUndefined();
|
||||
|
||||
abort.abort("test done");
|
||||
await login;
|
||||
});
|
||||
|
||||
it("suppresses launchUrl for fixed non-loopback HTTP redirects", async () => {
|
||||
const abort = new AbortController();
|
||||
const authFired = Promise.withResolvers<OAuthAuthInfo>();
|
||||
const flow = new LaunchProbeFlow(
|
||||
{
|
||||
onAuth: info => {
|
||||
authFired.resolve(info);
|
||||
},
|
||||
signal: abort.signal,
|
||||
},
|
||||
{
|
||||
preferredPort: 0,
|
||||
allowPortFallback: true,
|
||||
// The provider redirects to a hosted endpoint; this machine's
|
||||
// callback server never sees the redirect, so no launch URL.
|
||||
redirectUri: "https://auth.example.com/oauth/callback",
|
||||
},
|
||||
);
|
||||
const login = flow.login().catch(() => undefined) as Promise<void>;
|
||||
const info = await authFired.promise;
|
||||
|
||||
expect(info.launchUrl).toBeUndefined();
|
||||
|
||||
abort.abort("test done");
|
||||
await login;
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,104 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { buildAnthropicClientOptions, streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type { Context, Model, ModelSpec, TJsonSchema, Tool } from "@oh-my-pi/pi-ai/types";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
|
||||
const STRUCTURED_OUTPUTS_BETA = "structured-outputs-2025-12-15";
|
||||
|
||||
const bashTool: Tool = {
|
||||
name: "bash",
|
||||
description: "run a bash command",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { command: { type: "string" } },
|
||||
required: ["command"],
|
||||
} satisfies TJsonSchema,
|
||||
};
|
||||
|
||||
const toolContext: Context = {
|
||||
systemPrompt: ["Stay concise."],
|
||||
messages: [{ role: "user", content: "Hi", timestamp: 0 }],
|
||||
tools: [bashTool],
|
||||
};
|
||||
|
||||
function anthropicSpec(baseUrl: string): ModelSpec<"anthropic-messages"> {
|
||||
return {
|
||||
id: "claude-sonnet-5",
|
||||
name: "Claude Sonnet 5",
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
baseUrl,
|
||||
reasoning: false,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 200_000,
|
||||
maxTokens: 8_192,
|
||||
};
|
||||
}
|
||||
|
||||
function buildOAuthUtilityBetaHeader(model: Model<"anthropic-messages">): string {
|
||||
const options = buildAnthropicClientOptions({
|
||||
model,
|
||||
apiKey: "oauth-token",
|
||||
isOAuth: true,
|
||||
hasTools: false,
|
||||
thinkingEnabled: false,
|
||||
});
|
||||
return options.defaultHeaders["anthropic-beta"] ?? "";
|
||||
}
|
||||
|
||||
function abortedSignal(): AbortSignal {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
return controller.signal;
|
||||
}
|
||||
|
||||
async function captureToolParams(
|
||||
model: Model<"anthropic-messages">,
|
||||
): Promise<{ tools?: Array<{ name: string; strict?: unknown }> }> {
|
||||
const { promise, resolve } = Promise.withResolvers<{ tools?: Array<{ name: string; strict?: unknown }> }>();
|
||||
void streamAnthropic(model, toolContext, {
|
||||
apiKey: "sk-ant-api-test",
|
||||
isOAuth: false,
|
||||
signal: abortedSignal(),
|
||||
onPayload: payload => {
|
||||
resolve(payload as { tools?: Array<{ name: string; strict?: unknown }> });
|
||||
return undefined;
|
||||
},
|
||||
});
|
||||
return promise;
|
||||
}
|
||||
|
||||
describe("issue #4679 Azure Foundry Anthropic strict tools", () => {
|
||||
it.each([
|
||||
["inference", "https://example.inference.ai.azure.com/anthropic/v1"],
|
||||
["services", "https://example.services.ai.azure.com/anthropic/v1"],
|
||||
])("disables strict tools and omits structured-output beta for Azure Foundry %s routes", (_kind, baseUrl) => {
|
||||
const model = buildModel(anthropicSpec(baseUrl));
|
||||
|
||||
expect(model.compat.disableStrictTools).toBe(true);
|
||||
expect(buildOAuthUtilityBetaHeader(model)).not.toContain(STRUCTURED_OUTPUTS_BETA);
|
||||
});
|
||||
|
||||
it("keeps structured-output beta on direct Anthropic OAuth utility headers", () => {
|
||||
const model = buildModel(anthropicSpec("https://api.anthropic.com"));
|
||||
|
||||
expect(model.compat.disableStrictTools).toBe(false);
|
||||
expect(buildOAuthUtilityBetaHeader(model)).toContain(STRUCTURED_OUTPUTS_BETA);
|
||||
});
|
||||
|
||||
it("omits strict tool schemas on Azure Foundry Anthropic requests without disabling direct Anthropic", async () => {
|
||||
const azureParams = await captureToolParams(
|
||||
buildModel(anthropicSpec("https://example.services.ai.azure.com/anthropic/v1")),
|
||||
);
|
||||
const directParams = await captureToolParams(buildModel(anthropicSpec("https://api.anthropic.com")));
|
||||
|
||||
const azureBashTool = azureParams.tools?.find(tool => tool.name === "bash");
|
||||
const directBashTool = directParams.tools?.find(tool => tool.name === "bash");
|
||||
|
||||
expect(azureBashTool).toBeDefined();
|
||||
expect(azureBashTool?.strict).toBeUndefined();
|
||||
expect(directBashTool).toBeDefined();
|
||||
expect(directBashTool?.strict).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -2708,6 +2708,114 @@ describe("openai-codex streaming", () => {
|
||||
lastPreviousResponseId: undefined,
|
||||
});
|
||||
});
|
||||
it("retries websocket continuations when a proxy reports a stale previous response anchor", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
const sentRequests: Array<Record<string, unknown>> = [];
|
||||
const fetchMock = vi.fn(async () => {
|
||||
throw new Error("SSE fallback should not be called");
|
||||
});
|
||||
|
||||
class ProxyStaleAnchorWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(data: string): void {
|
||||
const request = JSON.parse(data) as Record<string, unknown>;
|
||||
sentRequests.push(request);
|
||||
const requestIndex = sentRequests.length;
|
||||
|
||||
if (requestIndex === 1) {
|
||||
this.emitCodexResponse({
|
||||
messageId: "msg_1",
|
||||
responseId: "resp_1",
|
||||
text: "First answer",
|
||||
terminalType: "response.completed",
|
||||
includeCreated: true,
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
if (requestIndex === 2) {
|
||||
expect(request.previous_response_id).toBe("resp_1");
|
||||
this.sendJson({
|
||||
type: "error",
|
||||
code: "codex_previous_response_stale",
|
||||
message: "Upstream previous response anchor expired; retry without previous_response_id.",
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
if (requestIndex === 3) {
|
||||
expect(request.previous_response_id).toBeUndefined();
|
||||
this.emitCodexResponse({
|
||||
messageId: "msg_3",
|
||||
responseId: "resp_3",
|
||||
text: "Second answer",
|
||||
terminalType: "response.completed",
|
||||
includeCreated: true,
|
||||
});
|
||||
return;
|
||||
}
|
||||
|
||||
throw new Error(`Unexpected websocket request index: ${requestIndex}`);
|
||||
}
|
||||
}
|
||||
|
||||
global.WebSocket = ProxyStaleAnchorWebSocket as unknown as typeof WebSocket;
|
||||
const model = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const firstContext: Context = {
|
||||
systemPrompt: ["You are a helpful assistant."],
|
||||
messages: [{ role: "user", content: "First question", timestamp: Date.now() }],
|
||||
};
|
||||
const firstResponse = await streamOpenAICodexResponses(model, firstContext, {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-proxy-stale-anchor-session",
|
||||
providerSessionState,
|
||||
}).result();
|
||||
const secondContext: Context = {
|
||||
systemPrompt: ["You are a helpful assistant."],
|
||||
messages: [
|
||||
...firstContext.messages,
|
||||
firstResponse,
|
||||
{ role: "user", content: "Second question", timestamp: Date.now() + 1 },
|
||||
],
|
||||
};
|
||||
|
||||
const secondResponse = await streamOpenAICodexResponses(model, secondContext, {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-proxy-stale-anchor-session",
|
||||
providerSessionState,
|
||||
}).result();
|
||||
|
||||
expect(secondResponse.stopReason).toBe("stop");
|
||||
expect(JSON.stringify(secondResponse.content)).toContain("Second answer");
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
expect(sentRequests).toHaveLength(3);
|
||||
expect(sentRequests[2]?.prompt_cache_key).toBe("ws-proxy-stale-anchor-session");
|
||||
const retryInput = sentRequests[2]?.input;
|
||||
expect(Array.isArray(retryInput)).toBe(true);
|
||||
expect(JSON.stringify(retryInput)).toContain("First question");
|
||||
expect(JSON.stringify(retryInput)).toContain("Second question");
|
||||
|
||||
const stats = getOpenAICodexWebSocketDebugStats(model, {
|
||||
sessionId: "ws-proxy-stale-anchor-session",
|
||||
providerSessionState,
|
||||
});
|
||||
expect(stats).toMatchObject({
|
||||
fullContextRequests: 2,
|
||||
deltaRequests: 1,
|
||||
lastInputItems: (retryInput as unknown[]).length,
|
||||
lastDeltaInputItems: undefined,
|
||||
lastPreviousResponseId: undefined,
|
||||
});
|
||||
});
|
||||
|
||||
it("uses websocket v2 beta header when v2 mode is enabled", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
|
||||
@@ -63,7 +63,7 @@ describe("openai-codex usage parser", () => {
|
||||
expect(report).not.toBeNull();
|
||||
const main = report?.limits.filter(l => l.id === "openai-codex:primary" || l.id === "openai-codex:secondary");
|
||||
expect(main?.map(l => l.id)).toEqual(["openai-codex:primary", "openai-codex:secondary"]);
|
||||
expect(main?.[0].scope.tier).toBe("pro");
|
||||
expect(main?.[0].scope).toEqual({ provider: "openai-codex", windowId: "5h", shared: true });
|
||||
expect(main?.[0].amount.usedFraction).toBeCloseTo(0.04, 5);
|
||||
});
|
||||
|
||||
|
||||
@@ -705,6 +705,53 @@ describe("OpenAI-family first-event timeouts", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("accepts response.done with a completed response as an OpenAI responses terminal event", async () => {
|
||||
const completedResponse = createSseResponse([
|
||||
{ type: "response.created", response: { id: "resp_done" } },
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
item: { type: "message", id: "msg_done", role: "assistant", status: "in_progress", content: [] },
|
||||
},
|
||||
{ type: "response.content_part.added", part: { type: "output_text", text: "" } },
|
||||
{ type: "response.output_text.delta", delta: "Hello done" },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_done",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "Hello done" }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.done",
|
||||
response: {
|
||||
id: "resp_done",
|
||||
status: "completed",
|
||||
usage: {
|
||||
input_tokens: 3,
|
||||
output_tokens: 2,
|
||||
total_tokens: 5,
|
||||
},
|
||||
},
|
||||
},
|
||||
]);
|
||||
const fetch: FetchImpl = () => Promise.resolve(completedResponse);
|
||||
const result = await streamOpenAIResponses(openAIResponsesModel, baseContext(), {
|
||||
apiKey: "test-key",
|
||||
fetch,
|
||||
}).result();
|
||||
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.content as unknown[]).toContainEqual({
|
||||
type: "text",
|
||||
text: "Hello done",
|
||||
textSignature: '{"v":1,"id":"msg_done"}',
|
||||
});
|
||||
});
|
||||
|
||||
it("errors when Azure OpenAI responses stream closes without a terminal response event", async () => {
|
||||
const incompleteResponse = createSseResponse([
|
||||
{ type: "response.created", response: { id: "resp_incomplete_azure" } },
|
||||
|
||||
@@ -405,7 +405,7 @@ describe("streamSimple resolver auth retry", () => {
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["credential-A", "credential-B"]);
|
||||
expect(eventTypes).toEqual(["start", "text_start", "text_delta", "text_end", "done"]);
|
||||
expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([false, true]);
|
||||
expect(retryContexts.map(ctx => ctx.lastChance)).toEqual([true]);
|
||||
}
|
||||
});
|
||||
|
||||
@@ -501,10 +501,9 @@ describe("streamSimple resolver auth retry", () => {
|
||||
expect((await stream.result()).content).toEqual([{ type: "text", text: "ok" }]);
|
||||
expect(keys).toEqual(["old-key", "next-key"]);
|
||||
expect(retryContexts.map(ctx => ({ lastChance: ctx.lastChance, hasError: ctx.error !== undefined }))).toEqual([
|
||||
{ lastChance: false, hasError: true },
|
||||
{ lastChance: true, hasError: true },
|
||||
]);
|
||||
expect((retryContexts[1]?.error as Error).message).toContain("Resource exhausted");
|
||||
expect((retryContexts[0]?.error as Error).message).toContain("Resource exhausted");
|
||||
});
|
||||
|
||||
it("surfaces the original error when the resolver declines every retry", async () => {
|
||||
|
||||
@@ -2,6 +2,12 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed LiteLLM discovery stopping at `/model_group/info` when that endpoint omitted `supports_vision`; it now continues to `/model/info` and preserves `model_info.supports_vision=true` for vision-capable proxy models. ([#4747](https://github.com/can1357/oh-my-pi/issues/4747))
|
||||
- Fixed LiteLLM discovery to fall back to bundled catalog metadata when `models.dev` lacks a model reference, preserving reasoning and thinking support for models such as `glm-5.2`. ([#4695](https://github.com/can1357/oh-my-pi/issues/4695))
|
||||
- Detected Azure AI Inference / Foundry Anthropic routes as strict-tool-incompatible so resolved Anthropic compat disables strict tools before request construction ([#4679](https://github.com/can1357/oh-my-pi/issues/4679)).
|
||||
|
||||
## [16.3.11] - 2026-07-06
|
||||
|
||||
### Added
|
||||
|
||||
@@ -99,18 +99,15 @@ export function buildAnthropicCompat(spec: ModelSpec<"anthropic-messages">): Res
|
||||
// (issue #4192).
|
||||
const isZenmux = modelMatchesHost(spec, "zenmux");
|
||||
const requiresThinkingEnabled = modelMatchesHost(spec, "moonshotNative") && matchesKimiK27CodeFamily(spec);
|
||||
const isVertex = isVertexAnthropicRoute(baseUrl);
|
||||
const isBedrock = isBedrockAnthropicRoute(baseUrl);
|
||||
const isAzure = isAzureAnthropicRoute(baseUrl);
|
||||
const signingEndpoint =
|
||||
official ||
|
||||
isCopilot ||
|
||||
isZenmux ||
|
||||
isCloudflareAnthropicGateway(baseUrl) ||
|
||||
isVertexAnthropicRoute(baseUrl) ||
|
||||
isBedrockAnthropicRoute(baseUrl) ||
|
||||
isAzureAnthropicRoute(baseUrl);
|
||||
official || isCopilot || isZenmux || isCloudflareAnthropicGateway(baseUrl) || isVertex || isBedrock || isAzure;
|
||||
const compat: ResolvedAnthropicCompat = {
|
||||
officialEndpoint: official,
|
||||
signingEndpoint,
|
||||
disableStrictTools: false,
|
||||
disableStrictTools: isAzure,
|
||||
disableAdaptiveThinking: false,
|
||||
supportsEagerToolInputStreaming: !isCopilot,
|
||||
// Long cache retention is only sent to the official API by default;
|
||||
|
||||
@@ -3032,6 +3032,15 @@ export interface FetchLiteLLMRichModelsOptions<TApi extends Api> {
|
||||
}
|
||||
|
||||
type LiteLLMRichModelEntry = Record<string, unknown>;
|
||||
type LiteLLMRichEndpointModel<TApi extends Api> = {
|
||||
model: ModelSpec<TApi>;
|
||||
supportsVision: unknown;
|
||||
supportsReasoning: unknown;
|
||||
hasContextWindow: boolean;
|
||||
hasMaxTokens: boolean;
|
||||
hasToolMetadata: boolean;
|
||||
hasSupportedOpenAIParams: boolean;
|
||||
};
|
||||
|
||||
const LITELLM_RICH_ENDPOINTS = ["/model_group/info", "/v2/model/info", "/model/info", "/v1/model/info"] as const;
|
||||
export const OPENAI_COMPAT_DISCOVERY_DEFAULT_CONTEXT_WINDOW = 128_000;
|
||||
@@ -3264,7 +3273,7 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
|
||||
managementBaseUrl: string,
|
||||
runtimeBaseUrl: string,
|
||||
signal?: AbortSignal,
|
||||
): Promise<ModelSpec<TApi>[] | null> {
|
||||
): Promise<{ models: LiteLLMRichEndpointModel<TApi>[]; incompleteVisionMetadata: boolean } | null> {
|
||||
const fetchImpl = discoveryFetch(options.fetch);
|
||||
const requestHeaders: Record<string, string> = {
|
||||
Accept: "application/json",
|
||||
@@ -3296,17 +3305,39 @@ async function fetchLiteLLMRichEndpoint<TApi extends Api>(
|
||||
if (!entries || entries.length === 0) {
|
||||
return null;
|
||||
}
|
||||
const deduped = new Map<string, ModelSpec<TApi>>();
|
||||
const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
|
||||
let incompleteVisionMetadata = false;
|
||||
for (const entry of entries) {
|
||||
const model = mapLiteLLMRichEntry(entry, options, runtimeBaseUrl);
|
||||
if (model) {
|
||||
deduped.set(model.id, model);
|
||||
const supportsVision = getLiteLLMMetadataValue(entry, "supports_vision");
|
||||
const supportsReasoning = getLiteLLMMetadataValue(entry, "supports_reasoning");
|
||||
const supportsFunctionCalling = getLiteLLMMetadataValue(entry, "supports_function_calling");
|
||||
const supportedOpenAIParams = getSupportedOpenAIParams(entry);
|
||||
if (supportsVision !== true && supportsVision !== false) {
|
||||
incompleteVisionMetadata = true;
|
||||
}
|
||||
deduped.set(model.id, {
|
||||
model,
|
||||
supportsVision,
|
||||
supportsReasoning,
|
||||
hasContextWindow: toPositiveNumber(getLiteLLMMetadataValue(entry, "max_input_tokens"), null) !== null,
|
||||
hasMaxTokens: toPositiveNumber(getLiteLLMMetadataValue(entry, "max_output_tokens"), null) !== null,
|
||||
hasToolMetadata:
|
||||
supportsFunctionCalling === true ||
|
||||
supportsFunctionCalling === false ||
|
||||
supportedOpenAIParams !== undefined,
|
||||
hasSupportedOpenAIParams: supportedOpenAIParams !== undefined,
|
||||
});
|
||||
}
|
||||
}
|
||||
if (deduped.size === 0) {
|
||||
return null;
|
||||
}
|
||||
return Array.from(deduped.values()).sort((left, right) => left.id.localeCompare(right.id));
|
||||
return {
|
||||
models: Array.from(deduped.values()).sort((left, right) => left.model.id.localeCompare(right.model.id)),
|
||||
incompleteVisionMetadata,
|
||||
};
|
||||
}
|
||||
|
||||
export async function fetchLiteLLMRichModels<TApi extends Api>(
|
||||
@@ -3318,13 +3349,55 @@ export async function fetchLiteLLMRichModels<TApi extends Api>(
|
||||
return null;
|
||||
}
|
||||
const fetchModels = async (signal?: AbortSignal): Promise<ModelSpec<TApi>[] | null> => {
|
||||
const deduped = new Map<string, LiteLLMRichEndpointModel<TApi>>();
|
||||
for (const endpoint of LITELLM_RICH_ENDPOINTS) {
|
||||
const models = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal);
|
||||
if (models) {
|
||||
return models;
|
||||
const result = await fetchLiteLLMRichEndpoint(endpoint, options, managementBaseUrl, runtimeBaseUrl, signal);
|
||||
if (!result) {
|
||||
continue;
|
||||
}
|
||||
const hadPriorModels = deduped.size > 0;
|
||||
for (const next of result.models) {
|
||||
const existing = deduped.get(next.model.id);
|
||||
if (!existing) {
|
||||
if (!hadPriorModels) {
|
||||
deduped.set(next.model.id, next);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
const model: ModelSpec<TApi> = {
|
||||
...existing.model,
|
||||
name: next.model.name === next.model.id ? existing.model.name : next.model.name,
|
||||
contextWindow: next.hasContextWindow ? next.model.contextWindow : existing.model.contextWindow,
|
||||
maxTokens: next.hasMaxTokens ? next.model.maxTokens : existing.model.maxTokens,
|
||||
input:
|
||||
next.supportsVision === true || next.supportsVision === false
|
||||
? next.model.input
|
||||
: existing.model.input,
|
||||
reasoning: typeof next.supportsReasoning === "boolean" ? next.model.reasoning : existing.model.reasoning,
|
||||
compat: next.hasSupportedOpenAIParams ? next.model.compat : existing.model.compat,
|
||||
};
|
||||
if (next.hasToolMetadata) {
|
||||
model.supportsTools = next.model.supportsTools;
|
||||
}
|
||||
deduped.set(next.model.id, { ...next, model });
|
||||
}
|
||||
let hasIncompleteVisionMetadata = false;
|
||||
for (const entry of deduped.values()) {
|
||||
if (entry.supportsVision !== true && entry.supportsVision !== false) {
|
||||
hasIncompleteVisionMetadata = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!hasIncompleteVisionMetadata) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
if (deduped.size === 0) {
|
||||
return null;
|
||||
}
|
||||
return Array.from(deduped.values())
|
||||
.map(entry => entry.model)
|
||||
.sort((left, right) => left.id.localeCompare(right.id));
|
||||
};
|
||||
if (options.signal !== undefined) {
|
||||
return fetchModels(options.signal);
|
||||
@@ -3339,18 +3412,20 @@ export function litellmModelManagerOptions(
|
||||
const baseUrl = config?.baseUrl ?? Bun.env.LITELLM_BASE_URL ?? "http://localhost:4000/v1";
|
||||
return {
|
||||
providerId: "litellm",
|
||||
// rich-v3 invalidates rows cached before reseller usage-suffix stripping
|
||||
// and placeholder-only `all-team-models` filtering; bump the version whenever
|
||||
// the mappers below change, or warm authoritative caches keep serving
|
||||
// pre-change rows for the full TTL.
|
||||
cacheProviderId: `litellm:rich-v3:${Bun.hash(baseUrl).toString(36)}`,
|
||||
// rich-v4 invalidates rows cached before LiteLLM ids gained bundled
|
||||
// reference fallback and before discovery continued past `/model_group/info`
|
||||
// when that endpoint omitted vision metadata. Earlier versions handled
|
||||
// reseller usage-suffix stripping and placeholder-only `all-team-models`
|
||||
// filtering; bump the version whenever the mappers below change, or warm
|
||||
// authoritative caches keep serving pre-change rows for the full TTL.
|
||||
cacheProviderId: `litellm:rich-v4:${Bun.hash(baseUrl).toString(36)}`,
|
||||
// litellm is a local-only proxy and is never bundled in models.json (that
|
||||
// would leak the machine's localhost catalog). Prefer the proxy's richer
|
||||
// management metadata, then fall back to /v1/models and enrich bare ids
|
||||
// against models.dev like the gateway providers (fireworks et al.) do.
|
||||
// management metadata, then enrich ids against models.dev with the bundled
|
||||
// catalog as a fallback before using /v1/models.
|
||||
fetchDynamicModels: async () => {
|
||||
const modelsDevReferences = await loadModelsDevReferences<"openai-completions">(config?.fetch);
|
||||
const resolveReference = (id: string) => modelsDevReferences.get(id);
|
||||
const resolveReference = createReferenceResolver(modelsDevReferences);
|
||||
const richModels = await fetchLiteLLMRichModels({
|
||||
api: "openai-completions",
|
||||
provider: "litellm",
|
||||
|
||||
@@ -125,7 +125,7 @@ describe("LiteLLM provider discovery", () => {
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(options.cacheProviderId).toBe(
|
||||
`litellm:rich-v3:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`,
|
||||
`litellm:rich-v4:${Bun.hash("http://litellm.example:4100/v1").toString(36)}`,
|
||||
);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(6);
|
||||
expect(models).toHaveLength(1);
|
||||
@@ -148,7 +148,7 @@ describe("LiteLLM provider discovery", () => {
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(options.cacheProviderId).toBe(
|
||||
`litellm:rich-v3:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`,
|
||||
`litellm:rich-v4:${Bun.hash("http://litellm-config.example:4200/v1/").toString(36)}`,
|
||||
);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(6);
|
||||
expect(models).toHaveLength(1);
|
||||
@@ -238,6 +238,54 @@ describe("LiteLLM provider discovery", () => {
|
||||
});
|
||||
});
|
||||
|
||||
test("enriches LiteLLM rich models missing from models.dev with bundled reasoning metadata", async () => {
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request) => {
|
||||
const url = inputUrl(input);
|
||||
if (url === MODELS_DEV_URL) {
|
||||
return Response.json({});
|
||||
}
|
||||
if (url === "http://primary:4000/model_group/info") {
|
||||
return Response.json({
|
||||
data: [
|
||||
{
|
||||
model_group: "glm-5.2",
|
||||
model_name: "GLM-5.2",
|
||||
},
|
||||
],
|
||||
});
|
||||
}
|
||||
if (url === "http://primary:4000/v1/models") {
|
||||
throw new Error("/v1/models should not be called when model_group info has a real model");
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
}) as FetchImpl;
|
||||
const options = litellmModelManagerOptions({
|
||||
apiKey: "sk-rich",
|
||||
baseUrl: "http://primary:4000/v1",
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(models).toHaveLength(1);
|
||||
expect(models?.[0]).toMatchObject({
|
||||
id: "glm-5.2",
|
||||
name: "GLM-5.2",
|
||||
api: "openai-completions",
|
||||
provider: "litellm",
|
||||
baseUrl: "http://primary:4000/v1",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: ["minimal", "low", "medium", "high", "xhigh"],
|
||||
effortMap: {
|
||||
minimal: "none",
|
||||
xhigh: "max",
|
||||
},
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
test("uses LiteLLM tool support metadata when rich endpoints succeed", async () => {
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request) => {
|
||||
const url = inputUrl(input);
|
||||
@@ -247,8 +295,13 @@ describe("LiteLLM provider discovery", () => {
|
||||
if (url === "http://primary:4000/model_group/info") {
|
||||
return Response.json({
|
||||
data: [
|
||||
{ model_group: "no-tools", providers: ["openai"], supports_function_calling: false },
|
||||
{ model_group: "params-tools", supported_openai_params: ["tools"] },
|
||||
{
|
||||
model_group: "no-tools",
|
||||
providers: ["openai"],
|
||||
supports_vision: false,
|
||||
supports_function_calling: false,
|
||||
},
|
||||
{ model_group: "params-tools", supports_vision: false, supported_openai_params: ["tools"] },
|
||||
],
|
||||
});
|
||||
}
|
||||
@@ -339,6 +392,7 @@ describe("LiteLLM provider discovery", () => {
|
||||
max_input_tokens: 96_000,
|
||||
max_output_tokens: 8_000,
|
||||
supports_function_calling: true,
|
||||
supports_vision: false,
|
||||
},
|
||||
],
|
||||
});
|
||||
@@ -422,6 +476,71 @@ describe("LiteLLM provider discovery", () => {
|
||||
});
|
||||
});
|
||||
|
||||
test("continues to LiteLLM model info when model_group omits vision metadata", async () => {
|
||||
const calls: string[] = [];
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request) => {
|
||||
const url = inputUrl(input);
|
||||
calls.push(url);
|
||||
if (url === MODELS_DEV_URL) {
|
||||
return Response.json({});
|
||||
}
|
||||
if (url === "http://primary:4000/model_group/info") {
|
||||
return Response.json({
|
||||
data: [
|
||||
{
|
||||
model_group: "vision-proxy-model",
|
||||
model_name: "Vision Proxy Model",
|
||||
max_input_tokens: 64_000,
|
||||
max_output_tokens: 8_000,
|
||||
},
|
||||
],
|
||||
});
|
||||
}
|
||||
if (url === "http://primary:4000/v2/model/info") {
|
||||
return Response.json({
|
||||
data: [{ model_name: "unrelated-v2-model", model_info: { supports_vision: false } }],
|
||||
});
|
||||
}
|
||||
if (url === "http://primary:4000/model/info") {
|
||||
return Response.json({
|
||||
data: [
|
||||
{ model_name: "text-only-model", model_info: { supports_vision: false } },
|
||||
{
|
||||
model_name: "vision-proxy-model",
|
||||
model_info: {
|
||||
supports_vision: true,
|
||||
},
|
||||
},
|
||||
],
|
||||
});
|
||||
}
|
||||
if (url === "http://primary:4000/v1/models") {
|
||||
throw new Error("/v1/models should not be called when LiteLLM model info succeeds");
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
}) as FetchImpl;
|
||||
const options = litellmModelManagerOptions({
|
||||
apiKey: "sk-rich",
|
||||
baseUrl: "http://primary:4000/v1",
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(calls).toContain("http://primary:4000/model_group/info");
|
||||
expect(calls).toContain("http://primary:4000/v2/model/info");
|
||||
expect(calls).toContain("http://primary:4000/model/info");
|
||||
expect(calls).not.toContain("http://primary:4000/v1/models");
|
||||
expect(models).toHaveLength(1);
|
||||
expect(models?.find(model => model.id === "vision-proxy-model")).toMatchObject({
|
||||
id: "vision-proxy-model",
|
||||
name: "Vision Proxy Model",
|
||||
input: ["text", "image"],
|
||||
contextWindow: 64_000,
|
||||
maxTokens: 8_000,
|
||||
});
|
||||
});
|
||||
|
||||
test("falls back from v2 model info to LiteLLM model info", async () => {
|
||||
const calls: string[] = [];
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request) => {
|
||||
@@ -431,7 +550,9 @@ describe("LiteLLM provider discovery", () => {
|
||||
return new Response("{}", { status: 404 });
|
||||
}
|
||||
if (url === "http://primary:4000/model/info") {
|
||||
return Response.json({ data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000 } }] });
|
||||
return Response.json({
|
||||
data: [{ model_name: "legacy-gpt", model_info: { max_input_tokens: 96_000, supports_vision: false } }],
|
||||
});
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
}) as FetchImpl;
|
||||
@@ -547,4 +668,53 @@ describe("LiteLLM provider discovery", () => {
|
||||
maxTokens: 8_192,
|
||||
});
|
||||
});
|
||||
|
||||
test("enriches LiteLLM /v1/models fallback entries missing from models.dev with bundled reasoning metadata", async () => {
|
||||
const calls: string[] = [];
|
||||
const fetchMock = vi.fn(async (input: string | URL | Request) => {
|
||||
const url = inputUrl(input);
|
||||
calls.push(url);
|
||||
if (url === MODELS_DEV_URL) {
|
||||
return Response.json({});
|
||||
}
|
||||
if (
|
||||
url === "http://primary:4000/model_group/info" ||
|
||||
url === "http://primary:4000/v2/model/info" ||
|
||||
url === "http://primary:4000/model/info" ||
|
||||
url === "http://primary:4000/v1/model/info"
|
||||
) {
|
||||
return new Response("{}", { status: 404 });
|
||||
}
|
||||
if (url === "http://primary:4000/v1/models") {
|
||||
return Response.json({ data: [{ id: "glm-5.2" }] });
|
||||
}
|
||||
throw new Error(`Unexpected URL: ${url}`);
|
||||
}) as FetchImpl;
|
||||
const options = litellmModelManagerOptions({
|
||||
apiKey: "sk-fallback",
|
||||
baseUrl: "http://primary:4000/v1",
|
||||
fetch: fetchMock,
|
||||
});
|
||||
|
||||
const models = await options.fetchDynamicModels?.();
|
||||
|
||||
expect(calls).toContain("http://primary:4000/v1/models");
|
||||
expect(models).toHaveLength(1);
|
||||
expect(models?.[0]).toMatchObject({
|
||||
id: "glm-5.2",
|
||||
name: "GLM-5.2",
|
||||
api: "openai-completions",
|
||||
provider: "litellm",
|
||||
baseUrl: "http://primary:4000/v1",
|
||||
reasoning: true,
|
||||
thinking: {
|
||||
mode: "effort",
|
||||
efforts: ["minimal", "low", "medium", "high", "xhigh"],
|
||||
effortMap: {
|
||||
minimal: "none",
|
||||
xhigh: "max",
|
||||
},
|
||||
},
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -6,10 +6,53 @@
|
||||
|
||||
- Added `error.notify` so failed model turns can emit distinct terminal/desktop notifications without changing completion notifications ([#2691](https://github.com/can1357/oh-my-pi/issues/2691)).
|
||||
|
||||
### Added
|
||||
|
||||
- Typing `#<number>` (e.g. `#3164`) in the prompt now offers PR and Issue autocomplete candidates that rewrite to the `pr://`/`issue://` internal URL, resolved from the current repo's git remote via the existing `read` tool → InternalUrlRouter → `gh` pipeline. Naming the type (`pr #3164` / `issue #3164`) constrains the candidates to that kind, and embedded hashes like `owner/repo#N`, `foo#N`, or URL fragments are left untouched ([#3218](https://github.com/can1357/oh-my-pi/issues/3218))
|
||||
|
||||
### Changed
|
||||
|
||||
- Memoized non-message token totals (system prompt, tool schemas, skills) so the per-turn compaction and context-threshold paths recompute them at most once per input change instead of on every call. `getContextBreakdown` and `#estimateStoredContextTokens` previously re-tokenized the system prompt and every tool's wire schema (per-tool `JSON.stringify`) several times per turn over inputs that change at most once per turn.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Improved handling of unawaited promises in JS eval cells to prevent process crashes
|
||||
- Added warning logs for unhandled rejections originating from finished eval cells
|
||||
- Improved advisor robustness by blocking exhausted accounts during consecutive turn failures
|
||||
- Fixed advisor turns hammering the same usage-limited account: a failed advisor turn now marks the exhausted credential blocked (with the provider's retry hint and usage-report reset time), so the next retry rotates to a sibling instead of re-picking the blocked account every few seconds. Previously the in-stream auth retry rotated within a request but never blocked the last failing credential, and the advisor loop — unlike the primary retry pipeline — never called `markUsageLimitReached`.
|
||||
- Added the account key to the `codex-auto-reset: skipped` debug log so skip reasons (e.g. `weekly-not-exhausted`) can be attributed to the evaluated account.
|
||||
- Fixed unawaited promise rejections in JS eval cells crashing the session: a floating rejection now fails the owning cell run (`Unhandled rejection (missing await?): …`) instead of escaping to the global `unhandledRejection` handler, which printed `[Unhandled Rejection]` and killed the process (inline fallback) or tore down the eval worker (dedicated worker). Rejections surfacing after a cell settled are downgraded to a warn log attributed to the finished cell.
|
||||
- Fixed project `.omp/RULES.md` sticky rules being shadowed by user `~/.omp/agent/RULES.md` rules with the same synthesized `RULES` name, so both user and project sticky rules now inject ([#4739](https://github.com/can1357/oh-my-pi/issues/4739)).
|
||||
- Fixed bash internal-URL expansion so unresolved literal `memory://` / `skill://` text stays verbatim instead of aborting command execution ([#4737](https://github.com/can1357/oh-my-pi/issues/4737)).
|
||||
- Fixed `agent://<id>` (and the `output()` eval helper) failing with `Not found` for a subagent spawned by another subagent (any spawn chain 2+ levels deep). `artifactsDirsFromRegistry` scanned only each ref's adopted (root-wide) `ArtifactManager` dir, but a subagent's own children are written one level deeper under its `sessionFile`-derived dir — so a live, addressable nested peer's output was unresolvable. The resolver now collects both candidate dirs per registered agent. ([#4650](https://github.com/can1357/oh-my-pi/issues/4650))
|
||||
- Fixed plan mode to document `local://` artifacts as writable session-local planning files and to carry every pre-approval `local://` artifact into the fresh session created by Approve and Execute.
|
||||
- Fixed the browser tool failing to launch Microsoft Edge-only Windows installs with Puppeteer's empty `Code: 0` launch error by keeping Edge's required `--enable-automation` default while preserving Chrome/Chromium stealth launch defaults.
|
||||
- Fixed bash/tool command environments inheriting Bun-autoloaded launch `.env.local` values, so nested apps can load their own dotenv values without parent deployment variables taking precedence. ([#4723](https://github.com/can1357/oh-my-pi/issues/4723))
|
||||
- Fixed legacy plugin validation for extension graphs that import JSON with `with { type: "json" }`, leaving JSON files on Bun's native loader instead of parsing them as JavaScript ([#4687](https://github.com/can1357/oh-my-pi/issues/4687)).
|
||||
- Fixed pasted terminal transcripts beginning with a shell prompt (`$ ...`) being mistaken for local Python shortcuts instead of being submitted as normal prompts ([#4678](https://github.com/can1357/oh-my-pi/issues/4678)).
|
||||
- Fixed wrapped OAuth copy-URL rows corrupting on paste: continuation chunks no longer carry a leading indent, so a multi-row terminal selection reassembles to the exact authorize URL (browsers strip newlines on paste but preserve or percent-encode embedded spaces, which previously corrupted the URL at every chunk boundary).
|
||||
- Fixed Windows browser-launch failures being unobservable: the opener now uses `%SystemRoot%`-resolved PowerShell `Start-Process` (via `-EncodedCommand`) instead of `rundll32`, which exits 0 unconditionally. Failures ShellExecute itself reports — missing target, no handler executable, access denied — now surface as non-zero exits and are logged; the encoded payload also keeps OAuth query strings (`&`-bearing) opaque to shell metacharacter parsing.
|
||||
- Fixed system prompt date rendering to use the host local calendar date instead of UTC.
|
||||
- Fixed `bash` tool `timeout: 0` so it disables the command deadline instead of falling back to the minimum timeout.
|
||||
- Fixed `read` and `grep` refusing to access filesystem paths whose names end in a selector-shaped suffix (e.g. `test:1-2`, `log:raw`) by preferring a literal match over the trailing `:<selector>` peel when the raw path exists on disk ([#4618](https://github.com/can1357/oh-my-pi/issues/4618)).
|
||||
- Fixed wrapped Edit-diff rows leaking inverse video into the result card's right-edge padding: a row that broke inside an intra-line highlight left inverse active at the row end, so the frame padding after it rendered as a default-foreground block ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103))
|
||||
- Fixed Edit-diff continuation rows escaping into the line-number column when the row's gutter was left-padded (line number narrower than the widest in the diff) or blanked by the gutter dedup (the bare `+` row of a single-line replacement); such rows now wrap behind a continuation gutter, while body lines that merely start with `|` keep wrapping generically ([#4616](https://github.com/can1357/oh-my-pi/pull/4616) by [@chan1103](https://github.com/chan1103))
|
||||
- Fixed live advisors continuing to use a stale `modelRoles.advisor` selection after `/model` changed the advisor model. ([#4612](https://github.com/can1357/oh-my-pi/issues/4612))
|
||||
- Fixed Claude plugin slash commands and skills silently vanishing when the plugin manifest declares `commands`/`slash-commands`/`skills` as a JSON array — the shape the Claude plugins reference documents and real plugins like `addyosmani/agent-skills` ship. `resolvePluginDir` in `packages/coding-agent/src/discovery/claude-plugins.ts` typed those fields as `string` and dropped array values on the floor; it now normalizes both shapes, loads every in-root entry, and reports one out-of-plugin-root warning per bad entry. The resolver also now honours Claude's per-field merge semantic — `skills` adds to the default `skills/` scan; `commands`/`slash-commands` replace the default `commands/` — so plugins like `{"skills":["./extra-skills"]}` no longer lose their default `skills/` folder while `{"commands":["./admin"]}` still replaces `commands/` as documented. ([#4609](https://github.com/can1357/oh-my-pi/issues/4609))
|
||||
- Fixed macOS Backspace on empty search not deleting sessions in the `/resume` picker; Fn+Backspace terminals that deliver `\x7f` instead of `\e[3~` now reach the delete confirmation dialog. ([#4580](https://github.com/can1357/oh-my-pi/pull/4580) by [@JagravNaik](https://github.com/JagravNaik))
|
||||
- Fixed `/rename` title arguments treating `#` prompt-action tokens as autocomplete triggers instead of literal session title text. ([#4600](https://github.com/can1357/oh-my-pi/issues/4600))
|
||||
- Fixed empty session `.jsonl` files accumulating in `~/.omp/agent/sessions/<cwd>/` after a draft-then-clear exit cycle. `SessionManager.saveDraft(text)` materializes the session file so the draft sidecar has a parent; a subsequent `saveDraft("")` unlinked the sidecar but left the metadata-only JSONL behind (title slot + session header + startup selector entries, ~500–750 B), and `#shouldHaveSessionFile()` could no longer prune it once `#fileIsCurrent`/`#forceFileCreation` were latched. `SessionManager.close()` now drops only draft-owned metadata-only sessions with no saved draft sidecar to reattach to, while keeping real conversations, meaningful non-message entries such as handoff custom messages, explicit `ensureOnDisk()` sessions, drafts still pending for `--resume`, and never-materialized sessions untouched ([#4571](https://github.com/can1357/oh-my-pi/issues/4571)).
|
||||
- Fixed the advisor being disabled for the entire session when the advisor role resolves to a reasoning model that exposes no controllable effort surface (Devin `devin/glm-5-2*`: `reasoning: true`, `thinking: undefined` — Cascade routes by sibling model id rather than a wire param). `#resolveAdvisorRuntimeDescriptors` in `packages/coding-agent/src/session/agent-session.ts` used to hardcode `ThinkingLevel.Medium`, which tripped `requireSupportedEffort` on the first advisor prompt with `Thinking effort medium is not supported by devin/glm-5-2. Supported efforts:` (empty list). The advisor descriptor now clamps the requested effort against the resolved model via `resolveThinkingLevelForModel` and forwards no explicit effort when the model has no controllable efforts — matching the `auto`-path fix (`clampAutoThinkingEffort`) and the Autonomous Memory stage fix (`clampThinkingLevelForModel`). Explicit `:off` still disables reasoning, and models that support `medium` (e.g. Anthropic) keep receiving it ([#4579](https://github.com/can1357/oh-my-pi/issues/4579)).
|
||||
- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports `calculateCost` (or `modelsAreEqual` / `getBundledProviders`) from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)).
|
||||
- Fixed legacy extension plugin validation failing with `Export named 'calculateCost' not found in module '.../legacy-pi-ai-shim.ts'` when the extension imports relocated catalog symbols such as `calculateCost`, `modelsAreEqual`, `getBundledProviders`, `getBundledModel`, or `getBundledModels` from `@oh-my-pi/pi-ai`. Those symbols were relocated to `@oh-my-pi/pi-catalog/models` during the catalog split but were never bridged back through the legacy `pi-ai` root shim; the shim now re-exports them alongside the existing `getModel` / `getModels` aliases so plugins written against pre-split pi-ai load again ([#4584](https://github.com/can1357/oh-my-pi/issues/4584)).
|
||||
- Fixed legacy pi extension imports of `DefaultResourceLoader` from `@mariozechner/pi-coding-agent` / `@earendil-works/pi-coding-agent` by adding a compatibility loader shim that translates `resourceLoader` into OMP's native session discovery options. ([#4567](https://github.com/can1357/oh-my-pi/issues/4567))
|
||||
- Fixed legacy Pi extension reloads on POSIX so `loadLegacyPiModule` imports the entry through a cache-busting filesystem path, refreshes load-time graph hooks when reloads add new modules, and threads the current load's `?mtime` tag through the extension source graph — relative `./helper.ts` siblings, `#alias/*` package-imports, extension-local bare dependency entries, and their relative children all rekey per reload, so same-process re-imports pick up edits across the whole graph. ([#4565](https://github.com/can1357/oh-my-pi/issues/4565))
|
||||
- Fixed bash tool pipeline execution preserving stale upstream output when the final stage was a stripped `head`/`tail` limiter; the tool now runs the command as written so `seq 1 5 | head -n2` returns only `1` and `2`. ([#4562](https://github.com/can1357/oh-my-pi/issues/4562))
|
||||
- Fixed the status-line token-rate segment rendering as `<number>/s`, which Ghostty auto-detected as a hyperlink on Ctrl+hover. ([#4541](https://github.com/can1357/oh-my-pi/issues/4541))
|
||||
- Fixed retry fallback model recovery by exposing `retry.fallbackChains` in `/settings`, adding a `/model` action to assign the selected default fallback model, and clearing a selected model's retry cooldown marker on manual model switches. ([#4533](https://github.com/can1357/oh-my-pi/issues/4533))
|
||||
- Fixed `/handoff` and auto-handoff skipping extension lifecycle hooks by emitting cancellable `session_before_switch` hooks and a `session_switch` with `reason: "handoff"` after the replacement session is ready ([#4434](https://github.com/can1357/oh-my-pi/issues/4434)).
|
||||
- Fixed TTSR stream interrupts so only the tool call whose stream matched a rule receives the rule-named abort result; sibling tool-call placeholders now use a neutral abort reason ([#2783](https://github.com/can1357/oh-my-pi/issues/2783)).
|
||||
|
||||
## [16.3.11] - 2026-07-06
|
||||
|
||||
### Changed
|
||||
|
||||
@@ -1294,6 +1294,151 @@ describe("advisor", () => {
|
||||
expect(failures).toHaveLength(2);
|
||||
});
|
||||
|
||||
it("calls onTurnError with state.error before retrying the batch", async () => {
|
||||
const promptInputs: string[] = [];
|
||||
const turnErrors: unknown[] = [];
|
||||
const events: string[] = [];
|
||||
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
|
||||
let promptCalls = 0;
|
||||
const agent: AdvisorAgent = {
|
||||
prompt: async input => {
|
||||
promptCalls++;
|
||||
promptInputs.push(input);
|
||||
events.push(`prompt:${promptCalls}`);
|
||||
state.error = promptCalls === 1 ? "provider failed" : undefined;
|
||||
},
|
||||
abort: () => {},
|
||||
reset: () => {
|
||||
state.error = undefined;
|
||||
},
|
||||
state,
|
||||
};
|
||||
const messages: AgentMessage[] = [{ role: "user", content: "aaa", timestamp: 1 } as AgentMessage];
|
||||
const host: AdvisorRuntimeHost = {
|
||||
snapshotMessages: () => messages,
|
||||
enqueueAdvice: () => {},
|
||||
onTurnError: error => {
|
||||
turnErrors.push(error);
|
||||
events.push(`hook:${error instanceof Error ? error.message : String(error)}`);
|
||||
},
|
||||
};
|
||||
const runtime = new AdvisorRuntime(agent, host, 1);
|
||||
|
||||
runtime.onTurnEnd(messages);
|
||||
await runtime.waitForCatchup(1000, 1);
|
||||
|
||||
expect(promptInputs).toHaveLength(2);
|
||||
expect(turnErrors).toHaveLength(1);
|
||||
const error = turnErrors[0];
|
||||
if (!(error instanceof Error)) throw new Error("expected advisor turn error");
|
||||
expect(error.message).toBe("provider failed");
|
||||
expect(events).toEqual(["prompt:1", "hook:provider failed", "prompt:2"]);
|
||||
expect(runtime.backlog).toBe(0);
|
||||
});
|
||||
|
||||
it("calls onTurnError for each consecutive failure including the dropped third turn", async () => {
|
||||
const promptInputs: string[] = [];
|
||||
const turnErrors: unknown[] = [];
|
||||
const failures: unknown[] = [];
|
||||
const events: string[] = [];
|
||||
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
|
||||
let promptCalls = 0;
|
||||
const agent: AdvisorAgent = {
|
||||
prompt: async input => {
|
||||
promptCalls++;
|
||||
promptInputs.push(input);
|
||||
events.push(`prompt:${promptCalls}`);
|
||||
state.error = `provider failed ${promptCalls}`;
|
||||
},
|
||||
abort: () => {},
|
||||
reset: () => {
|
||||
state.error = undefined;
|
||||
},
|
||||
state,
|
||||
};
|
||||
const messages: AgentMessage[] = [{ role: "user", content: "aaa", timestamp: 1 } as AgentMessage];
|
||||
const host: AdvisorRuntimeHost = {
|
||||
snapshotMessages: () => messages,
|
||||
enqueueAdvice: () => {},
|
||||
onTurnError: error => {
|
||||
turnErrors.push(error);
|
||||
events.push(`hook:${error instanceof Error ? error.message : String(error)}`);
|
||||
},
|
||||
notifyFailure: error => {
|
||||
failures.push(error);
|
||||
events.push(`notify:${error instanceof Error ? error.message : String(error)}`);
|
||||
},
|
||||
};
|
||||
const runtime = new AdvisorRuntime(agent, host, 1);
|
||||
|
||||
runtime.onTurnEnd(messages);
|
||||
await runtime.waitForCatchup(1000, 1);
|
||||
|
||||
expect(promptInputs).toHaveLength(3);
|
||||
expect(turnErrors.map(error => (error instanceof Error ? error.message : String(error)))).toEqual([
|
||||
"provider failed 1",
|
||||
"provider failed 2",
|
||||
"provider failed 3",
|
||||
]);
|
||||
expect(failures).toHaveLength(1);
|
||||
const failure = failures[0];
|
||||
if (!(failure instanceof Error)) throw new Error("expected advisor failure error");
|
||||
expect(failure.message).toBe("provider failed 3");
|
||||
expect(events).toEqual([
|
||||
"prompt:1",
|
||||
"hook:provider failed 1",
|
||||
"prompt:2",
|
||||
"hook:provider failed 2",
|
||||
"prompt:3",
|
||||
"hook:provider failed 3",
|
||||
"notify:provider failed 3",
|
||||
]);
|
||||
expect(runtime.backlog).toBe(0);
|
||||
});
|
||||
|
||||
it("continues retrying when onTurnError rejects", async () => {
|
||||
const promptInputs: string[] = [];
|
||||
const turnErrors: unknown[] = [];
|
||||
const events: string[] = [];
|
||||
const state: { messages: AgentMessage[]; error?: string } = { messages: [] };
|
||||
let promptCalls = 0;
|
||||
const agent: AdvisorAgent = {
|
||||
prompt: async input => {
|
||||
promptCalls++;
|
||||
promptInputs.push(input);
|
||||
events.push(`prompt:${promptCalls}`);
|
||||
state.error = promptCalls === 1 ? "provider failed" : undefined;
|
||||
},
|
||||
abort: () => {},
|
||||
reset: () => {
|
||||
state.error = undefined;
|
||||
},
|
||||
state,
|
||||
};
|
||||
const messages: AgentMessage[] = [{ role: "user", content: "aaa", timestamp: 1 } as AgentMessage];
|
||||
const host: AdvisorRuntimeHost = {
|
||||
snapshotMessages: () => messages,
|
||||
enqueueAdvice: () => {},
|
||||
onTurnError: async error => {
|
||||
turnErrors.push(error);
|
||||
events.push(`hook:${error instanceof Error ? error.message : String(error)}`);
|
||||
throw new Error("hook failed");
|
||||
},
|
||||
};
|
||||
const runtime = new AdvisorRuntime(agent, host, 1);
|
||||
|
||||
runtime.onTurnEnd(messages);
|
||||
await runtime.waitForCatchup(1000, 1);
|
||||
|
||||
expect(promptInputs).toHaveLength(2);
|
||||
expect(turnErrors).toHaveLength(1);
|
||||
const error = turnErrors[0];
|
||||
if (!(error instanceof Error)) throw new Error("expected advisor turn error");
|
||||
expect(error.message).toBe("provider failed");
|
||||
expect(events).toEqual(["prompt:1", "hook:provider failed", "prompt:2"]);
|
||||
expect(runtime.backlog).toBe(0);
|
||||
});
|
||||
|
||||
it("rolls advisor state back after each failed prompt so retries don't replay duplicate turns", async () => {
|
||||
// The real `Agent` appends the user batch + a synthetic `stopReason: "error"`
|
||||
// assistant turn before `state.error` is read. Without rollback, the runtime's
|
||||
|
||||
@@ -48,6 +48,17 @@ export interface AdvisorRuntimeHost {
|
||||
* one that routes `advise()` results back to the primary.
|
||||
*/
|
||||
beginAdvisorUpdate?(): void;
|
||||
/**
|
||||
* Called with the error of every failed advisor turn, before the retry sleep
|
||||
* or the dropped-after-3 path. Lets the host apply credential-level remedies
|
||||
* the advisor loop lacks: the in-stream a/b/c auth retry rotates through
|
||||
* sibling credentials within one request but never blocks the LAST failing
|
||||
* one — the primary agent's retry pipeline does that via
|
||||
* `markUsageLimitReached`, so without this hook the advisor re-picks the
|
||||
* same usage-limited account on every retry. Errors thrown here are logged
|
||||
* and swallowed.
|
||||
*/
|
||||
onTurnError?(error: unknown): Promise<void> | void;
|
||||
/** Surface a non-recovering advisor failure to the host UI without adding model-visible context. */
|
||||
notifyFailure?(error: unknown): void;
|
||||
}
|
||||
@@ -352,6 +363,14 @@ export class AdvisorRuntime {
|
||||
if (this.#epoch !== epoch) continue;
|
||||
this.#rollbackFailedTurn(messageSnapshot);
|
||||
logger.debug("advisor turn failed", { err: String(err) });
|
||||
try {
|
||||
await this.host.onTurnError?.(err);
|
||||
} catch (hookErr) {
|
||||
logger.debug("advisor onTurnError hook failed", { err: String(hookErr) });
|
||||
}
|
||||
// The hook awaits; a reset during it invalidates this batch like the
|
||||
// prompt await above — drop it instead of requeueing stale content.
|
||||
if (this.#epoch !== epoch) continue;
|
||||
this.#consecutiveFailures++;
|
||||
if (this.#consecutiveFailures >= 3) {
|
||||
logger.warn("advisor failed consecutively 3 times; dropping backlog to prevent stall");
|
||||
|
||||
@@ -49,7 +49,7 @@ export function createApiKeyResolver(
|
||||
options: ApiKeyResolverOptions = {},
|
||||
): ApiKeyResolver {
|
||||
const { sessionId, baseUrl, modelId } = options;
|
||||
return async ({ lastChance, error, signal }) => {
|
||||
return async ({ lastChance, error, signal, previousKey }) => {
|
||||
if (error === undefined) {
|
||||
return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId });
|
||||
}
|
||||
@@ -59,7 +59,12 @@ export function createApiKeyResolver(
|
||||
// sibling exists we switch immediately; the precise no-sibling backoff
|
||||
// is owned by `markUsageLimitReached` (default + server usage-report
|
||||
// reset) and the outer whole-turn retry layer.
|
||||
await registry.authStorage.rotateSessionCredential(provider, sessionId, { error, modelId, signal });
|
||||
await registry.authStorage.rotateSessionCredential(provider, sessionId, {
|
||||
error,
|
||||
modelId,
|
||||
signal,
|
||||
apiKey: previousKey,
|
||||
});
|
||||
return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId });
|
||||
}
|
||||
return registry.getApiKeyForProvider(provider, sessionId, { baseUrl, modelId, forceRefresh: true, signal });
|
||||
|
||||
@@ -2266,6 +2266,15 @@ export class ModelRegistry {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Clear the cooldown suppression for one selector after an explicit user selection.
|
||||
*/
|
||||
clearSuppressedSelector(selector: string): void {
|
||||
this.#suppressedSelectors.delete(
|
||||
normalizeSuppressedSelector(selector, (provider, id) => this.find(provider, id) !== undefined),
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Clear all cooldown suppressions recorded via {@link suppressSelector}.
|
||||
* Used to reset retry-fallback cooldown state without a full {@link refresh}.
|
||||
|
||||
@@ -1367,7 +1367,17 @@ export const SETTINGS_SCHEMA = {
|
||||
description: "Allow retry recovery to switch to configured fallback models",
|
||||
},
|
||||
},
|
||||
"retry.fallbackChains": { type: "record", default: {} as Record<string, string[]> },
|
||||
"retry.fallbackChains": {
|
||||
type: "record",
|
||||
default: {} as Record<string, string[]>,
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Retry & Fallback",
|
||||
label: "Retry Fallback Chains",
|
||||
description:
|
||||
'JSON object mapping model roles to ordered fallback model selectors, e.g. {"default":["openai/gpt-4o-mini"]}.',
|
||||
},
|
||||
},
|
||||
"retry.fallbackRevertPolicy": {
|
||||
type: "enum",
|
||||
values: ["cooldown-expiry", "never"] as const,
|
||||
|
||||
@@ -429,6 +429,9 @@ export class Settings {
|
||||
if (path === "statusLine.sessionAccent") {
|
||||
statusLineSessionAccentSignal.fire();
|
||||
}
|
||||
if (path === "modelRoles") {
|
||||
modelRolesSignal.fire();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -480,11 +483,13 @@ export class Settings {
|
||||
async reloadForCwd(cwd: string): Promise<void> {
|
||||
const normalized = path.normalize(cwd);
|
||||
if (normalized === this.#cwd) return;
|
||||
const prevModelRoles = this.get("modelRoles");
|
||||
this.#cwd = normalized;
|
||||
if (this.#persist) {
|
||||
this.#project = await this.#loadProjectSettings();
|
||||
}
|
||||
this.#rebuildMerged();
|
||||
this.#fireEffectiveSettingChanged("modelRoles", this.get("modelRoles"), prevModelRoles);
|
||||
this.#fireAllHooks();
|
||||
}
|
||||
|
||||
@@ -1477,6 +1482,12 @@ const appendOnlyModeSignal = new SettingSignal<[value: string]>("provider.append
|
||||
*/
|
||||
export const onAppendOnlyModeChanged = (cb: (value: string) => void) => appendOnlyModeSignal.on(cb);
|
||||
|
||||
/** Fires when any model role changes at runtime. */
|
||||
const modelRolesSignal = new SettingSignal("modelRoles");
|
||||
|
||||
/** Subscribe to model role changes. Returns an unsubscribe function. */
|
||||
export const onModelRolesChanged: (cb: () => void) => () => void = modelRolesSignal.on.bind(modelRolesSignal);
|
||||
|
||||
/** Fires when `statusLine.sessionAccent` changes at runtime. */
|
||||
const statusLineSessionAccentSignal = new SettingSignal("statusLine.sessionAccent");
|
||||
|
||||
|
||||
@@ -401,7 +401,8 @@ async function loadStickyRulesFile(filePath: string, level: "user" | "project"):
|
||||
const content = await readFile(filePath);
|
||||
if (!content) return null;
|
||||
const source = createSourceMeta(PROVIDER_ID, filePath, level);
|
||||
const rule = buildRuleFromMarkdown("RULES.md", content, filePath, source, { ruleName: "RULES" });
|
||||
const ruleName = level === "project" ? "RULES@project" : "RULES";
|
||||
const rule = buildRuleFromMarkdown("RULES.md", content, filePath, source, { ruleName });
|
||||
// Force alwaysApply regardless of frontmatter — the whole point of RULES.md
|
||||
// is to be reattached every turn.
|
||||
return { ...rule, alwaysApply: true };
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
* Loads configuration from ~/.claude/plugins/cache/ based on installed_plugins.json registry.
|
||||
* Priority: 70 (below claude.ts at 80, so user overrides in .claude/ take precedence)
|
||||
*/
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import { logger } from "@oh-my-pi/pi-utils";
|
||||
import { registerProvider } from "../capability";
|
||||
@@ -30,14 +31,14 @@ const DISPLAY_NAME = "Claude Code Marketplace";
|
||||
const PRIORITY = 70; // Below claude.ts (80) so user .claude/ overrides win
|
||||
|
||||
interface ClaudePluginManifest {
|
||||
skills?: string;
|
||||
"slash-commands"?: string;
|
||||
commands?: string;
|
||||
skills?: string | string[];
|
||||
"slash-commands"?: string | string[];
|
||||
commands?: string | string[];
|
||||
}
|
||||
|
||||
interface ResolvedPluginDir {
|
||||
dir: string;
|
||||
warning?: string;
|
||||
dirs: string[];
|
||||
warnings: string[];
|
||||
}
|
||||
|
||||
async function readPluginManifest(root: ClaudePluginRoot): Promise<ClaudePluginManifest | null> {
|
||||
@@ -54,43 +55,116 @@ async function readPluginManifest(root: ClaudePluginRoot): Promise<ClaudePluginM
|
||||
}
|
||||
}
|
||||
|
||||
function isRecord(value: unknown): value is Record<string, unknown> {
|
||||
return value !== null && typeof value === "object" && !Array.isArray(value);
|
||||
}
|
||||
|
||||
async function skillsManifestReplacesFallback(root: ClaudePluginRoot): Promise<boolean> {
|
||||
const raw = await readFile(path.join(root.path, "marketplace.json"));
|
||||
if (raw === null) return false;
|
||||
|
||||
try {
|
||||
const parsed: unknown = JSON.parse(raw);
|
||||
if (!isRecord(parsed)) return false;
|
||||
const plugins = parsed.plugins;
|
||||
return (
|
||||
Array.isArray(plugins) &&
|
||||
plugins.some(entry => isRecord(entry) && entry.name === root.plugin && entry.source === "./")
|
||||
);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
function isWithinPluginRoot(rootPath: string, targetPath: string): boolean {
|
||||
const relative = path.relative(rootPath, targetPath);
|
||||
return relative === "" || (!relative.startsWith("..") && !path.isAbsolute(relative));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a manifest-declared directory field to absolute paths within the
|
||||
* plugin root.
|
||||
*
|
||||
* Manifest path fields may be `string` or `string[]`
|
||||
* (https://code.claude.com/docs/en/plugins-reference#path-behavior-rules);
|
||||
* both shapes are normalized here. The first `manifestKeys` entry that
|
||||
* supplies at least one non-empty path wins (later keys are ignored — used for
|
||||
* the `commands` > `slash-commands` legacy fallback).
|
||||
*
|
||||
* `fallback` is the default subdirectory (e.g. `skills/`, `commands/`) and
|
||||
* `includeFallback` controls the Claude-documented merge semantic per field:
|
||||
*
|
||||
* - `skills` **adds to** the default: `fallback` is always scanned, and any
|
||||
* manifest entries load alongside it. Callers pass `includeFallback: true`.
|
||||
* - `commands` / `slash-commands` **replace** the default: an explicit
|
||||
* manifest key means the default `commands/` directory is not scanned.
|
||||
* Callers pass `includeFallback: false` (the manifest itself may still
|
||||
* list `./commands` explicitly to keep it).
|
||||
*
|
||||
* When no matching key is set, the fallback is used regardless. Entries that
|
||||
* resolve outside the plugin root are dropped with a warning so misconfigured
|
||||
* manifests remain observable and cannot escape via traversal.
|
||||
*/
|
||||
async function resolvePluginDir(
|
||||
root: ClaudePluginRoot,
|
||||
manifestKeys: ReadonlyArray<keyof ClaudePluginManifest>,
|
||||
fallback: string,
|
||||
includeFallback: boolean,
|
||||
): Promise<ResolvedPluginDir> {
|
||||
const manifest = await readPluginManifest(root);
|
||||
const fallbackDir = path.join(root.path, fallback);
|
||||
|
||||
let configured: string | undefined;
|
||||
let configured: string[] | undefined;
|
||||
let matchedKey: keyof ClaudePluginManifest | undefined;
|
||||
for (const key of manifestKeys) {
|
||||
const val = manifest?.[key];
|
||||
if (typeof val === "string" && val.trim()) {
|
||||
configured = val.trim();
|
||||
const candidates: string[] = [];
|
||||
if (typeof val === "string") {
|
||||
const trimmed = val.trim();
|
||||
if (trimmed) candidates.push(trimmed);
|
||||
} else if (Array.isArray(val)) {
|
||||
for (const entry of val) {
|
||||
if (typeof entry !== "string") continue;
|
||||
const trimmed = entry.trim();
|
||||
if (trimmed) candidates.push(trimmed);
|
||||
}
|
||||
}
|
||||
if (candidates.length > 0) {
|
||||
configured = candidates;
|
||||
matchedKey = key;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (configured === undefined) {
|
||||
return { dir: fallbackDir };
|
||||
return { dirs: [fallbackDir], warnings: [] };
|
||||
}
|
||||
|
||||
const resolved = path.resolve(root.path, configured);
|
||||
if (isWithinPluginRoot(root.path, resolved)) {
|
||||
return { dir: resolved };
|
||||
// Dedup preserves order: default entry (when included) first, then declared
|
||||
// entries in manifest order. Deduping the paths themselves means a plugin
|
||||
// author can still list `./commands` explicitly when they want the default
|
||||
// alongside extras without producing double-loads.
|
||||
const seen = new Set<string>();
|
||||
const dirs: string[] = [];
|
||||
const warnings: string[] = [];
|
||||
if (includeFallback) {
|
||||
seen.add(fallbackDir);
|
||||
dirs.push(fallbackDir);
|
||||
}
|
||||
for (const entry of configured) {
|
||||
const resolved = path.resolve(root.path, entry);
|
||||
if (!isWithinPluginRoot(root.path, resolved)) {
|
||||
warnings.push(
|
||||
`[claude-plugins] Ignoring ${String(matchedKey)} path outside plugin root for ${root.id}: ${entry}`,
|
||||
);
|
||||
continue;
|
||||
}
|
||||
if (seen.has(resolved)) continue;
|
||||
seen.add(resolved);
|
||||
dirs.push(resolved);
|
||||
}
|
||||
|
||||
return {
|
||||
dir: fallbackDir,
|
||||
warning: `[claude-plugins] Ignoring ${String(matchedKey)} path outside plugin root for ${root.id}: ${configured}`,
|
||||
};
|
||||
return { dirs, warnings };
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
@@ -104,24 +178,37 @@ async function loadSkills(ctx: LoadContext): Promise<LoadResult<Skill>> {
|
||||
warnings.push(...rootWarnings);
|
||||
const results = await Promise.all(
|
||||
roots.map(async root => {
|
||||
const { dir: skillsDir, warning } = await resolvePluginDir(root, ["skills"], "skills");
|
||||
const result = await scanSkillsFromDir(ctx, {
|
||||
dir: skillsDir,
|
||||
providerId: PROVIDER_ID,
|
||||
level: root.scope,
|
||||
});
|
||||
return { root, result, warning };
|
||||
const includeFallback = !(await skillsManifestReplacesFallback(root));
|
||||
const { dirs: skillsDirs, warnings: resolveWarnings } = await resolvePluginDir(
|
||||
root,
|
||||
["skills"],
|
||||
"skills",
|
||||
includeFallback,
|
||||
);
|
||||
const scanResults = await Promise.all(
|
||||
skillsDirs.map(dir =>
|
||||
scanSkillsFromDir(ctx, {
|
||||
dir,
|
||||
providerId: PROVIDER_ID,
|
||||
level: root.scope,
|
||||
includeSelf: true,
|
||||
}),
|
||||
),
|
||||
);
|
||||
return { scanResults, resolveWarnings };
|
||||
}),
|
||||
);
|
||||
for (const { result, warning } of results) {
|
||||
if (warning) warnings.push(warning);
|
||||
for (const { scanResults, resolveWarnings } of results) {
|
||||
warnings.push(...resolveWarnings);
|
||||
// Intentionally do NOT prefix skill names with `root.plugin`.
|
||||
// The `plugin:name` format breaks skill:// URL parsing (colons are
|
||||
// ambiguous with port separators) and is unintuitive for callers.
|
||||
// Dedup-by-key in the capability layer already handles name collisions
|
||||
// across providers using priority ordering.
|
||||
items.push(...result.items);
|
||||
if (result.warnings) warnings.push(...result.warnings);
|
||||
for (const result of scanResults) {
|
||||
items.push(...result.items);
|
||||
if (result.warnings) warnings.push(...result.warnings);
|
||||
}
|
||||
}
|
||||
return { items, warnings };
|
||||
}
|
||||
@@ -139,28 +226,62 @@ async function loadSlashCommands(ctx: LoadContext): Promise<LoadResult<SlashComm
|
||||
|
||||
const results = await Promise.all(
|
||||
roots.map(async root => {
|
||||
const { dir: commandsDir, warning } = await resolvePluginDir(root, ["commands", "slash-commands"], "commands");
|
||||
const commandResult = await loadFilesFromDir<SlashCommand>(ctx, commandsDir, PROVIDER_ID, root.scope, {
|
||||
extensions: ["md"],
|
||||
transform: (name, content, filePath, source) => {
|
||||
const cmdName = name.replace(/\.md$/, "");
|
||||
return {
|
||||
name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName,
|
||||
path: filePath,
|
||||
content,
|
||||
level: root.scope,
|
||||
_source: source,
|
||||
};
|
||||
},
|
||||
});
|
||||
return { commandResult, warning };
|
||||
const { dirs: commandsDirs, warnings: resolveWarnings } = await resolvePluginDir(
|
||||
root,
|
||||
["commands", "slash-commands"],
|
||||
"commands",
|
||||
false,
|
||||
);
|
||||
const commandResults = await Promise.all(
|
||||
commandsDirs.map(async dir => {
|
||||
try {
|
||||
const stats = await fs.stat(dir);
|
||||
if (stats.isFile()) {
|
||||
if (path.extname(dir) !== ".md") return { items: [], warnings: [] };
|
||||
const content = await readFile(dir);
|
||||
if (content === null) return { items: [], warnings: [`Failed to read file: ${dir}`] };
|
||||
const cmdName = path.basename(dir).replace(/\.md$/, "");
|
||||
return {
|
||||
items: [
|
||||
{
|
||||
name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName,
|
||||
path: dir,
|
||||
content,
|
||||
level: root.scope,
|
||||
_source: createSourceMeta(PROVIDER_ID, dir, root.scope),
|
||||
},
|
||||
],
|
||||
warnings: [],
|
||||
};
|
||||
}
|
||||
} catch {
|
||||
// Missing entries behave like missing directories: no items, no warning.
|
||||
}
|
||||
return loadFilesFromDir<SlashCommand>(ctx, dir, PROVIDER_ID, root.scope, {
|
||||
extensions: ["md"],
|
||||
transform: (name, content, filePath, source) => {
|
||||
const cmdName = name.replace(/\.md$/, "");
|
||||
return {
|
||||
name: root.plugin ? `${root.plugin}:${cmdName}` : cmdName,
|
||||
path: filePath,
|
||||
content,
|
||||
level: root.scope,
|
||||
_source: source,
|
||||
};
|
||||
},
|
||||
});
|
||||
}),
|
||||
);
|
||||
return { commandResults, resolveWarnings };
|
||||
}),
|
||||
);
|
||||
|
||||
for (const { commandResult, warning } of results) {
|
||||
if (warning) warnings.push(warning);
|
||||
items.push(...commandResult.items);
|
||||
if (commandResult.warnings) warnings.push(...commandResult.warnings);
|
||||
for (const { commandResults, resolveWarnings } of results) {
|
||||
warnings.push(...resolveWarnings);
|
||||
for (const commandResult of commandResults) {
|
||||
items.push(...commandResult.items);
|
||||
if (commandResult.warnings) warnings.push(...commandResult.warnings);
|
||||
}
|
||||
}
|
||||
|
||||
return { items, warnings };
|
||||
|
||||
@@ -312,6 +312,15 @@ export interface ScanSkillsFromDirOptions {
|
||||
providerId: string;
|
||||
level: "user" | "project";
|
||||
requireDescription?: boolean;
|
||||
/**
|
||||
* When true, treat a `SKILL.md` sitting directly under `dir` as a single skill in addition to
|
||||
* scanning `<dir>/<name>/SKILL.md` children. Matches the Claude plugin manifest convention
|
||||
* that lets a skill path point at a directory containing `SKILL.md` directly (e.g.
|
||||
* `"skills": ["./"]`), where the frontmatter `name` determines the invocation name and the
|
||||
* directory basename is the fallback. Default `false` preserves the strict child-scan
|
||||
* semantic every non-Claude provider relies on.
|
||||
*/
|
||||
includeSelf?: boolean;
|
||||
}
|
||||
|
||||
// Stable ordering used for skill lists in prompts: name (case-insensitive), then name, then path.
|
||||
@@ -368,7 +377,13 @@ export async function scanSkillsFromDir(
|
||||
}
|
||||
};
|
||||
|
||||
const work = [];
|
||||
const work: Promise<void>[] = [];
|
||||
if (options.includeSelf) {
|
||||
const selfSkillPath = path.join(dir, "SKILL.md");
|
||||
if (fs.existsSync(selfSkillPath)) {
|
||||
work.push(loadSkill(selfSkillPath));
|
||||
}
|
||||
}
|
||||
for (const entry of entries) {
|
||||
if (entry.name.startsWith(".")) continue;
|
||||
if (!entry.isDirectory() && !entry.isSymbolicLink()) continue;
|
||||
|
||||
@@ -679,21 +679,35 @@ function wrapEditRendererLine(line: string, width: number): string[] {
|
||||
const startAnsi = line.match(/^((?:\x1b\[[0-9;]*m)*)/)?.[1] ?? "";
|
||||
const bodyWithReset = line.slice(startAnsi.length);
|
||||
const body = bodyWithReset.endsWith("\x1b[39m") ? bodyWithReset.slice(0, -"\x1b[39m".length) : bodyWithReset;
|
||||
const diffMatch = /^([+\-\s])(\s*\d+)([|│])(.*)$/s.exec(body);
|
||||
// Gutter shapes produced by formatCodeFrameLine: "-315│", " 313│", "+322│",
|
||||
// plus the deduplicated forms " +│" and " │" whose repeated line number
|
||||
// renderDiff blanked (single-line replacement pairs and insert-then-context
|
||||
// runs) — all │-separated. ASCII "|" gutters exist only in raw canonical
|
||||
// diff rows passed through by the plain fallback ("-42|old", " 42|ctx"),
|
||||
// which always carry a marker column ("+"/"-"/space) and a line number. So
|
||||
// the number is optional for "│", while "|" requires the full canonical
|
||||
// shape; anything else (a body line merely starting with "|", error text
|
||||
// like "123|…") is not a diff row and wraps generically.
|
||||
const diffMatch = /^(\s*[+-]?\s*\d*)([|│])(.*)$/s.exec(body);
|
||||
|
||||
if (!diffMatch) {
|
||||
if (!diffMatch || diffMatch[1].length === 0 || (diffMatch[2] === "|" && !/^[+\-\s]\s*\d+$/.test(diffMatch[1]))) {
|
||||
return wrapTextWithAnsi(line, width);
|
||||
}
|
||||
|
||||
const [, marker, lineNum, separator, content] = diffMatch;
|
||||
const prefix = `${marker}${lineNum}${separator}`;
|
||||
const [, gutter, separator, content] = diffMatch;
|
||||
const prefix = `${gutter}${separator}`;
|
||||
const prefixWidth = visibleWidth(prefix);
|
||||
const contentWidth = Math.max(1, width - prefixWidth);
|
||||
const continuationPrefix = `${" ".repeat(Math.max(0, prefixWidth - 1))}${separator}`;
|
||||
const wrappedContent = wrapTextWithAnsi(content ?? "", contentWidth);
|
||||
|
||||
// Each visual row is a standalone terminal line: wrapTextWithAnsi re-opens
|
||||
// active SGR state at the next row's start, so a row that breaks inside an
|
||||
// intra-line diff highlight still ends with inverse video active. Close it
|
||||
// alongside the foreground reset — otherwise the frame padding appended
|
||||
// after the row is painted as an inverse block (default-foreground cells).
|
||||
return wrappedContent.map(
|
||||
(segment, index) => `${startAnsi}${index === 0 ? prefix : continuationPrefix}${segment}\x1b[39m`,
|
||||
(segment, index) => `${startAnsi}${index === 0 ? prefix : continuationPrefix}${segment}\x1b[27m\x1b[39m`,
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -1,6 +1,15 @@
|
||||
import { isMainThread } from "node:worker_threads";
|
||||
import { postmortem } from "@oh-my-pi/pi-utils";
|
||||
import { ToolError } from "../../tools/tool-errors";
|
||||
import { JsRuntime, type RuntimeHooks } from "./shared/runtime";
|
||||
import type { RunErrorPayload, SessionSnapshot, ToolReply, Transport, WorkerInbound } from "./worker-protocol";
|
||||
import type {
|
||||
RunErrorPayload,
|
||||
SessionSnapshot,
|
||||
ToolReply,
|
||||
Transport,
|
||||
WorkerInbound,
|
||||
WorkerOutbound,
|
||||
} from "./worker-protocol";
|
||||
|
||||
interface PendingTool {
|
||||
runId: string;
|
||||
@@ -10,9 +19,17 @@ interface PendingTool {
|
||||
|
||||
interface ActiveRun {
|
||||
runId: string;
|
||||
filename: string;
|
||||
pendingTools: Map<string, PendingTool>;
|
||||
/** Rejections floated by this run's cell code, captured before its result was sent. */
|
||||
floatingRejections: unknown[];
|
||||
}
|
||||
|
||||
type RunResult = Extract<WorkerOutbound, { type: "result" }>;
|
||||
|
||||
/** Finished-cell filenames retained for attributing rejections that surface after the run settled. */
|
||||
const RECENT_CELL_FILES_MAX = 256;
|
||||
|
||||
function errorPayload(error: unknown): RunErrorPayload {
|
||||
if (error instanceof Error) {
|
||||
return {
|
||||
@@ -34,15 +51,134 @@ function errorFromPayload(payload: RunErrorPayload): Error {
|
||||
return error;
|
||||
}
|
||||
|
||||
/**
|
||||
* Fold rejections floated by cell code into the run result: an otherwise
|
||||
* successful run fails with the first floating rejection (an unawaited promise
|
||||
* failing is a cell failure, not a success with noise); the rest surface as
|
||||
* output text so nothing is silently dropped.
|
||||
*/
|
||||
function foldFloatingRejections(active: ActiveRun, result: RunResult, hooks: RuntimeHooks): RunResult {
|
||||
const rejections = active.floatingRejections;
|
||||
if (rejections.length === 0) return result;
|
||||
let folded = result;
|
||||
let reported = rejections;
|
||||
if (result.ok) {
|
||||
const error = errorPayload(rejections[0]);
|
||||
error.message = `Unhandled rejection (missing await?): ${error.message}`;
|
||||
folded = { type: "result", runId: active.runId, ok: false, error };
|
||||
reported = rejections.slice(1);
|
||||
}
|
||||
for (const reason of reported) {
|
||||
const payload = errorPayload(reason);
|
||||
hooks.onText(`[unhandled rejection] ${payload.name ?? "Error"}: ${payload.message}\n`);
|
||||
}
|
||||
return folded;
|
||||
}
|
||||
|
||||
export class WorkerCore {
|
||||
#transport: Transport;
|
||||
#runtime: JsRuntime | null = null;
|
||||
#runs = new Map<string, ActiveRun>();
|
||||
#recentCellFiles = new Set<string>();
|
||||
#unsubscribe: () => void;
|
||||
#uninstallRejectionGuard: () => void;
|
||||
|
||||
constructor(transport: Transport) {
|
||||
this.#transport = transport;
|
||||
this.#unsubscribe = transport.onMessage(msg => this.#handle(msg));
|
||||
this.#uninstallRejectionGuard = this.#installRejectionGuard();
|
||||
}
|
||||
|
||||
/**
|
||||
* Capture unhandled rejections floated by eval-cell code (unawaited async
|
||||
* calls) so they fail the owning run instead of tearing down the worker or —
|
||||
* via the global postmortem handler — the whole session. On the main thread
|
||||
* (inline fallback) only cell-attributable rejections are consumed; in the
|
||||
* dedicated worker realm a rejection during a live run is cell activity even
|
||||
* without a usable stack, while anything else keeps its default fatality.
|
||||
*/
|
||||
#installRejectionGuard(): () => void {
|
||||
if (isMainThread) {
|
||||
return postmortem.interceptUnhandledRejections(reason => this.#consumeRejection(reason));
|
||||
}
|
||||
const onRejection = (reason: unknown): void => {
|
||||
if (this.#consumeRejection(reason)) return;
|
||||
// Not cell-attributable: restore default fatality. Rethrowing from a
|
||||
// timer surfaces it as an uncaught exception, which reaches the host
|
||||
// as a worker `error` event exactly like an unhandled rejection did
|
||||
// before this listener existed.
|
||||
setTimeout(() => {
|
||||
throw reason;
|
||||
}, 0);
|
||||
};
|
||||
process.on("unhandledRejection", onRejection);
|
||||
return () => {
|
||||
process.off("unhandledRejection", onRejection);
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Attribute an unhandled rejection to eval-cell code. Live runs are stashed
|
||||
* on the run (folded into its result after the settle drain); finished cells
|
||||
* downgrade to a host-side warn log. Returns false when the rejection is not
|
||||
* cell activity and must keep the default fatal path.
|
||||
*/
|
||||
#consumeRejection(reason: unknown): boolean {
|
||||
const stack = reason instanceof Error && typeof reason.stack === "string" ? reason.stack : undefined;
|
||||
if (stack) {
|
||||
// The stack can name several cells (helper defined by an earlier cell,
|
||||
// called from the live one); the outermost matching frame is the caller
|
||||
// that owns the floating promise.
|
||||
let owner: ActiveRun | undefined;
|
||||
let ownerIndex = -1;
|
||||
for (const run of this.#runs.values()) {
|
||||
const index = stack.lastIndexOf(run.filename);
|
||||
if (index > ownerIndex) {
|
||||
ownerIndex = index;
|
||||
owner = run;
|
||||
}
|
||||
}
|
||||
if (owner) {
|
||||
owner.floatingRejections.push(reason);
|
||||
return true;
|
||||
}
|
||||
let recent: string | undefined;
|
||||
let recentIndex = -1;
|
||||
for (const filename of this.#recentCellFiles) {
|
||||
const index = stack.lastIndexOf(filename);
|
||||
if (index > recentIndex) {
|
||||
recentIndex = index;
|
||||
recent = filename;
|
||||
}
|
||||
}
|
||||
if (recent) {
|
||||
this.#transport.send({
|
||||
type: "log",
|
||||
level: "warn",
|
||||
msg: "Unhandled rejection from a finished eval cell (missing await?)",
|
||||
meta: { filename: recent, error: errorPayload(reason) },
|
||||
});
|
||||
return true;
|
||||
}
|
||||
}
|
||||
if (!isMainThread && this.#runs.size > 0) {
|
||||
// Dedicated eval worker: during a live run, a rejection without a cell
|
||||
// frame (e.g. `Promise.reject("msg")` or a library-created reason) is
|
||||
// still cell activity — nothing else runs user code in this realm.
|
||||
if (this.#runs.size === 1) {
|
||||
const only = this.#runs.values().next().value;
|
||||
only?.floatingRejections.push(reason);
|
||||
return true;
|
||||
}
|
||||
this.#transport.send({
|
||||
type: "log",
|
||||
level: "warn",
|
||||
msg: "Unhandled rejection during concurrent eval runs; cannot attribute to a cell",
|
||||
meta: { error: errorPayload(reason) },
|
||||
});
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
#handle(msg: WorkerInbound): void {
|
||||
@@ -77,23 +213,42 @@ export class WorkerCore {
|
||||
}
|
||||
|
||||
async #runOne(runId: string, code: string, filename: string, snapshot: SessionSnapshot): Promise<void> {
|
||||
const runtime = this.#ensureRuntime(snapshot);
|
||||
runtime.setCwd(snapshot.cwd);
|
||||
const active: ActiveRun = { runId, pendingTools: new Map() };
|
||||
const active: ActiveRun = { runId, filename, pendingTools: new Map(), floatingRejections: [] };
|
||||
this.#runs.set(runId, active);
|
||||
const hooks: RuntimeHooks = {
|
||||
onText: chunk => this.#transport.send({ type: "text", runId, chunk }),
|
||||
onDisplay: output => this.#transport.send({ type: "display", runId, output }),
|
||||
callTool: (name, args) => this.#callTool(active, name, args),
|
||||
};
|
||||
let result: RunResult;
|
||||
try {
|
||||
const runtime = this.#ensureRuntime(snapshot);
|
||||
runtime.setCwd(snapshot.cwd);
|
||||
const value = await runtime.run(code, filename, hooks, { runId, cwd: snapshot.cwd });
|
||||
runtime.displayValue(value, hooks);
|
||||
this.#transport.send({ type: "result", runId, ok: true });
|
||||
result = { type: "result", runId, ok: true };
|
||||
} catch (error) {
|
||||
this.#transport.send({ type: "result", runId, ok: false, error: errorPayload(error) });
|
||||
result = { type: "result", runId, ok: false, error: errorPayload(error) };
|
||||
}
|
||||
try {
|
||||
// One event-loop turn so rejections the cell already floated surface
|
||||
// while this run can still own them (rejection callbacks run before
|
||||
// timers fire).
|
||||
await Bun.sleep(0);
|
||||
result = foldFloatingRejections(active, result, hooks);
|
||||
} finally {
|
||||
this.#runs.delete(runId);
|
||||
this.#rememberCellFile(filename);
|
||||
this.#transport.send(result);
|
||||
}
|
||||
}
|
||||
|
||||
#rememberCellFile(filename: string): void {
|
||||
this.#recentCellFiles.delete(filename);
|
||||
this.#recentCellFiles.add(filename);
|
||||
if (this.#recentCellFiles.size > RECENT_CELL_FILES_MAX) {
|
||||
const oldest = this.#recentCellFiles.values().next().value;
|
||||
if (oldest !== undefined) this.#recentCellFiles.delete(oldest);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -127,6 +282,7 @@ export class WorkerCore {
|
||||
this.#runtime?.dispose?.();
|
||||
this.#runtime = null;
|
||||
this.#transport.send({ type: "closed" });
|
||||
this.#uninstallRejectionGuard();
|
||||
this.#unsubscribe();
|
||||
this.#transport.close();
|
||||
}
|
||||
@@ -141,6 +297,7 @@ export class WorkerCore {
|
||||
this.#runs.clear();
|
||||
this.#runtime?.dispose?.();
|
||||
this.#runtime = null;
|
||||
this.#uninstallRejectionGuard();
|
||||
this.#unsubscribe();
|
||||
try {
|
||||
this.#transport.close();
|
||||
|
||||
@@ -14,6 +14,7 @@ import { buildNonInteractiveEnv } from "./non-interactive-env";
|
||||
|
||||
export interface BashExecutorOptions {
|
||||
cwd?: string;
|
||||
/** Milliseconds before aborting the command; 0 disables the executor deadline. */
|
||||
timeout?: number;
|
||||
onChunk?: (chunk: string) => void;
|
||||
chunkThrottleMs?: number;
|
||||
@@ -296,11 +297,15 @@ export async function executeBash(command: string, options?: BashExecutorOptions
|
||||
|
||||
let timeoutTimer: NodeJS.Timeout | undefined;
|
||||
const timeoutDeferred = Promise.withResolvers<"timeout">();
|
||||
const baseTimeoutMs = Math.max(1_000, options?.timeout ?? 300_000);
|
||||
timeoutTimer = setTimeout(() => {
|
||||
abortCurrentExecution();
|
||||
timeoutDeferred.resolve("timeout");
|
||||
}, baseTimeoutMs);
|
||||
const requestedTimeoutMs = options?.timeout;
|
||||
const deadlineTimeoutMs = requestedTimeoutMs === 0 ? undefined : Math.max(1_000, requestedTimeoutMs ?? 300_000);
|
||||
const nativeTimeoutMs = requestedTimeoutMs !== undefined && requestedTimeoutMs > 0 ? requestedTimeoutMs : undefined;
|
||||
if (deadlineTimeoutMs !== undefined) {
|
||||
timeoutTimer = setTimeout(() => {
|
||||
abortCurrentExecution();
|
||||
timeoutDeferred.resolve("timeout");
|
||||
}, deadlineTimeoutMs);
|
||||
}
|
||||
|
||||
let resetSession = false;
|
||||
|
||||
@@ -311,7 +316,7 @@ export async function executeBash(command: string, options?: BashExecutorOptions
|
||||
command: finalCommand,
|
||||
cwd: commandCwd,
|
||||
env: commandEnv,
|
||||
timeoutMs: options?.timeout,
|
||||
timeoutMs: nativeTimeoutMs,
|
||||
signal: runAbortController.signal,
|
||||
},
|
||||
(err, chunk) => {
|
||||
@@ -328,7 +333,7 @@ export async function executeBash(command: string, options?: BashExecutorOptions
|
||||
sessionEnv: shellEnv,
|
||||
snapshotPath: snapshotPath ?? undefined,
|
||||
minimizer,
|
||||
timeoutMs: options?.timeout,
|
||||
timeoutMs: nativeTimeoutMs,
|
||||
signal: runAbortController.signal,
|
||||
},
|
||||
(err, chunk) => {
|
||||
@@ -359,8 +364,8 @@ export async function executeBash(command: string, options?: BashExecutorOptions
|
||||
exitCode: undefined,
|
||||
cancelled: true,
|
||||
...(await sink.dump(
|
||||
winner.kind === "timeout"
|
||||
? `Command timed out after ${Math.round(baseTimeoutMs / 1000)} seconds`
|
||||
winner.kind === "timeout" && deadlineTimeoutMs !== undefined
|
||||
? `Command timed out after ${Math.round(deadlineTimeoutMs / 1000)} seconds`
|
||||
: "Command cancelled",
|
||||
)),
|
||||
};
|
||||
|
||||
@@ -1059,7 +1059,8 @@ async function collectExtensionModules(entryRealPath: string): Promise<Map<strin
|
||||
let resolved: string | null = null;
|
||||
let nextFollowsBareDependencies = followBareDependencies;
|
||||
if (specifier.startsWith(".")) {
|
||||
resolved = await realpathOrSelf(Bun.resolveSync(specifier, dir));
|
||||
const candidate = Bun.resolveSync(specifier, dir);
|
||||
resolved = hasSourceModuleExtension(candidate) ? await realpathOrSelf(candidate) : null;
|
||||
} else if (specifier.startsWith("#")) {
|
||||
resolved = await resolvePackageImportSpecifier(specifier, file);
|
||||
} else if (
|
||||
@@ -1078,7 +1079,10 @@ async function collectExtensionModules(entryRealPath: string): Promise<Map<strin
|
||||
dependencyExtension === ".cjs" ||
|
||||
dependencyExtension === ".cts" ||
|
||||
((dependencyExtension === ".js" || dependencyExtension === ".jsx") && manifest?.type !== "module");
|
||||
resolved = dependencyEntry && !isCommonJsEntry ? await realpathOrSelf(dependencyEntry) : null;
|
||||
resolved =
|
||||
dependencyEntry && hasSourceModuleExtension(dependencyEntry) && !isCommonJsEntry
|
||||
? await realpathOrSelf(dependencyEntry)
|
||||
: null;
|
||||
nextFollowsBareDependencies = false;
|
||||
}
|
||||
if (resolved && !modules.has(resolved)) {
|
||||
|
||||
@@ -196,19 +196,20 @@ export function parseMarketplaceCatalog(content: string, filePath: string): Mark
|
||||
* Catalog paths tried in priority order: omp-namespaced override first, then
|
||||
* the Claude Code-compatible fallback so existing marketplaces keep loading.
|
||||
*/
|
||||
const CATALOG_RELATIVE_PATHS: readonly string[] = [
|
||||
path.join(".omp-plugin", "marketplace.json"),
|
||||
path.join(".claude-plugin", "marketplace.json"),
|
||||
];
|
||||
const CATALOG_RELATIVE_PATHS: readonly string[] = [".omp-plugin/marketplace.json", ".claude-plugin/marketplace.json"];
|
||||
|
||||
async function readMarketplaceCatalog(root: string): Promise<{ catalogPath: string; content: string }> {
|
||||
async function readMarketplaceCatalog(
|
||||
root: string,
|
||||
options: { relativeDisplayPaths?: boolean } = {},
|
||||
): Promise<{ catalogPath: string; displayPath: string; content: string }> {
|
||||
const tried: string[] = [];
|
||||
for (const rel of CATALOG_RELATIVE_PATHS) {
|
||||
const catalogPath = path.join(root, rel);
|
||||
tried.push(catalogPath);
|
||||
const catalogPath = path.join(root, ...rel.split("/"));
|
||||
const displayPath = options.relativeDisplayPaths ? rel : catalogPath;
|
||||
tried.push(displayPath);
|
||||
try {
|
||||
const content = await Bun.file(catalogPath).text();
|
||||
return { catalogPath, content };
|
||||
return { catalogPath, displayPath, content };
|
||||
} catch (err) {
|
||||
if (isEnoent(err)) continue;
|
||||
throw err;
|
||||
@@ -252,11 +253,11 @@ export async function fetchMarketplace(source: string, cacheDir: string): Promis
|
||||
|
||||
if (type === "github") {
|
||||
const url = `https://github.com/${source}.git`;
|
||||
return cloneAndReadCatalog(url, cacheDir);
|
||||
return cloneAndReadCatalog(url, source, cacheDir);
|
||||
}
|
||||
|
||||
if (type === "git") {
|
||||
return cloneAndReadCatalog(source, cacheDir);
|
||||
return cloneAndReadCatalog(source, source, cacheDir);
|
||||
}
|
||||
|
||||
// type === "url"
|
||||
@@ -284,7 +285,7 @@ export async function fetchMarketplace(source: string, cacheDir: string): Promis
|
||||
* responsible for promoting the clone to its final cache location via
|
||||
* `promoteCloneToCache` after any duplicate/drift checks pass.
|
||||
*/
|
||||
async function cloneAndReadCatalog(url: string, cacheDir: string): Promise<FetchResult> {
|
||||
async function cloneAndReadCatalog(url: string, source: string, cacheDir: string): Promise<FetchResult> {
|
||||
const tmpDir = path.join(cacheDir, `.tmp-clone-${Date.now()}`);
|
||||
await fs.mkdir(cacheDir, { recursive: true });
|
||||
|
||||
@@ -292,12 +293,12 @@ async function cloneAndReadCatalog(url: string, cacheDir: string): Promise<Fetch
|
||||
await git.clone(url, tmpDir);
|
||||
|
||||
try {
|
||||
const { catalogPath, content } = await readMarketplaceCatalog(tmpDir);
|
||||
const catalog = parseMarketplaceCatalog(content, catalogPath);
|
||||
const { displayPath, content } = await readMarketplaceCatalog(tmpDir, { relativeDisplayPaths: true });
|
||||
const catalog = parseMarketplaceCatalog(content, displayPath);
|
||||
return { catalog, clonePath: tmpDir };
|
||||
} catch (err) {
|
||||
await fs.rm(tmpDir, { recursive: true, force: true }).catch(() => {});
|
||||
throw new Error(`Cloned repository ${url}: ${(err as Error).message}`, { cause: err });
|
||||
throw new Error(`Cloned repository ${url}: ${(err as Error).message} (source: ${source})`, { cause: err });
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -33,7 +33,7 @@ export interface SessionStartEvent {
|
||||
export interface SessionBeforeSwitchEvent {
|
||||
type: "session_before_switch";
|
||||
/** Reason for the switch */
|
||||
reason: "new" | "resume" | "fork";
|
||||
reason: "new" | "resume" | "fork" | "handoff";
|
||||
/** Session file we're switching to (only for "resume") */
|
||||
targetSessionFile?: string;
|
||||
}
|
||||
@@ -42,7 +42,7 @@ export interface SessionBeforeSwitchEvent {
|
||||
export interface SessionSwitchEvent {
|
||||
type: "session_switch";
|
||||
/** Reason for the switch */
|
||||
reason: "new" | "resume" | "fork";
|
||||
reason: "new" | "resume" | "fork" | "handoff";
|
||||
/** Session file we came from */
|
||||
previousSessionFile: string | undefined;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
import { afterAll, afterEach, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as path from "node:path";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { AgentRegistry } from "../../registry/agent-registry";
|
||||
import type { AgentSession } from "../../session/agent-session";
|
||||
import { ArtifactManager } from "../../session/artifacts";
|
||||
import { AgentProtocolHandler } from "../agent-protocol";
|
||||
import { resetRegisteredArtifactDirsForTests } from "../registry-helpers";
|
||||
|
||||
const tempDir = TempDir.createSync("omp-nested-agent-repro-");
|
||||
afterEach(() => {
|
||||
AgentRegistry.resetGlobalForTests();
|
||||
resetRegisteredArtifactDirsForTests();
|
||||
});
|
||||
afterAll(() => {
|
||||
tempDir.removeSync();
|
||||
});
|
||||
|
||||
it("agent:// resolves a depth-2 subagent's .md output while its session is live and artifact-manager-adopted", async () => {
|
||||
const root = tempDir.path();
|
||||
const rootSessionFile = path.join(root, "session.jsonl");
|
||||
const rootArtifactsDir = rootSessionFile.slice(0, -6);
|
||||
await fs.mkdir(rootArtifactsDir, { recursive: true });
|
||||
// Every subagent adopts the root ArtifactManager and reports its dir.
|
||||
const sharedArtifactManager = new ArtifactManager(rootArtifactsDir);
|
||||
|
||||
// A depth-1 subagent's OWN children are written under its own
|
||||
// sessionFile.slice(0, -6) (task/index.ts), i.e. one level deeper.
|
||||
const midSessionFile = path.join(rootArtifactsDir, "CodexDeepDive.jsonl");
|
||||
const midOwnArtifactsDir = midSessionFile.slice(0, -6);
|
||||
await fs.mkdir(midOwnArtifactsDir, { recursive: true });
|
||||
|
||||
const grandchildId = "CodexDeepDive.GraphStore";
|
||||
const grandchildSessionFile = path.join(midOwnArtifactsDir, `${grandchildId}.jsonl`);
|
||||
await fs.writeFile(path.join(midOwnArtifactsDir, `${grandchildId}.md`), "full report content");
|
||||
|
||||
const fakeSession = {
|
||||
sessionManager: { getArtifactsDir: () => sharedArtifactManager.dir },
|
||||
} as unknown as AgentSession;
|
||||
const registry = AgentRegistry.global();
|
||||
registry.register({
|
||||
id: "Main",
|
||||
displayName: "main",
|
||||
kind: "main",
|
||||
session: fakeSession,
|
||||
sessionFile: rootSessionFile,
|
||||
});
|
||||
registry.register({
|
||||
id: "CodexDeepDive",
|
||||
displayName: "sub",
|
||||
kind: "sub",
|
||||
parentId: "Main",
|
||||
session: fakeSession,
|
||||
sessionFile: midSessionFile,
|
||||
});
|
||||
registry.register({
|
||||
id: grandchildId,
|
||||
displayName: "sub",
|
||||
kind: "sub",
|
||||
parentId: "CodexDeepDive",
|
||||
session: fakeSession,
|
||||
sessionFile: grandchildSessionFile,
|
||||
});
|
||||
|
||||
const resource = await new AgentProtocolHandler().resolve(new URL(`agent://${grandchildId}`) as never);
|
||||
expect(resource.content).toBe("full report content");
|
||||
});
|
||||
@@ -20,11 +20,13 @@ export function resetRegisteredArtifactDirsForTests(): void {
|
||||
/**
|
||||
* Snapshot of artifacts dirs for every registered session, deduped.
|
||||
*
|
||||
* Prefers `sessionManager.getArtifactsDir()` because subagents adopt their
|
||||
* parent's `ArtifactManager` and report the parent's dir there; dedup then
|
||||
* collapses parent + N subagents (the whole agent tree) to one entry. Falls
|
||||
* back to the raw session file (with the `.jsonl` suffix stripped) when no
|
||||
* live session reference is attached.
|
||||
* Collects TWO candidate dirs per ref, because a subagent reads from its
|
||||
* adopted (root-wide) `ArtifactManager.dir` but its own children are written
|
||||
* one level deeper, under `sessionFile.slice(0, -6)` (`task/index.ts`). A
|
||||
* depth-2+ subagent's output therefore lives in the write-time dir, not the
|
||||
* adopted one, so `agent://` must scan both or it 404s a live nested peer.
|
||||
* `addDir` dedup collapses the depth-0 case (both formulas agree) back to a
|
||||
* single entry.
|
||||
*/
|
||||
export function artifactsDirsFromRegistry(): string[] {
|
||||
const dirs: string[] = [];
|
||||
@@ -33,7 +35,8 @@ export function artifactsDirsFromRegistry(): string[] {
|
||||
if (!dirs.includes(dir)) dirs.push(dir);
|
||||
};
|
||||
for (const ref of AgentRegistry.global().list()) {
|
||||
addDir(ref.session?.sessionManager.getArtifactsDir() ?? (ref.sessionFile ? ref.sessionFile.slice(0, -6) : null));
|
||||
addDir(ref.session?.sessionManager.getArtifactsDir());
|
||||
if (ref.sessionFile) addDir(ref.sessionFile.slice(0, -6));
|
||||
}
|
||||
for (const dir of extraArtifactsDirs) addDir(dir);
|
||||
return dirs;
|
||||
|
||||
@@ -90,16 +90,20 @@ interface RoleAssignment {
|
||||
autoSelected: boolean;
|
||||
}
|
||||
|
||||
type ModelSelectorAction = "modelRole" | "retryFallback";
|
||||
|
||||
type RoleSelectCallback = (
|
||||
model: Model,
|
||||
role: string | null,
|
||||
thinkingLevel?: ConfiguredThinkingLevel,
|
||||
selector?: string,
|
||||
action?: ModelSelectorAction,
|
||||
) => void;
|
||||
type CancelCallback = () => void;
|
||||
interface MenuRoleAction {
|
||||
label: string;
|
||||
role: string; // now accepts custom role strings
|
||||
role: string;
|
||||
action: ModelSelectorAction;
|
||||
}
|
||||
|
||||
interface ProviderTabState {
|
||||
@@ -284,14 +288,19 @@ export class ModelSelectorComponent extends Container {
|
||||
}
|
||||
|
||||
#buildMenuRoleActions(): void {
|
||||
this.#menuRoleActions = getKnownRoleIds(this.#settings).map(role => {
|
||||
const roleActions = getKnownRoleIds(this.#settings).map(role => {
|
||||
const roleInfo = getRoleInfo(role, this.#settings);
|
||||
const roleLabel = roleInfo.tag ? `${roleInfo.tag} (${roleInfo.name})` : roleInfo.name;
|
||||
return {
|
||||
label: `Set as ${roleLabel}`,
|
||||
role,
|
||||
action: "modelRole" as const,
|
||||
};
|
||||
});
|
||||
this.#menuRoleActions = [
|
||||
...roleActions,
|
||||
{ label: "Set as DEFAULT retry fallback", role: "default", action: "retryFallback" },
|
||||
];
|
||||
}
|
||||
|
||||
#loadRoleModels(autoCandidateModels?: ReadonlyArray<Model>): void {
|
||||
@@ -1195,6 +1204,11 @@ export class ModelSelectorComponent extends Container {
|
||||
if (this.#menuStep === "role") {
|
||||
const action = this.#menuRoleActions[this.#menuSelectedIndex];
|
||||
if (!action) return;
|
||||
if (action.action === "retryFallback") {
|
||||
this.#handleSelect(selectedItem, action.role, undefined, action.action);
|
||||
this.#closeMenu();
|
||||
return;
|
||||
}
|
||||
this.#menuSelectedRole = action.role;
|
||||
this.#menuStep = "thinking";
|
||||
this.#menuSelectedIndex = this.#getThinkingPreselectIndex(action.role, selectedItem.model);
|
||||
@@ -1206,7 +1220,7 @@ export class ModelSelectorComponent extends Container {
|
||||
const thinkingOptions = this.#getThinkingLevelsForModel(selectedItem.model);
|
||||
const thinkingLevel = thinkingOptions[this.#menuSelectedIndex];
|
||||
if (!thinkingLevel) return;
|
||||
this.#handleSelect(selectedItem, this.#menuSelectedRole, thinkingLevel);
|
||||
this.#handleSelect(selectedItem, this.#menuSelectedRole, thinkingLevel, "modelRole");
|
||||
this.#closeMenu();
|
||||
return;
|
||||
}
|
||||
@@ -1225,13 +1239,23 @@ export class ModelSelectorComponent extends Container {
|
||||
}
|
||||
}
|
||||
|
||||
#handleSelect(item: ModelItem, role: string | null, thinkingLevel?: ConfiguredThinkingLevel): void {
|
||||
#handleSelect(
|
||||
item: ModelItem,
|
||||
role: string | null,
|
||||
thinkingLevel?: ConfiguredThinkingLevel,
|
||||
action: ModelSelectorAction = "modelRole",
|
||||
): void {
|
||||
if (this.#isItemDisabled(item)) {
|
||||
return;
|
||||
}
|
||||
// For temporary role, don't save to settings - just notify caller
|
||||
if (role === null) {
|
||||
this.#onSelectCallback(item.model, null, undefined, item.selector);
|
||||
this.#onSelectCallback(item.model, null, undefined, item.selector, action);
|
||||
return;
|
||||
}
|
||||
|
||||
if (action === "retryFallback") {
|
||||
this.#onSelectCallback(item.model, role, undefined, item.selector, action);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1241,7 +1265,7 @@ export class ModelSelectorComponent extends Container {
|
||||
this.#roles[role] = { model: item.model, thinkingLevel: selectedThinkingLevel, autoSelected: false };
|
||||
|
||||
// Notify caller (for updating agent state if needed)
|
||||
this.#onSelectCallback(item.model, role, selectedThinkingLevel, item.selector);
|
||||
this.#onSelectCallback(item.model, role, selectedThinkingLevel, item.selector, action);
|
||||
|
||||
// Update list to show new badges
|
||||
this.#updateList();
|
||||
|
||||
@@ -180,7 +180,7 @@ function pathToSettingDef(path: SettingPath): SettingDef | null {
|
||||
}
|
||||
|
||||
if (schemaType === "record") {
|
||||
return path === "providers.maxInFlightRequests" ? { ...base, type: "providerLimits" } : null;
|
||||
return path === "providers.maxInFlightRequests" ? { ...base, type: "providerLimits" } : { ...base, type: "text" };
|
||||
}
|
||||
|
||||
return null;
|
||||
|
||||
@@ -1235,9 +1235,21 @@ export class StatusLineComponent implements Component {
|
||||
}
|
||||
}
|
||||
}
|
||||
const leftOverflowDropIndex = (): number => {
|
||||
// Preserve the current working directory as long as possible. The
|
||||
// previous right-to-left pop could collapse a normal-width bar to
|
||||
// just the model segment, hiding the path before less-critical left
|
||||
// segments such as model/mode/collab were removed.
|
||||
for (let i = leftSegIds.length - 1; i >= 0; i--) {
|
||||
if (leftSegIds[i] !== "path") return i;
|
||||
}
|
||||
return left.length - 1;
|
||||
};
|
||||
|
||||
while (totalWidth() > topFillWidth && left.length > 0) {
|
||||
left.pop();
|
||||
leftSegIds.pop();
|
||||
const dropIdx = leftOverflowDropIndex();
|
||||
left.splice(dropIdx, 1);
|
||||
leftSegIds.splice(dropIdx, 1);
|
||||
leftWidth = groupWidth(left, leftCapWidth, leftSepWidth);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -849,11 +849,7 @@ export class CommandController {
|
||||
}
|
||||
|
||||
async #runNewSessionFlow(options?: NewSessionOptions, label: string = "New session started"): Promise<void> {
|
||||
if (this.ctx.loadingAnimation) {
|
||||
this.ctx.loadingAnimation.stop();
|
||||
this.ctx.loadingAnimation = undefined;
|
||||
}
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.clearTransientSessionUi();
|
||||
|
||||
if (this.ctx.session.isCompacting) {
|
||||
this.ctx.session.abortCompaction();
|
||||
@@ -867,14 +863,9 @@ export class CommandController {
|
||||
|
||||
this.ctx.statusLine.invalidate();
|
||||
this.ctx.statusLine.resetActiveTime();
|
||||
this.ctx.ui.requestRender();
|
||||
this.ctx.updateEditorBorderColor();
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.pendingMessagesContainer.clear();
|
||||
this.ctx.compactionQueuedMessages = [];
|
||||
this.ctx.streamingComponent = undefined;
|
||||
this.ctx.streamingMessage = undefined;
|
||||
this.ctx.pendingTools.clear();
|
||||
this.ctx.clearTransientSessionUi();
|
||||
this.ctx.resetTranscript();
|
||||
|
||||
this.ctx.present([new Spacer(1), new Text(`${theme.fg("accent", `${theme.status.success} ${label}`)}`, 1, 1)]);
|
||||
await this.ctx.reloadTodos();
|
||||
@@ -914,7 +905,7 @@ export class CommandController {
|
||||
this.ctx.loadingAnimation.stop();
|
||||
this.ctx.loadingAnimation = undefined;
|
||||
}
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
|
||||
const success = await this.ctx.session.fork();
|
||||
if (!success) {
|
||||
@@ -1177,7 +1168,7 @@ export class CommandController {
|
||||
this.ctx.loadingAnimation.stop();
|
||||
this.ctx.loadingAnimation = undefined;
|
||||
}
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
|
||||
const label = isAuto ? "Auto-compacting context... (esc to cancel)" : "Compacting context... (esc to cancel)";
|
||||
const compactingLoader = new Loader(
|
||||
@@ -1207,7 +1198,7 @@ export class CommandController {
|
||||
await this.ctx.session.compact(instructions, options);
|
||||
|
||||
compactingLoader.stop();
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
this.ctx.rebuildChatFromMessages();
|
||||
|
||||
this.ctx.statusLine.invalidate();
|
||||
@@ -1223,7 +1214,7 @@ export class CommandController {
|
||||
}
|
||||
} finally {
|
||||
compactingLoader.stop();
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
}
|
||||
// Run the caller's pre-flush hook (e.g. the plan-approval model transition)
|
||||
// before queued user input is dispatched, so any turn queued during
|
||||
@@ -1252,7 +1243,7 @@ export class CommandController {
|
||||
this.ctx.loadingAnimation.stop();
|
||||
this.ctx.loadingAnimation = undefined;
|
||||
}
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
|
||||
const handoffLoader = new Loader(
|
||||
this.ctx.ui,
|
||||
@@ -1273,11 +1264,10 @@ export class CommandController {
|
||||
return;
|
||||
}
|
||||
|
||||
// Rebuild chat from the new session (which now contains the handoff document)
|
||||
this.ctx.rebuildChatFromMessages();
|
||||
|
||||
// Rebuild chat from the new session (which now contains the handoff document).
|
||||
this.ctx.clearTransientSessionUi();
|
||||
this.ctx.renderInitialMessages();
|
||||
this.ctx.statusLine.invalidate();
|
||||
this.ctx.ui.requestRender();
|
||||
this.ctx.updateEditorBorderColor();
|
||||
await this.ctx.reloadTodos();
|
||||
|
||||
@@ -1297,9 +1287,9 @@ export class CommandController {
|
||||
}
|
||||
} finally {
|
||||
handoffLoader.stop();
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
}
|
||||
this.ctx.ui.requestRender();
|
||||
this.ctx.ui.requestRender(true, { clearScrollback: true });
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -379,7 +379,7 @@ export class EventController {
|
||||
if (this.ctx.retryLoader) {
|
||||
this.ctx.retryLoader.stop();
|
||||
this.ctx.retryLoader = undefined;
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
}
|
||||
this.#cancelIdleCompaction();
|
||||
this.#cancelIdleRecap();
|
||||
@@ -1083,7 +1083,7 @@ export class EventController {
|
||||
if (this.ctx.loadingAnimation) {
|
||||
this.ctx.loadingAnimation.stop();
|
||||
this.ctx.loadingAnimation = undefined;
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
}
|
||||
if (this.ctx.streamingComponent) {
|
||||
this.ctx.chatContainer.removeChild(this.ctx.streamingComponent);
|
||||
@@ -1126,9 +1126,9 @@ export class EventController {
|
||||
|
||||
/**
|
||||
* Tear down the live "Working…" loader: stop its animation timer AND clear the
|
||||
* reference. A transient overlay (auto-compaction / auto-retry) that only ran
|
||||
* `statusContainer.clear()` detached the loader from the container but left
|
||||
* `ctx.loadingAnimation` set, so the resumed turn's `agent_start` →
|
||||
* reference. A transient overlay (auto-compaction / auto-retry) can remove the
|
||||
* loader from the container while leaving `ctx.loadingAnimation` set, so the
|
||||
* resumed turn's `agent_start` →
|
||||
* `ensureLoadingAnimation()` (guarded by `if (!this.loadingAnimation)`) skipped
|
||||
* re-adding it and the spinner vanished while the agent kept streaming. Nulling
|
||||
* the reference here lets the next `agent_start` recreate and re-attach it.
|
||||
@@ -1169,7 +1169,7 @@ export class EventController {
|
||||
this.#cancelIdleRecap();
|
||||
this.#setTerminalProgress(true);
|
||||
this.#stopWorkingLoader();
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
const reasonText =
|
||||
event.reason === "overflow"
|
||||
? "Context overflow detected, "
|
||||
@@ -1204,7 +1204,7 @@ export class EventController {
|
||||
if (this.ctx.autoCompactionLoader) {
|
||||
this.ctx.autoCompactionLoader.stop();
|
||||
this.ctx.autoCompactionLoader = undefined;
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
}
|
||||
const isHandoffAction = event.action === "handoff";
|
||||
const isShakeAction = event.action === "shake";
|
||||
@@ -1246,12 +1246,12 @@ export class EventController {
|
||||
} else if (event.errorMessage) {
|
||||
this.ctx.showWarning(event.errorMessage);
|
||||
} else if (isHandoffAction) {
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.clearTransientSessionUi();
|
||||
this.ctx.lastAssistantUsage = undefined;
|
||||
this.ctx.rebuildChatFromMessages();
|
||||
this.ctx.renderInitialMessages();
|
||||
this.ctx.statusLine.invalidate();
|
||||
this.ctx.ui.requestRender();
|
||||
await this.ctx.reloadTodos();
|
||||
this.ctx.ui.requestRender(true, { clearScrollback: true });
|
||||
this.ctx.showStatus("Auto-handoff completed");
|
||||
} else if (event.skipped) {
|
||||
// Benign skip: no model selected, no candidate models available, or nothing
|
||||
@@ -1269,7 +1269,7 @@ export class EventController {
|
||||
async #handleAutoRetryStart(event: Extract<AgentSessionEvent, { type: "auto_retry_start" }>): Promise<void> {
|
||||
this.#trackRetrySupersededAssistantComponent(this.#lastAssistantComponent);
|
||||
this.#stopWorkingLoader();
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
if (AIError.is(event.errorId, AIError.Flag.ThinkingLoop)) {
|
||||
// The retry path drops the failed assistant from runtime context. Do not
|
||||
// restore its inline Error row; just unpin the fixed-region banner so the
|
||||
@@ -1293,7 +1293,7 @@ export class EventController {
|
||||
if (this.ctx.retryLoader) {
|
||||
this.ctx.retryLoader.stop();
|
||||
this.ctx.retryLoader = undefined;
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
}
|
||||
if (event.success) {
|
||||
let appliedRecovered = false;
|
||||
|
||||
@@ -162,18 +162,12 @@ export class ExtensionUiController {
|
||||
waitForIdle: () => this.ctx.session.agent.waitForIdle(),
|
||||
reload: async () => {
|
||||
await this.ctx.session.reload();
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
this.ctx.showStatus("Reloaded session");
|
||||
},
|
||||
newSession: async options => {
|
||||
// Stop any loading animation
|
||||
if (this.ctx.loadingAnimation) {
|
||||
this.ctx.loadingAnimation.stop();
|
||||
this.ctx.loadingAnimation = undefined;
|
||||
}
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.clearTransientSessionUi();
|
||||
|
||||
// Create new session
|
||||
this.clearExtensionTerminalInputListeners();
|
||||
@@ -192,15 +186,8 @@ export class ExtensionUiController {
|
||||
// Reset and update status line
|
||||
this.ctx.statusLine.invalidate();
|
||||
this.ctx.statusLine.resetActiveTime();
|
||||
this.ctx.ui.requestRender();
|
||||
|
||||
// Clear UI state
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.pendingMessagesContainer.clear();
|
||||
this.ctx.compactionQueuedMessages = [];
|
||||
this.ctx.streamingComponent = undefined;
|
||||
this.ctx.streamingMessage = undefined;
|
||||
this.ctx.pendingTools.clear();
|
||||
this.ctx.clearTransientSessionUi();
|
||||
this.ctx.resetTranscript();
|
||||
|
||||
this.ctx.present([
|
||||
new Spacer(1),
|
||||
@@ -218,7 +205,6 @@ export class ExtensionUiController {
|
||||
}
|
||||
|
||||
// Update UI
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
this.ctx.editor.setText(result.selectedText);
|
||||
@@ -233,7 +219,6 @@ export class ExtensionUiController {
|
||||
}
|
||||
|
||||
// Update UI
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
if (result.editorText && !this.ctx.editor.getText().trim()) {
|
||||
@@ -251,7 +236,6 @@ export class ExtensionUiController {
|
||||
return { cancelled: true };
|
||||
}
|
||||
setSessionTerminalTitle(this.ctx.sessionManager.getSessionName(), this.ctx.sessionManager.getCwd());
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
return { cancelled: false };
|
||||
@@ -398,18 +382,12 @@ export class ExtensionUiController {
|
||||
waitForIdle: () => this.ctx.session.agent.waitForIdle(),
|
||||
reload: async () => {
|
||||
await this.ctx.session.reload();
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
this.ctx.showStatus("Reloaded session");
|
||||
},
|
||||
newSession: async options => {
|
||||
// Stop any loading animation
|
||||
if (this.ctx.loadingAnimation) {
|
||||
this.ctx.loadingAnimation.stop();
|
||||
this.ctx.loadingAnimation = undefined;
|
||||
}
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.clearTransientSessionUi();
|
||||
|
||||
// Create new session
|
||||
this.clearExtensionTerminalInputListeners();
|
||||
@@ -425,12 +403,8 @@ export class ExtensionUiController {
|
||||
}
|
||||
|
||||
// Clear UI state
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.pendingMessagesContainer.clear();
|
||||
this.ctx.compactionQueuedMessages = [];
|
||||
this.ctx.streamingComponent = undefined;
|
||||
this.ctx.streamingMessage = undefined;
|
||||
this.ctx.pendingTools.clear();
|
||||
this.ctx.clearTransientSessionUi();
|
||||
this.ctx.resetTranscript();
|
||||
|
||||
this.ctx.present([
|
||||
new Spacer(1),
|
||||
@@ -448,7 +422,6 @@ export class ExtensionUiController {
|
||||
}
|
||||
|
||||
// Update UI
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
this.ctx.editor.setText(result.selectedText);
|
||||
@@ -463,7 +436,6 @@ export class ExtensionUiController {
|
||||
}
|
||||
|
||||
// Update UI
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
if (result.editorText && !this.ctx.editor.getText().trim()) {
|
||||
@@ -480,7 +452,6 @@ export class ExtensionUiController {
|
||||
if (!result) {
|
||||
return { cancelled: true };
|
||||
}
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
return { cancelled: false };
|
||||
|
||||
@@ -86,6 +86,20 @@ function hasPasteText(value: unknown): value is PasteTarget {
|
||||
return typeof value === "object" && value !== null && typeof (value as PasteTarget).pasteText === "function";
|
||||
}
|
||||
|
||||
const SHELL_PROMPT_COMMAND_RE =
|
||||
/^(?:\.{0,2}\/|~\/|cd(?:\s|$)|sudo(?:\s|$)|git(?:\s|$)|bun(?:\s|$)|npm(?:\s|$)|pnpm(?:\s|$)|yarn(?:\s|$)|node(?:\s|$)|python\d*(?:\s|$)|cargo(?:\s|$)|go(?:\s|$)|make(?:\s|$)|docker(?:\s|$)|kubectl(?:\s|$))/;
|
||||
const SHELL_PROMPT_OPERATOR_RE = /(?:^|\s)(?:&&|\|\||\||2>&1|[<>]{1,2})(?:\s|$)/;
|
||||
const OMP_STATUS_LINE_RE = /^\s*in:\s+\d+\s+out:\s+\d+(?:\s+cache\s+\S+)?\s+t:\s+\S+\s+tok\/s:\s+\S+/m;
|
||||
|
||||
function looksLikePastedShellPrompt(code: string): boolean {
|
||||
const firstLine = code.split("\n", 1)[0]?.trimStart() ?? "";
|
||||
return (
|
||||
SHELL_PROMPT_COMMAND_RE.test(firstLine) ||
|
||||
SHELL_PROMPT_OPERATOR_RE.test(firstLine) ||
|
||||
OMP_STATUS_LINE_RE.test(code)
|
||||
);
|
||||
}
|
||||
|
||||
function pythonCommandPrefixLength(trimmedText: string): 0 | 1 | 2 {
|
||||
if (trimmedText.charCodeAt(0) !== 36 /* $ */) return 0;
|
||||
if (trimmedText.charCodeAt(1) === 123 /* { */) return 0;
|
||||
@@ -100,8 +114,10 @@ function parsePythonCommandInput(text: string): { code: string; isExcluded: bool
|
||||
const trimmed = text.trimStart();
|
||||
const prefixLength = pythonCommandPrefixLength(trimmed);
|
||||
if (prefixLength === 0) return undefined;
|
||||
const code = trimmed.slice(prefixLength).trim();
|
||||
if (prefixLength === 1 && looksLikePastedShellPrompt(code)) return undefined;
|
||||
return {
|
||||
code: trimmed.slice(prefixLength).trim(),
|
||||
code,
|
||||
isExcluded: prefixLength === 2,
|
||||
};
|
||||
}
|
||||
@@ -536,7 +552,7 @@ export class InputController {
|
||||
const wasPythonMode = this.ctx.isPythonMode;
|
||||
const trimmed = text.trimStart();
|
||||
this.ctx.isBashMode = trimmed.startsWith("!");
|
||||
this.ctx.isPythonMode = pythonCommandPrefixLength(trimmed) > 0;
|
||||
this.ctx.isPythonMode = parsePythonCommandInput(trimmed) !== undefined;
|
||||
if (wasBashMode !== this.ctx.isBashMode || wasPythonMode !== this.ctx.isPythonMode) {
|
||||
this.ctx.updateEditorBorderColor();
|
||||
}
|
||||
|
||||
@@ -88,12 +88,14 @@ function raceAbortSignal<T>(promise: Promise<T>, signal: AbortSignal, createErro
|
||||
const MCP_AUTH_MIN_WRAP_WIDTH = 16;
|
||||
|
||||
/**
|
||||
* Wrap `url` into rows that each fit inside `width`, prefixed by a shared
|
||||
* single-column indent so nested composition doesn't touch column 0. When the
|
||||
* label + URL fit on one line, returns a single row; otherwise puts the label
|
||||
* on its own row and slices the URL into fixed-width chunks. URL chunks are
|
||||
* plain code points — browsers strip whitespace when pasted into the address
|
||||
* bar, so a multi-row selection copies back to the intact URL.
|
||||
* Wrap `url` into rows that each fit inside `width`. When the label + URL fit
|
||||
* on one line, returns a single indented row; otherwise puts the label on its
|
||||
* own indented row and slices the URL into fixed-width chunks that start at
|
||||
* column 0. Continuation chunks carry ZERO leading bytes on purpose: a
|
||||
* multi-row terminal selection includes the newline plus any leading indent,
|
||||
* and while address bars strip newlines they preserve or percent-encode
|
||||
* embedded spaces — an indent would corrupt the URL at every chunk boundary
|
||||
* (silently, when the damage lands inside a query value).
|
||||
*/
|
||||
function wrapUrlRows(label: string, url: string, width: number): string[] {
|
||||
const indent = " ";
|
||||
@@ -103,10 +105,9 @@ function wrapUrlRows(label: string, url: string, width: number): string[] {
|
||||
if (inlineWidth <= effective) {
|
||||
return [`${indent}${theme.fg("muted", `${label} ${sanitized}`)}`];
|
||||
}
|
||||
const chunkWidth = Math.max(1, effective - indent.length);
|
||||
const rows: string[] = [`${indent}${theme.fg("muted", label)}`];
|
||||
for (let i = 0; i < sanitized.length; i += chunkWidth) {
|
||||
rows.push(`${indent}${theme.fg("muted", sanitized.slice(i, i + chunkWidth))}`);
|
||||
for (let i = 0; i < sanitized.length; i += effective) {
|
||||
rows.push(theme.fg("muted", sanitized.slice(i, i + effective)));
|
||||
}
|
||||
return rows;
|
||||
}
|
||||
|
||||
@@ -593,13 +593,27 @@ export class SelectorController {
|
||||
this.ctx.settings,
|
||||
this.ctx.session.modelRegistry,
|
||||
this.ctx.session.scopedModels,
|
||||
async (model, role, thinkingLevel, selector) => {
|
||||
async (model, role, thinkingLevel, selector, action) => {
|
||||
// `auto` is session-global: never baked into a per-role model value
|
||||
// (it can't round-trip through `model:<level>`). Apply it to the session
|
||||
// separately and persist via `defaultThinkingLevel`.
|
||||
const isAuto = thinkingLevel === AUTO_THINKING;
|
||||
const concreteThinking = isAuto ? undefined : thinkingLevel;
|
||||
const selectorValue = selector ?? `${model.provider}/${model.id}`;
|
||||
try {
|
||||
if (action === "retryFallback" && role !== null) {
|
||||
const fallbackSelector = formatModelSelectorValue(selectorValue, concreteThinking);
|
||||
const fallbackChains = this.ctx.settings.get("retry.fallbackChains");
|
||||
const chain = Array.isArray(fallbackChains[role]) ? fallbackChains[role] : [];
|
||||
this.ctx.settings.set("retry.fallbackChains", {
|
||||
...fallbackChains,
|
||||
[role]: [fallbackSelector, ...chain.filter(existing => existing !== fallbackSelector)],
|
||||
});
|
||||
const roleInfo = getRoleInfo(role, settings);
|
||||
const roleLabel = roleInfo?.name ?? role;
|
||||
this.ctx.showStatus(`${roleLabel} fallback model: ${fallbackSelector}`);
|
||||
return;
|
||||
}
|
||||
if (role === null) {
|
||||
// Temporary: update agent state but don't persist the model to settings
|
||||
await this.ctx.session.setModelTemporary(model);
|
||||
@@ -771,7 +785,6 @@ export class SelectorController {
|
||||
return;
|
||||
}
|
||||
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
this.ctx.editor.setText(result.selectedText);
|
||||
done();
|
||||
@@ -915,7 +928,6 @@ export class SelectorController {
|
||||
|
||||
// Update UI — rebuild the display transcript for the new leaf (the
|
||||
// context from navigateTree is the LLM context, not the transcript).
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
if (result.editorText && !this.ctx.editor.getText().trim()) {
|
||||
@@ -927,7 +939,7 @@ export class SelectorController {
|
||||
} finally {
|
||||
if (summaryLoader) {
|
||||
summaryLoader.stop();
|
||||
this.ctx.statusContainer.clear();
|
||||
this.ctx.statusContainer.disposeChildren();
|
||||
}
|
||||
this.ctx.editor.onEscape = originalOnEscape;
|
||||
}
|
||||
@@ -1067,7 +1079,6 @@ export class SelectorController {
|
||||
this.ctx.updateEditorBorderColor();
|
||||
|
||||
// Clear and re-render the chat
|
||||
this.ctx.chatContainer.clear();
|
||||
this.ctx.renderInitialMessages({ clearTerminalHistory: true });
|
||||
await this.ctx.reloadTodos();
|
||||
this.ctx.showStatus(movedProject ? `Resumed session in ${shortenPath(newCwd)}` : "Resumed session");
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
/**
|
||||
* Autocomplete for GitHub issue/PR references typed as `#<number>` (e.g. `#3164`).
|
||||
*
|
||||
* Mirrors the `@` file-reference and `scheme://` internal-url conventions: the
|
||||
* token is rewritten to an internal URL (`pr://3164` or `issue://3164`) plus a
|
||||
* trailing space, and the existing tool-mediated pipeline (the `read` tool →
|
||||
* InternalUrlRouter → `gh`) resolves it from the session cwd's git remote.
|
||||
*
|
||||
* No network at suggestion time — candidates are generated locally. GitHub
|
||||
* shares the issue/PR number space and there is no cheap way to tell which a
|
||||
* given number is while typing, so both a PR and an Issue candidate are offered
|
||||
* by default. Naming the type first (`pr #3164` / `issue #3164`) constrains the
|
||||
* candidates to that kind. Anything that is not a standalone `#<number>` token
|
||||
* keeps falling through to the existing prompt-action menu.
|
||||
*/
|
||||
import type { AutocompleteItem } from "@oh-my-pi/pi-tui";
|
||||
|
||||
/** Candidate kinds, in default display order. */
|
||||
const GITHUB_REF_KINDS = [
|
||||
{ qualifier: "pr", scheme: "pr", label: "PR", description: "GitHub pull request" },
|
||||
{ qualifier: "issue", scheme: "issue", label: "Issue", description: "GitHub issue" },
|
||||
] as const;
|
||||
|
||||
export interface GithubRefContext {
|
||||
/** Text to replace on accept: `#3164`, or `pr #3164` when a qualifier precedes it. */
|
||||
prefix: string;
|
||||
/** Type the user named (`pr`/`pull` → `pr`, `issue` → `issue`), or null to offer both. */
|
||||
qualifier: "pr" | "issue" | null;
|
||||
/** The numeric reference, e.g. `3164`. */
|
||||
number: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* A standalone `#<positive-number>` token ending at the cursor. The `#` must be
|
||||
* preceded by a token boundary (start, whitespace, or an opening quote/paren/`<`/`=`,
|
||||
* matching the internal-URL boundary set) so embedded hashes like `owner/repo#N`,
|
||||
* `foo#N`, `C#12`, or a URL fragment do not match. An optional `pr`/`pull`/`issue`
|
||||
* qualifier word (case-insensitive) immediately before the `#` constrains the kind.
|
||||
*/
|
||||
const GITHUB_REF_TOKEN_RE = /(?:^|[\s"'`(<=])(?:(pr|pull|issue)(\s+))?#([1-9]\d*)$/i;
|
||||
|
||||
export function getGithubRefContext(textBeforeCursor: string): GithubRefContext | null {
|
||||
const match = textBeforeCursor.match(GITHUB_REF_TOKEN_RE);
|
||||
if (!match) return null;
|
||||
const qualifierWord = match[1];
|
||||
const whitespace = match[2] ?? "";
|
||||
const number = match[3] ?? "";
|
||||
return {
|
||||
prefix: qualifierWord ? `${qualifierWord}${whitespace}#${number}` : `#${number}`,
|
||||
qualifier: !qualifierWord ? null : qualifierWord.toLowerCase() === "issue" ? "issue" : "pr",
|
||||
number,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Suggestions for a `#<number>` token. Both kinds are offered unless the user
|
||||
* named a type (`pr #3164` / `issue #3164`), in which case only that kind is
|
||||
* offered. Returns `null` when the text before the cursor is not a standalone
|
||||
* `#<number>` token.
|
||||
*/
|
||||
export function getGithubRefSuggestions(
|
||||
textBeforeCursor: string,
|
||||
): { items: AutocompleteItem[]; prefix: string } | null {
|
||||
const context = getGithubRefContext(textBeforeCursor);
|
||||
if (!context) return null;
|
||||
const kinds = context.qualifier
|
||||
? GITHUB_REF_KINDS.filter(kind => kind.qualifier === context.qualifier)
|
||||
: GITHUB_REF_KINDS;
|
||||
const items: AutocompleteItem[] = kinds.map(kind => ({
|
||||
value: `${kind.scheme}://${context.number}`,
|
||||
label: `${kind.label} #${context.number}`,
|
||||
description: kind.description,
|
||||
}));
|
||||
return { items, prefix: context.prefix };
|
||||
}
|
||||
@@ -579,10 +579,10 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
this.retryLoader.stop();
|
||||
this.retryLoader = undefined;
|
||||
}
|
||||
this.statusContainer.clear();
|
||||
this.pendingMessagesContainer.clear();
|
||||
this.statusContainer.disposeChildren();
|
||||
this.pendingMessagesContainer.disposeChildren();
|
||||
this.#cancelModelCycleClearTimer();
|
||||
this.modelCycleContainer.clear();
|
||||
this.modelCycleContainer.disposeChildren();
|
||||
this.compactionQueuedMessages = [];
|
||||
this.streamingComponent = undefined;
|
||||
this.streamingMessage = undefined;
|
||||
@@ -2602,6 +2602,49 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
}
|
||||
}
|
||||
|
||||
#resolveLocalRoot(): string {
|
||||
return resolveLocalUrlToPath("local://", {
|
||||
getArtifactsDir: () => this.sessionManager.getArtifactsDir(),
|
||||
getSessionId: () => this.sessionManager.getSessionId(),
|
||||
});
|
||||
}
|
||||
|
||||
async #copyLocalArtifactsForFreshSession(sourceRoot: string, destinationRoot: string): Promise<void> {
|
||||
if (sourceRoot === destinationRoot) return;
|
||||
|
||||
let sourceRootStat: { isDirectory(): boolean };
|
||||
try {
|
||||
sourceRootStat = await fs.lstat(sourceRoot);
|
||||
} catch (error) {
|
||||
if (isEnoent(error)) return;
|
||||
throw error;
|
||||
}
|
||||
|
||||
if (!sourceRootStat.isDirectory()) return;
|
||||
|
||||
await fs.mkdir(destinationRoot, { recursive: true });
|
||||
await this.#copyLocalArtifactEntries(sourceRoot, destinationRoot);
|
||||
}
|
||||
|
||||
async #copyLocalArtifactEntries(sourceDir: string, destinationDir: string): Promise<void> {
|
||||
const entries = await fs.readdir(sourceDir, { withFileTypes: true });
|
||||
for (const entry of entries) {
|
||||
const sourcePath = path.join(sourceDir, entry.name);
|
||||
const destinationPath = path.join(destinationDir, entry.name);
|
||||
|
||||
if (entry.isDirectory()) {
|
||||
await fs.mkdir(destinationPath, { recursive: true });
|
||||
await this.#copyLocalArtifactEntries(sourcePath, destinationPath);
|
||||
continue;
|
||||
}
|
||||
|
||||
if (entry.isFile()) {
|
||||
await fs.mkdir(path.dirname(destinationPath), { recursive: true });
|
||||
await fs.copyFile(sourcePath, destinationPath);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async #approvePlan(
|
||||
planContent: string,
|
||||
options: {
|
||||
@@ -2632,14 +2675,16 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
});
|
||||
|
||||
if (!options.preserveContext) {
|
||||
const oldLocalRoot = this.#resolveLocalRoot();
|
||||
await this.handleClearCommand();
|
||||
// The new session has a fresh local:// root — persist the approved plan there
|
||||
// so `local://<slug>-plan.md` resolves correctly in the execution session.
|
||||
const newLocalRoot = this.#resolveLocalRoot();
|
||||
await this.#copyLocalArtifactsForFreshSession(oldLocalRoot, newLocalRoot);
|
||||
const newLocalPath = resolveLocalUrlToPath(options.planFilePath, {
|
||||
getArtifactsDir: () => this.sessionManager.getArtifactsDir(),
|
||||
getSessionId: () => this.sessionManager.getSessionId(),
|
||||
});
|
||||
await Bun.write(newLocalPath, planContent);
|
||||
await fs.mkdir(path.dirname(newLocalPath), { recursive: true });
|
||||
await fs.writeFile(newLocalPath, planContent);
|
||||
} else if (options.compactBeforeExecute) {
|
||||
// Distill the plan-mode transcript before the execution turn is queued so
|
||||
// the plan-approved synthetic prompt lands as a fresh cache anchor.
|
||||
@@ -3626,7 +3671,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
ensureLoadingAnimation(): void {
|
||||
if (!this.loadingAnimation) {
|
||||
this.#clearWorkingMessageAccentCache();
|
||||
this.statusContainer.clear();
|
||||
this.statusContainer.disposeChildren();
|
||||
const messageColorFn = ((message: string) =>
|
||||
renderWorkingMessage(message, this.#getWorkingMessageAccent())) as LoaderMessageColorFn & {
|
||||
animated?: true;
|
||||
@@ -3647,7 +3692,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
);
|
||||
this.statusContainer.addChild(this.loadingAnimation);
|
||||
} else if (!this.statusContainer.children.includes(this.loadingAnimation)) {
|
||||
this.statusContainer.clear();
|
||||
this.statusContainer.disposeChildren();
|
||||
this.statusContainer.addChild(this.loadingAnimation);
|
||||
this.ui.requestRender();
|
||||
}
|
||||
@@ -3660,7 +3705,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
this.loadingAnimation = undefined;
|
||||
this.#clearWorkingMessageAccentCache();
|
||||
if (clearStatusContainer) {
|
||||
this.statusContainer.clear();
|
||||
this.statusContainer.disposeChildren();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4123,7 +4168,6 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
}
|
||||
this.#btwController.dispose();
|
||||
this.#omfgController.dispose();
|
||||
this.chatContainer.clear();
|
||||
this.renderInitialMessages({ clearTerminalHistory: true });
|
||||
this.updateEditorBorderColor();
|
||||
this.showStatus(
|
||||
|
||||
@@ -9,6 +9,7 @@ import {
|
||||
import { formatKeyHints, type KeybindingsManager } from "../config/keybindings";
|
||||
import { isSettingsInitialized, settings } from "../config/settings";
|
||||
import { applyEmojiCompletion, getEmojiSuggestions, isEmojiPrefix, tryEmojiInlineReplace } from "./emoji-autocomplete";
|
||||
import { getGithubRefContext, getGithubRefSuggestions } from "./github-ref-autocomplete";
|
||||
import {
|
||||
applyInternalUrlCompletion,
|
||||
getInternalUrlSuggestions,
|
||||
@@ -94,6 +95,36 @@ function getPromptActionPrefix(textBeforeCursor: string): string | null {
|
||||
return textBeforeCursor.slice(hashIndex);
|
||||
}
|
||||
|
||||
function applyGithubRefCompletion(
|
||||
lines: string[],
|
||||
cursorLine: number,
|
||||
cursorCol: number,
|
||||
item: AutocompleteItem,
|
||||
prefix: string,
|
||||
): { lines: string[]; cursorLine: number; cursorCol: number } | null {
|
||||
if (!getGithubRefContext(prefix)) return null;
|
||||
const scheme: "pr" | "issue" | null = item.value.startsWith("pr://")
|
||||
? "pr"
|
||||
: item.value.startsWith("issue://")
|
||||
? "issue"
|
||||
: null;
|
||||
if (!scheme) return { lines, cursorLine, cursorCol };
|
||||
|
||||
const currentLine = lines[cursorLine] || "";
|
||||
const liveContext = getGithubRefContext(currentLine.slice(0, cursorCol));
|
||||
if (!liveContext || (liveContext.qualifier && liveContext.qualifier !== scheme)) {
|
||||
return { lines, cursorLine, cursorCol };
|
||||
}
|
||||
|
||||
return applyInternalUrlCompletion(
|
||||
lines,
|
||||
cursorLine,
|
||||
cursorCol,
|
||||
{ ...item, value: `${scheme}://${liveContext.number}` },
|
||||
liveContext.prefix,
|
||||
);
|
||||
}
|
||||
|
||||
export class PromptActionAutocompleteProvider implements AutocompleteProvider {
|
||||
#commands: SlashCommand[];
|
||||
#baseProvider: CombinedAutocompleteProvider;
|
||||
@@ -129,6 +160,8 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider {
|
||||
}
|
||||
}
|
||||
|
||||
const githubRefSuggestions = getGithubRefSuggestions(textBeforeCursor);
|
||||
if (githubRefSuggestions) return githubRefSuggestions;
|
||||
const promptActionPrefix = getPromptActionPrefix(textBeforeCursor);
|
||||
if (promptActionPrefix) {
|
||||
const query = promptActionPrefix.slice(1).toLowerCase();
|
||||
@@ -176,6 +209,8 @@ export class PromptActionAutocompleteProvider implements AutocompleteProvider {
|
||||
cursorCol: number;
|
||||
onApplied?: () => void;
|
||||
} {
|
||||
const githubRefCompletion = applyGithubRefCompletion(lines, cursorLine, cursorCol, item, prefix);
|
||||
if (githubRefCompletion) return githubRefCompletion;
|
||||
if (prefix.startsWith("#") && isPromptActionItem(item)) {
|
||||
if (item.actionId === "undo") {
|
||||
return {
|
||||
|
||||
@@ -52,7 +52,8 @@ export function buildHotkeysMarkdown(bindings: HotkeysMarkdownBindings): string
|
||||
`| \`${appKey(bindings, "app.clipboard.pasteImage")}\` | Paste image or text from clipboard |`,
|
||||
"| Hold `Space` | Speech-to-text (push-to-talk): hold to record, release to transcribe |",
|
||||
`| \`${appKey(bindings, "app.agents.hub")}\` / \`${appKey(bindings, "app.session.observe")}\` / double-tap \`←\` (empty editor) | Open the agent hub |`,
|
||||
"| `#` | Open prompt actions |",
|
||||
"| `#<number>` | GitHub issue/PR reference (e.g. `#3164` → `pr://`/`issue://`) |",
|
||||
"| `#` / `#<text>` | Prompt actions (copy / undo / move cursor) |",
|
||||
"| `/` | Slash commands |",
|
||||
"| `!` | Run bash command |",
|
||||
"| `!!` | Run bash command (excluded from context) |",
|
||||
|
||||
@@ -581,7 +581,7 @@ export class UiHelpers {
|
||||
} else {
|
||||
this.ctx.resetTranscript();
|
||||
}
|
||||
this.ctx.pendingMessagesContainer.clear();
|
||||
this.ctx.pendingMessagesContainer.disposeChildren();
|
||||
this.ctx.pendingBashComponents = [];
|
||||
this.ctx.pendingPythonComponents = [];
|
||||
|
||||
@@ -647,7 +647,7 @@ export class UiHelpers {
|
||||
}
|
||||
|
||||
updatePendingMessagesDisplay(): void {
|
||||
this.ctx.pendingMessagesContainer.clear();
|
||||
this.ctx.pendingMessagesContainer.disposeChildren();
|
||||
const queuedMessages = this.ctx.viewSession.getQueuedMessages() as QueuedMessages;
|
||||
|
||||
const steeringMessages: Array<{ message: string; label: string }> = [];
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import workflowNotice from "../prompts/system/workflow-notice.md" with { type: "text" };
|
||||
import { prompt } from "@oh-my-pi/pi-utils";
|
||||
import workflowNoticeTemplate from "../prompts/system/workflow-notice.md" with { type: "text" };
|
||||
import { createGradientHighlighter, type KeywordHighlighter } from "./gradient-highlight";
|
||||
import { keywordInProse } from "./markdown-prose";
|
||||
|
||||
@@ -7,18 +8,23 @@ import { keywordInProse } from "./markdown-prose";
|
||||
*
|
||||
* Typing the standalone word in the input editor paints it with a warm
|
||||
* amber→green gradient ({@link highlightWorkflow}); submitting a message that
|
||||
* mentions it appends a hidden {@link WORKFLOW_NOTICE} that steers the model to
|
||||
* author a deterministic multi-subagent workflow in eval cells (agent/parallel/
|
||||
* pipeline). Matching is whitespace-delimited and case-sensitive (lowercase
|
||||
* only) — "workflowz" triggers, but "workflowzed", "Workflowz", and
|
||||
* "workflowz.ts" never do.
|
||||
* mentions it appends a hidden workflow notice that steers the model to author
|
||||
* a deterministic multi-subagent workflow through the active task schema.
|
||||
* Matching is whitespace-delimited and case-sensitive (lowercase only) —
|
||||
* "workflowz" triggers, but "workflowzed", "Workflowz", and "workflowz.ts"
|
||||
* never do.
|
||||
*/
|
||||
|
||||
// Detection: lowercase keyword flanked by whitespace or a string edge. Non-global so `.test` stays stateless.
|
||||
const WORKFLOW_WORD = /(?<!\S)workflowz(?!\S)/;
|
||||
|
||||
/** Hidden system notice appended after a user message that mentions "workflowz". */
|
||||
export const WORKFLOW_NOTICE: string = workflowNotice.trim();
|
||||
/** WORKFLOW_NOTICE is the default hidden notice for sessions with batched task calls enabled. */
|
||||
export const WORKFLOW_NOTICE: string = renderWorkflowNotice({ taskBatch: true });
|
||||
|
||||
/** renderWorkflowNotice renders the workflow notice for the active task schema. */
|
||||
export function renderWorkflowNotice({ taskBatch }: { taskBatch: boolean }): string {
|
||||
return prompt.render(workflowNoticeTemplate, { taskBatch }).trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether `text` contains the standalone keyword "workflowz"
|
||||
|
||||
@@ -1,7 +1,10 @@
|
||||
<critical>
|
||||
Plan mode is active. You MUST perform READ-ONLY work only:
|
||||
- You NEVER create, edit, or delete files — except the single plan file named below.
|
||||
Plan mode is active. You MUST preserve read-only working-tree and system semantics:
|
||||
- You NEVER create, edit, delete, or rename working-tree files.
|
||||
- You NEVER run state-changing commands (`git commit`, `npm install`, migrations) or make any other system change.
|
||||
- `local://` artifacts are session-local planning artifacts. You MAY create or update them when explicitly requested or needed for the plan.
|
||||
- You NEVER delete or rename `local://` artifacts.
|
||||
- You MUST write the canonical plan to `local://<slug>-plan.md`.
|
||||
|
||||
To leave plan mode and implement: call `resolve` with `action: "apply"`, a `reason`, and `extra: { title: "<slug>" }`, where `<slug>` matches your `local://<slug>-plan.md`. The user then picks an execution option and full write access is restored. `<slug>` may contain only letters, numbers, underscores, and hyphens.
|
||||
|
||||
|
||||
@@ -110,9 +110,8 @@ You MUST use the specialized tool over its shell equivalent:
|
||||
{{#has tools "lsp"}}- Code intelligence → `{{toolRefs.lsp}}`.{{/has}}
|
||||
{{#has tools "grep"}}- Regex search → `{{toolRefs.grep}}`, not `grep`, `rg`, or `awk`.{{/has}}
|
||||
{{#has tools "glob"}}- Globbing → `{{toolRefs.glob}}`, not `ls **/*.ext` or `fd`.{{/has}}
|
||||
{{#has tools "eval"}}- Default for any compute: `{{toolRefs.eval}}` cells. Bash is the EXCEPTION — only single binary calls or short fact-computing pipelines (`wc -l`, `sort | uniq -c`, `diff`, checksums). The moment a command grows a loop, conditional, heredoc, `-e`/`-c` script, `$(…)` nesting, or >2 pipe stages, it's a program → `{{toolRefs.eval}}`. NEVER write multiline or inline-script bash.{{/has}}
|
||||
{{#has tools "bash"}}- `{{toolRefs.bash}}`: real binaries and short fact pipelines only. Commands shadowing the specialized tools above are blocked.{{/has}}
|
||||
{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash.{{#has tools "eval"}} Needs control flow, state, or fights shell quoting → `{{toolRefs.eval}}`.{{/has}} Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}}
|
||||
{{#has tools "bash"}}- Litmus: one external-CLI call or short pipeline returning a count, frequency, set difference, or checksum → bash. Merely moves, pages, or trims bytes a tool can fetch → use the tool.{{/has}}
|
||||
|
||||
{{#has tools "report_tool_issue"}}
|
||||
<critical>
|
||||
|
||||
@@ -1,70 +1,89 @@
|
||||
<system-notice>
|
||||
The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Author the orchestration as Python in the `eval` tool and fan out subagents — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough.
|
||||
The user's message above contains the **workflowz** keyword: drive this task as a deterministic multi-subagent workflow. Use the `task` tool {{#if taskBatch}}for batched fan-out{{else}}once per independent subagent{{/if}} — to be comprehensive (decompose and cover in parallel), to be confident (independent perspectives and adversarial checks before you commit), or to take on scale one context can't hold (audits, migrations, broad sweeps). This overrides any default tendency to do the whole task inline when fanning out would be more thorough.
|
||||
|
||||
<when>
|
||||
Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline FIRST (list the files, scope the diff, find the call sites) to discover the work-list, then fan out over it — you don't need to know the shape before the *task*, only before the *fan-out*. Common shapes, each a well-scoped `eval` call you can chain across turns:
|
||||
- **Understand** — parallel readers over subsystems → structured map
|
||||
- **Design** — judge panel of N independent approaches → scored synthesis
|
||||
- **Review** — split into dimensions → find per dimension → adversarially verify each finding
|
||||
- **Research** — multi-modal sweep → deep-read the hits → synthesize
|
||||
- **Migrate** — discover sites → transform each → verify
|
||||
Worth it when the task benefits from decomposition + parallel coverage, or from independent/adversarial cross-checking before you commit. For a quick lookup or single edit, just do it directly — don't spin up agents. Scout inline first (list the files, scope the diff, find the call sites) to discover the work list, then fan out over it. Common shapes:
|
||||
- **Understand** — parallel readers over subsystems → structured map.
|
||||
- **Design** — independent approaches → scored synthesis.
|
||||
- **Review** — split dimensions → find per dimension → adversarially verify each finding.
|
||||
- **Research** — multi-modal sweep → deep-read the hits → synthesize.
|
||||
- **Migrate** — discover sites → transform each → verify.
|
||||
</when>
|
||||
|
||||
<helpers>
|
||||
State persists across eval calls, so scout in one call and fan out in the next. Every eval call has:
|
||||
<task-contract>
|
||||
{{#if taskBatch}}
|
||||
Call `task` once per independent fan-out batch. Put shared background in `context`, and put each independent work item in `tasks[]`. Do not emulate batching with shell loops or eval helper APIs.
|
||||
|
||||
- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes. Recursion follows `task.maxRecursionDepth` (default 2; `-1` uses eval's hard cap 3): main agent depth = 0, each `agent()` child increments depth by 1, and a spawner may call `agent()` only while its current `taskDepth < effective cap`. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts.
|
||||
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
|
||||
- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`.
|
||||
- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out.
|
||||
- `log(message)` — emit a progress line above the status tree. `phase(title)` — start a phase; the status lines that follow group under it.
|
||||
- `budget` — `budget.total` (output-token ceiling, or `None` when none is set), `budget.spent()` (tokens spent this turn — main loop + eval subagents), `budget.remaining()` (`math.inf` when total is `None`), `budget.hard` (whether it's enforced). A ceiling is set by the user: `+Nk` in their message is advisory (you self-limit via `budget.remaining()`), `+Nk!` (or Goal Mode) is hard — `agent()` refuses to spawn once spent reaches it. Gate loops on `budget.total` first, since it's `None` when the user set no budget.
|
||||
`context` must carry the shared contract:
|
||||
|
||||
Everything runs INLINE and synchronously inside the eval call — no background mode, no resume, no separate progress app. Each eval call is one well-scoped fan-out; chain several across calls and turns for multi-phase work, reading each result before you decide the next phase.
|
||||
</helpers>
|
||||
# Goal
|
||||
What the batch accomplishes.
|
||||
# Constraints
|
||||
Rules, non-goals, permissions, and verification limits.
|
||||
# Contract
|
||||
Shared interfaces, output shape, branch/base assumptions, and coordination rules.
|
||||
|
||||
Each task assignment must be self-contained:
|
||||
|
||||
# Target
|
||||
Exact files, symbols, subsystem, or evidence surface; explicit non-goals.
|
||||
# Change
|
||||
What to inspect or modify, step by step, including APIs and patterns to reuse.
|
||||
# Acceptance
|
||||
Observable result, return packet, and local verification. Subagents skip formatters,
|
||||
linters, and project-wide tests; the parent runs shared proof once.
|
||||
{{else}}
|
||||
Call `task` once per independent subagent. Put the full shared background and the leaf work in that call's `assignment`. Do not pass `context` or `tasks[]`: the flat task schema rejects them when batch calls are disabled.
|
||||
|
||||
Each assignment must be self-contained:
|
||||
|
||||
# Target
|
||||
Exact files, symbols, subsystem, or evidence surface; explicit non-goals.
|
||||
# Change
|
||||
Shared background plus what to inspect or modify, step by step, including APIs and patterns to reuse.
|
||||
# Acceptance
|
||||
Observable result, return packet, and local verification. Subagents skip formatters,
|
||||
linters, and project-wide tests; the parent runs shared proof once.
|
||||
{{/if}}
|
||||
|
||||
<structure>
|
||||
For independent per-item chains (review → verify, fetch → extract → score), wrap the WHOLE chain in one function and run it with `parallel()` — then each item flows through its own steps without waiting on the others:
|
||||
Decompose first, then {{#if taskBatch}}batch the independent leaves{{else}}issue one independent task call per leaf in the same turn{{/if}}:
|
||||
|
||||
DIMENSIONS = [{"key": "bugs", "prompt": "…"}, {"key": "perf", "prompt": "…"}]
|
||||
def review_and_verify(d):
|
||||
found = agent(d["prompt"], label=f"review:{d['key']}", schema=FINDINGS_SCHEMA)
|
||||
return parallel([lambda f=f: {**f, "verdict": agent(
|
||||
f"Refute if you can (default refuted when unsure): {f['title']}",
|
||||
label=f"verify:{f['file']}", schema=VERDICT_SCHEMA)} for f in found["findings"]])
|
||||
phase("Review")
|
||||
results = parallel([lambda d=d: review_and_verify(d) for d in DIMENSIONS])
|
||||
confirmed = [f for group in results for f in group if f["verdict"]["is_real"]]
|
||||
{{#if taskBatch}}
|
||||
task(
|
||||
context: "# Goal\nReview the auth diff...\n# Constraints\nRead-only...\n# Contract\nReturn findings as severity/file/line/fix...",
|
||||
tasks: [
|
||||
{ id: "AuthOwner", role: "Auth Storage Reviewer", assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nTrace credential selection...\n# Acceptance\nReturn confirmed findings only..." },
|
||||
{ id: "PromptOwner", role: "Prompt Contract Reviewer", assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance...\n# Acceptance\nReturn mismatches and exact prompt lines..." },
|
||||
]
|
||||
)
|
||||
{{else}}
|
||||
task(
|
||||
role: "Auth Storage Reviewer",
|
||||
assignment: "# Target\npackages/ai/src/auth-storage.ts\n# Change\nReview the auth diff. Shared contract: read-only; return findings as severity/file/line/fix.\n# Acceptance\nReturn confirmed findings only..."
|
||||
)
|
||||
task(
|
||||
role: "Prompt Contract Reviewer",
|
||||
assignment: "# Target\npackages/coding-agent/src/prompts/**\n# Change\nCheck active-tool guidance. Shared contract: read-only; return mismatches and exact prompt lines.\n# Acceptance\nReturn confirmed findings only..."
|
||||
)
|
||||
{{/if}}
|
||||
|
||||
Reach for `pipeline()` only when a stage genuinely needs ALL of the previous stage first — dedup/merge across the whole set, early-exit on zero, or "compare against the other findings" — because its inter-stage barrier makes every item wait for the slowest peer:
|
||||
|
||||
phase("Find")
|
||||
found = parallel([lambda d=d: agent(d["prompt"], schema=FINDINGS_SCHEMA) for d in DIMENSIONS])
|
||||
findings = dedupe([f for r in found for f in r["findings"]]) # needs everything at once
|
||||
phase("Verify")
|
||||
verdicts = parallel([lambda f=f: agent(verify_prompt(f), schema=VERDICT_SCHEMA) for f in findings])
|
||||
|
||||
Don't add a barrier just to flatten/map/filter — do that with plain Python between calls. Nested `parallel()` pools each cap independently, so keep total fan-out sane.
|
||||
{{#if taskBatch}}Prefer one wide batch over serial subagent calls when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing.{{else}}Prefer issuing all independent task calls in one assistant turn over serial dispatch when work items do not share files. If tasks overlap, name the overlap and have agents coordinate through IRC before editing.{{/if}}
|
||||
</structure>
|
||||
|
||||
<patterns>
|
||||
Compose the harness the task calls for:
|
||||
- **Adversarial verify** — N independent skeptics per finding, each prompted to REFUTE; keep it only if a majority survive. `votes = parallel([lambda i=i: agent(f"Refute: {claim}. refuted=true if unsure.", schema=VERDICT) for i in range(3)])`, then keep when `sum(not v["refuted"] for v in votes) ≥ 2`.
|
||||
- **Perspective-diverse verify** — give each verifier a distinct lens (correctness, security, perf, does-it-reproduce) instead of N identical refuters.
|
||||
- **Judge panel** — N attempts from different angles, scored by parallel judges; synthesize from the winner, graft the best of the rest.
|
||||
- **Loop-until-dry** — for unknown-size discovery, keep spawning finders until K consecutive rounds surface nothing new; dedup against everything SEEN, not just what was confirmed, or it never converges.
|
||||
- **Multi-modal sweep** — parallel finders each searching a different way (by-container, by-content, by-entity, by-time), each blind to the others.
|
||||
- **Completeness critic** — a final agent that asks "what's missing — modality not run, claim unverified, file unread?"; its answer is the next round.
|
||||
- **Budget/count loops** — `while len(bugs) < 10:` to hit a target, or `while budget.total and budget.remaining() > 50_000:` to scale depth to the turn budget; `log()` each round.
|
||||
- **No silent caps** — if you bound coverage (top-N, no-retry, sampling), `log()` what you dropped; silent truncation reads as "covered everything" when it didn't.
|
||||
|
||||
Scale to the ask: "find any bugs" → a few finders, single-vote verify. "thoroughly audit / be comprehensive" → larger finder pool, 3–5-vote adversarial pass, a synthesis stage.
|
||||
- **Adversarial verify** — dispatch skeptical reviewers with distinct targets, then keep only findings the parent can verify against source.
|
||||
- **Perspective-diverse review** — use separate correctness, security, performance, and maintainability roles instead of identical reviewers.
|
||||
- **Completeness critic** — after the first batch, dispatch one read-only critic that asks what modality, file, claim, or proof was missed.
|
||||
- **No silent caps** — if you bound coverage (top-N, no retry, sampling), state what was dropped and why before acting.
|
||||
- **Parent owns closure** — subagents return evidence; the parent reads it, resolves contradictions, runs proof, and makes the final decision.
|
||||
</patterns>
|
||||
|
||||
<execution>
|
||||
- Decompose the surface first; capture it in `todo` when it spans phases.
|
||||
- Prefer `schema=` for any agent whose output you branch on.
|
||||
- After a fan-out returns, YOU own correctness: read the artifacts, run the gate, verify before acting. Subagents do the legwork; they don't get the last word.
|
||||
- Keep going until the task is closed — a returned fan-out is a step, not a stopping point.
|
||||
- Capture multi-phase workflow state in the visible todo system when available.
|
||||
{{#if taskBatch}}- Batch independent subagents in one `task` call.{{else}}- Dispatch independent subagents as separate `task` calls in the same turn.{{/if}}
|
||||
- Give every subagent a narrow target, explicit non-goals, and a concrete return packet.
|
||||
- After fan-out returns, read the artifacts, patch or decide, and run the shared gate.
|
||||
- Keep going until the task is closed — returned fan-out is a step, not a stopping point.
|
||||
</execution>
|
||||
</system-notice>
|
||||
|
||||
@@ -6,14 +6,22 @@ The shell invokes **real binaries** with simple args. It is NOT full GNU Bash.
|
||||
|
||||
Use bash ONLY for: a single binary call, or one short pipeline that COMPUTES a fact and does not depend on shell-specific regex/quoting (`wc -l`, `sort | uniq -c`, `comm`, `diff`, a checksum, `git status`).
|
||||
|
||||
Anything below → `eval` cell, not bash:
|
||||
{{#if hasEval}}Anything below → `eval` cell, not bash:
|
||||
- Inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists for that language
|
||||
- Heredocs (`<<EOF`), `while`/`for`/`if`/`case` shell control flow
|
||||
- `$(…)` command substitution nested inside another command
|
||||
- Pipelines with more than two stages, or stages that need control flow or quote/JSON escaping
|
||||
- Multiline commands, `&&`-chains mixing control flow
|
||||
- Quote/JSON escaping that fights the shell
|
||||
- GNU grep BRE extensions are not guaranteed in the embedded shell: use `grep -E 'json|tool'` for alternation instead of `grep 'json\|tool'`; use the built-in `grep` tool with `pattern: "json|tool"` (Rust regex, so `\bword\b` works there), or `eval` for exact text processing.
|
||||
{{else}}Anything below means you are writing a shell program, not invoking one. Prefer a purpose-built tool, a checked-in script, or a single repo command instead:
|
||||
- Inline interpreter scripts (`-e`/`-c`/`--eval`)
|
||||
- Heredocs (`<<EOF`), `while`/`for`/`if`/`case` shell control flow
|
||||
- `$(…)` command substitution nested inside another command
|
||||
- Pipelines with more than two stages, or stages that need control flow or quote/JSON escaping
|
||||
- Multiline commands, `&&`-chains mixing control flow
|
||||
- Quote/JSON escaping that fights the shell
|
||||
{{/if}}
|
||||
{{#if hasGrep}}- GNU grep BRE extensions are not guaranteed in the embedded shell: use `grep -E 'json|tool'` for alternation instead of `grep 'json\|tool'`; use the built-in `grep` tool with `pattern: "json|tool"` (Rust regex, so `\bword\b` works there){{#if hasEval}}, or `eval` for exact text processing{{/if}}.{{else}}- GNU grep BRE extensions are not guaranteed in the embedded shell: use `grep -E 'json|tool'` for alternation instead of `grep 'json\|tool'`{{#if hasEval}}, or use `eval` for exact text processing{{/if}}.{{/if}}
|
||||
|
||||
<instruction>
|
||||
- `cwd` sets the working dir, not `cd dir && …`
|
||||
@@ -23,14 +31,17 @@ Anything below → `eval` cell, not bash:
|
||||
- `;` only when later commands should run despite earlier failures
|
||||
- Multiple bash calls per message run concurrently. NEVER split order-dependent commands across parallel calls — chain with `&&` in one call.
|
||||
- Internal URIs (`skill://`, `agent://`, …) auto-resolve to FS paths
|
||||
- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly.
|
||||
{{#if hasEval}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Prefer `eval` and process the stream directly.{{else}}- Need exact pipeline semantics (`cmd | head`, multi-stage filtering) or output truncation? Use a checked-in script, purpose-built tool, or single command that owns the output shape.{{/if}}
|
||||
{{#if asyncEnabled}}
|
||||
- `async: true` for long-running commands when you don't need immediate output: returns a background job ID; result delivered as a follow-up.
|
||||
{{/if}}
|
||||
</instruction>
|
||||
|
||||
<critical>
|
||||
- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.
|
||||
{{#if hasEval}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.{{else}}- The embedded shell invokes real binaries with simple args; it is NOT full GNU Bash and NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts, several piped stages, exact pipeline semantics, or quote/JSON escaping mean you're writing a shell program; use a purpose-built tool or checked-in script instead.{{/if}}
|
||||
{{#if hasGrep}}- NEVER shell out to search content or files: `grep/rg` → `grep`.{{else}}- Avoid shelling out for broad content search; use an active search/read tool when one is available.{{/if}}
|
||||
{{#if hasRead}}{{#if hasGlob}}- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing.{{else}}- Prefer `read` for known file and directory reads. Only use shell listing when no file-listing tool is active.{{/if}}{{else}}{{#if hasGlob}}- Prefer `glob` for file discovery; avoid `find` when `glob` is active.{{else}}- If no file read/listing tool is active, keep shell inspection narrow and state that limitation.{{/if}}{{/if}}
|
||||
- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://<id>`.
|
||||
</critical>
|
||||
|
||||
<output>
|
||||
@@ -41,9 +52,9 @@ Anything below → `eval` cell, not bash:
|
||||
{{#if asyncEnabled}}
|
||||
# Timeout and async
|
||||
|
||||
- `timeout` is seconds, clamped to `1..3600`; the process is killed on elapse.
|
||||
- `async: true` defers only reporting — it does NOT extend the timeout; a daemon with `async: true` is still killed at the clamped timeout.
|
||||
- Need >3600s? Detach/manage lifecycle yourself (`cmd &`, supervisor, self-restarting script). The shell session persists across calls.
|
||||
- `timeout` is seconds; nonzero values are clamped to `1..3600` and the process is killed on elapse. Set `timeout: 0` only for commands that must run until completion or explicit cancellation.
|
||||
- `async: true` defers only reporting — it does NOT extend a nonzero timeout; use `timeout: 0` when a daemon or watcher must be cancellation-owned.
|
||||
- Need a daemon or >3600s run? Use `async: true` with `timeout: 0` when the harness should keep it alive until cancellation, or detach/manage lifecycle yourself (`cmd &`, supervisor, self-restarting script). The shell session persists across calls.
|
||||
{{/if}}
|
||||
{{#if autoBackgroundEnabled}}
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@ Greps files using regex.
|
||||
|
||||
<instruction>
|
||||
- Rust regex (RE2-style): alternation is `foo|bar`, not GNU BRE-style `foo\|bar`; Rust word boundaries like `\bword\b` are supported. Use line anchors or post-filters instead of lookaround/backreferences.
|
||||
- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`).
|
||||
- `path`: SHOULD scope to a known path (e.g. `src`); pass several as a delimited list (`src; tests`). Literal colon filename + line range? Use `selector` (e.g. `{"path":"test:1-2","selector":"1-2"}`), not recursive `path:"test:1-2:1-2"`.
|
||||
- Cross-line patterns detected from literal `\n` or `\\n` in `pattern`.
|
||||
</instruction>
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
Read files, directories, archives, SQLite, images, documents, internal resources, and web URLs via one `path`.
|
||||
Read files, directories, archives, SQLite, images, documents, internal resources, and web URLs via `path` plus optional `selector`.
|
||||
|
||||
<instruction>
|
||||
- SHOULD parallelize independent reads.
|
||||
@@ -7,7 +7,8 @@ Read files, directories, archives, SQLite, images, documents, internal resources
|
||||
|
||||
## Parameters
|
||||
|
||||
- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`, `ssh://`), or URL. Append `:<sel>` for ranges/modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`).
|
||||
- `path` — required. Local path, internal URI (`skill://`, `agent://`, `artifact://`, `memory://`, `rule://`, `local://`, `vault://`, `mcp://`, `omp://`, `issue://`, `pr://`, `ssh://`), or URL. Inline `:<sel>` still works for ranges/modes (e.g. `src/foo.ts:50-200`, `src/foo.ts:raw`, `db.sqlite:users:42`).
|
||||
- `selector` — optional selector without leading `:` (e.g. `"50-200"`, `"raw"`, `"raw:50-100"`, `"conflicts"`). Use when `path` contains literal colons: `{"path":"test:1-2","selector":"1-2"}`.
|
||||
|
||||
## Selectors
|
||||
|
||||
@@ -72,6 +73,6 @@ All URI schemes take the same line selectors. `artifact://<id>` recovers spilled
|
||||
`ssh://host/<absolute-path>` reads a remote text file (UTF-8, ≤1 MiB) or lists a directory one level deep, on a pre-configured SSH host or `~/.ssh/config` alias; `ssh://host/` lists the remote root and bare `ssh://` lists the configured hosts. Files are also writable via `write` and searchable via `search`; a directory only lists (`search` refuses a directory, `write` refuses to overwrite one). A literal `:`, `?`, or `#` in the remote path must be percent-encoded (`%3A`/`%3F`/`%23`) — a trailing `:sel` is read as a line selector, and `?`/`#` start a URL query/fragment. Requires a POSIX login shell (`sh`/`bash`/`zsh`); a Windows host or a non-POSIX shell (fish, csh/tcsh) is rejected — use the `ssh` tool there.
|
||||
|
||||
<critical>
|
||||
- Line ranges go in the selector: `path="src/foo.ts:50-200"`.
|
||||
- Literal colon filename + selector? Use `selector`, not recursive `path:"file:sel:sel"`.
|
||||
- Summary footer names elided ranges? Re-issue ONLY those ranges. NEVER guess `..`/`…` content.
|
||||
</critical>
|
||||
|
||||
@@ -1525,10 +1525,19 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
// entries capture it at fetch time and are dropped at injection if a newer
|
||||
// mutation (any tool) bumped it in the meantime.
|
||||
const fileMutationVersions = new Map<string, number>();
|
||||
const activeToolNames = new Set<string>();
|
||||
const setActiveToolNames = (names: Iterable<string>): void => {
|
||||
activeToolNames.clear();
|
||||
for (const name of names) {
|
||||
activeToolNames.add(name);
|
||||
}
|
||||
};
|
||||
const toolSession: ToolSession = {
|
||||
get cwd() {
|
||||
return sessionManager.getCwd();
|
||||
},
|
||||
isToolActive: name => activeToolNames.has(name),
|
||||
setActiveToolNames,
|
||||
hasUI: options.hasUI ?? false,
|
||||
enableLsp,
|
||||
get hasEditTool() {
|
||||
@@ -2558,6 +2567,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
});
|
||||
hasRegistered = true;
|
||||
|
||||
setActiveToolNames(initialToolNames);
|
||||
const { systemPrompt } = await logger.time(
|
||||
"buildSystemPrompt",
|
||||
rebuildSystemPrompt,
|
||||
@@ -2852,6 +2862,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
rebuildSystemPrompt,
|
||||
reloadSshTool,
|
||||
requestedToolNames: requestedToolNameSet,
|
||||
setActiveToolNames,
|
||||
getMcpServerInstructions: mcpManager
|
||||
? () => {
|
||||
const raw = mcpManager.getServerInstructions();
|
||||
|
||||
@@ -35,6 +35,7 @@ import {
|
||||
type AsideMessage,
|
||||
type CompactionSummaryMessage,
|
||||
countTokens,
|
||||
createToolScopedAbortReason,
|
||||
resolveTelemetry,
|
||||
type StreamFn,
|
||||
ThinkingLevel,
|
||||
@@ -108,6 +109,7 @@ import {
|
||||
clearAnthropicFastModeFallback,
|
||||
deriveClaudeDeviceId,
|
||||
Effort,
|
||||
isUsageLimitOutcome,
|
||||
parseRateLimitReason,
|
||||
realizesPriorityServiceTier,
|
||||
resolveModelServiceTier,
|
||||
@@ -124,6 +126,7 @@ import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
|
||||
import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives";
|
||||
import {
|
||||
escapeXmlText,
|
||||
extractHttpStatusFromError,
|
||||
extractRetryHint,
|
||||
formatDuration,
|
||||
getAgentDbPath,
|
||||
@@ -179,7 +182,12 @@ import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles";
|
||||
import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates";
|
||||
import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier";
|
||||
import type { Settings, SkillsSettings } from "../config/settings";
|
||||
import { getDefault, onAppendOnlyModeChanged, validateProviderMaxInFlightRequests } from "../config/settings";
|
||||
import {
|
||||
getDefault,
|
||||
onAppendOnlyModeChanged,
|
||||
onModelRolesChanged,
|
||||
validateProviderMaxInFlightRequests,
|
||||
} from "../config/settings";
|
||||
import { RawSseDebugBuffer } from "../debug/raw-sse-buffer";
|
||||
import { loadCapability } from "../discovery";
|
||||
import { expandApplyPatchToEntries, normalizeDiff, normalizeToLF, ParseError, previewPatch, stripBom } from "../edit";
|
||||
@@ -237,7 +245,7 @@ import { theme } from "../modes/theme/theme";
|
||||
import { parseTurnBudget } from "../modes/turn-budget";
|
||||
import { containsUltrathink, ULTRATHINK_NOTICE } from "../modes/ultrathink";
|
||||
import { computeNonMessageBreakdown, computeNonMessageTokens } from "../modes/utils/context-usage";
|
||||
import { containsWorkflow, WORKFLOW_NOTICE } from "../modes/workflow";
|
||||
import { containsWorkflow, renderWorkflowNotice } from "../modes/workflow";
|
||||
import { createPlanReadMatcher } from "../plan-mode/plan-protection";
|
||||
import type { PlanModeState } from "../plan-mode/state";
|
||||
import advisorSystemPrompt from "../prompts/advisor/system.md" with { type: "text" };
|
||||
@@ -314,6 +322,7 @@ import { resolveFileDisplayMode } from "../utils/file-display-mode";
|
||||
import { extractFileMentions, generateFileMentionMessages } from "../utils/file-mentions";
|
||||
import { normalizeModelContextImages } from "../utils/image-loading";
|
||||
import { describeAttachedImagesForTextModel } from "../utils/image-vision-fallback";
|
||||
import { formatLocalCalendarDate } from "../utils/local-date";
|
||||
import { generateSessionTitle } from "../utils/title-generator";
|
||||
import { buildNamedToolChoice, isToolChoiceActive } from "../utils/tool-choice";
|
||||
import type { AuthStorage } from "./auth-storage";
|
||||
@@ -686,6 +695,8 @@ export interface AgentSessionConfig {
|
||||
toolRegistry?: Map<string, AgentTool>;
|
||||
/** Tool names whose current registry entry is still the built-in implementation. */
|
||||
builtInToolNames?: Iterable<string>;
|
||||
/** Update tool-session predicates that render guidance from the live active tool set. */
|
||||
setActiveToolNames?: (names: Iterable<string>) => void;
|
||||
/** Current session pre-LLM message transform pipeline */
|
||||
transformContext?: (messages: AgentMessage[], signal?: AbortSignal) => AgentMessage[] | Promise<AgentMessage[]>;
|
||||
/**
|
||||
@@ -724,6 +735,8 @@ export interface AgentSessionConfig {
|
||||
convertToLlm?: (messages: AgentMessage[]) => Message[] | Promise<Message[]>;
|
||||
/** System prompt builder that can consider tool availability. Returns ordered provider-facing blocks. */
|
||||
rebuildSystemPrompt?: (toolNames: string[], tools: Map<string, AgentTool>) => Promise<{ systemPrompt: string[] }>;
|
||||
/** Local calendar date provider used by prompt-cache invalidation. Defaults to the host local date. */
|
||||
getLocalCalendarDate?: () => string;
|
||||
/** Rebuild the SSH tool from current capability discovery results. */
|
||||
reloadSshTool?: () => Promise<AgentTool | null>;
|
||||
requestedToolNames?: ReadonlySet<string>;
|
||||
@@ -870,6 +883,7 @@ export interface HandoffResult {
|
||||
export interface SessionHandoffOptions {
|
||||
autoTriggered?: boolean;
|
||||
signal?: AbortSignal;
|
||||
onSwitchCancelled?: () => void;
|
||||
}
|
||||
|
||||
/** Result from cycleModel() */
|
||||
@@ -1555,6 +1569,7 @@ export class AgentSession {
|
||||
#cancelExitRecorder?: () => void;
|
||||
#exitRecorded = false;
|
||||
#unsubscribeAppendOnly?: () => void;
|
||||
#unsubscribeModelRoles?: () => void;
|
||||
/** Last (enable, providerId) tuple resolved by `#syncAppendOnlyContext` — used to skip no-op invalidations. */
|
||||
#lastAppendOnlyResolution?: { enable: boolean; providerId: string | undefined };
|
||||
#eventListeners: AgentSessionEventListener[] = [];
|
||||
@@ -1716,8 +1731,10 @@ export class AgentSession {
|
||||
#rebuildSystemPrompt:
|
||||
| ((toolNames: string[], tools: Map<string, AgentTool>) => Promise<{ systemPrompt: string[] }>)
|
||||
| undefined;
|
||||
#getLocalCalendarDate: () => string;
|
||||
#getMcpServerInstructions: (() => Map<string, string> | undefined) | undefined;
|
||||
#reloadSshTool: (() => Promise<AgentTool | null>) | undefined;
|
||||
#setActiveToolNames: ((names: Iterable<string>) => void) | undefined;
|
||||
#disconnectOwnedMcpManager: (() => Promise<void>) | undefined;
|
||||
#requestedToolNames: ReadonlySet<string> | undefined;
|
||||
#baseSystemPrompt: string[];
|
||||
@@ -2161,8 +2178,10 @@ export class AgentSession {
|
||||
});
|
||||
this.#convertToLlm = config.convertToLlm ?? convertToLlm;
|
||||
this.#rebuildSystemPrompt = config.rebuildSystemPrompt;
|
||||
this.#getLocalCalendarDate = config.getLocalCalendarDate ?? formatLocalCalendarDate;
|
||||
this.#getMcpServerInstructions = config.getMcpServerInstructions;
|
||||
this.#reloadSshTool = config.reloadSshTool;
|
||||
this.#setActiveToolNames = config.setActiveToolNames;
|
||||
this.#disconnectOwnedMcpManager = config.disconnectOwnedMcpManager;
|
||||
this.#baseSystemPrompt = this.agent.state.systemPrompt;
|
||||
this.#promptModelKey = this.#currentPromptModelKey();
|
||||
@@ -2259,6 +2278,11 @@ export class AgentSession {
|
||||
this.#unsubscribeAgent = this.agent.subscribe(this.#handleAgentEvent);
|
||||
// Re-evaluate append-only context mode when the setting changes at runtime.
|
||||
this.#unsubscribeAppendOnly = onAppendOnlyModeChanged(_value => this.#syncAppendOnlyContext(this.model));
|
||||
this.#unsubscribeModelRoles = onModelRolesChanged(() => {
|
||||
if (!this.#advisorEnabled || this.#isDisposed) return;
|
||||
if (this.#advisors.length > 0 && !this.#advisorRuntimeMatchesCurrentConfig()) this.#stopAdvisorRuntime();
|
||||
this.#buildAdvisorRuntime(true);
|
||||
});
|
||||
}
|
||||
// -------------------------------------------------------------------------
|
||||
// Advisor runtime lifecycle
|
||||
@@ -2391,7 +2415,9 @@ export class AgentSession {
|
||||
#advisorRuntimeSignature(config: AdvisorConfig, slug: string, model: Model, thinkingLevel: ThinkingLevel): string {
|
||||
const tools = config.tools?.length ? config.tools.join("\u001e") : "";
|
||||
const instructions = config.instructions?.trim() ?? "";
|
||||
return [config.name, slug, model.provider, model.id, thinkingLevel, tools, instructions].join("\u001f");
|
||||
return [config.name, slug, formatModelStringWithRouting(model), thinkingLevel, tools, instructions].join(
|
||||
"\u001f",
|
||||
);
|
||||
}
|
||||
|
||||
#advisorRuntimeMatchesCurrentConfig(): boolean {
|
||||
@@ -2539,6 +2565,21 @@ export class AgentSession {
|
||||
maintainContext: incomingTokens => this.#maintainAdvisorContext(advisorRef, incomingTokens),
|
||||
obfuscator: this.#obfuscator,
|
||||
beginAdvisorUpdate: () => advisorRef.emissionGuard.beginUpdate(),
|
||||
onTurnError: async error => {
|
||||
// Mirror the auth-gateway's usage-limit remedy: the in-stream a/b/c
|
||||
// auth retry rotates through siblings within one request but never
|
||||
// blocks the LAST failing credential, so without this the advisor
|
||||
// re-picks the same exhausted account every retry. Usage limits
|
||||
// only — other failures keep the plain retry/notify path (never
|
||||
// suspect-mark a credential on a transient advisor error).
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
if (!isUsageLimitOutcome(extractHttpStatusFromError(error), message)) return;
|
||||
await this.#modelRegistry.authStorage.markUsageLimitReached(advisorModel.provider, advisorSessionId, {
|
||||
retryAfterMs: extractRetryHint(undefined, message),
|
||||
baseUrl: advisorModel.baseUrl,
|
||||
modelId: advisorModel.id,
|
||||
});
|
||||
},
|
||||
notifyFailure: error => {
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
this.emitNotice(
|
||||
@@ -4666,7 +4707,8 @@ export class AgentSession {
|
||||
// Decide first: a non-interrupting tool-source match attaches to the
|
||||
// specific tool call's result instead of driving a loop-wide follow-up.
|
||||
const shouldInterrupt = this.#shouldInterruptForTtsrMatch(matches, matchContext);
|
||||
const perToolId = shouldInterrupt ? undefined : this.#extractTtsrToolCallId(matchContext);
|
||||
const matchedToolId = this.#extractTtsrToolCallId(matchContext);
|
||||
const perToolId = shouldInterrupt ? undefined : matchedToolId;
|
||||
if (perToolId) {
|
||||
this.#addPerToolTtsrInjections(perToolId, matches);
|
||||
this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {});
|
||||
@@ -4682,7 +4724,16 @@ export class AgentSession {
|
||||
// Abort the stream immediately — do not gate on extension callbacks
|
||||
this.#ttsrAbortPending = true;
|
||||
this.#ensureTtsrResumePromise();
|
||||
this.agent.abort(this.#formatTtsrAbortReason(matches));
|
||||
const abortReason = this.#formatTtsrAbortReason(matches);
|
||||
this.agent.abort(
|
||||
matchedToolId
|
||||
? createToolScopedAbortReason(
|
||||
abortReason,
|
||||
{ [matchedToolId]: abortReason },
|
||||
"TTSR interrupt on another tool call",
|
||||
)
|
||||
: abortReason,
|
||||
);
|
||||
// Notify extensions (fire-and-forget, does not block abort)
|
||||
this.#emitSessionEvent({ type: "ttsr_triggered", rules: matches }).catch(() => {});
|
||||
// Schedule retry after a short delay
|
||||
@@ -5755,6 +5806,10 @@ export class AgentSession {
|
||||
this.#unsubscribeAppendOnly();
|
||||
this.#unsubscribeAppendOnly = undefined;
|
||||
}
|
||||
if (this.#unsubscribeModelRoles) {
|
||||
this.#unsubscribeModelRoles();
|
||||
this.#unsubscribeModelRoles = undefined;
|
||||
}
|
||||
this.#eventListeners = [];
|
||||
}
|
||||
|
||||
@@ -6298,6 +6353,7 @@ export class AgentSession {
|
||||
),
|
||||
);
|
||||
}
|
||||
this.#setActiveToolNames?.(validToolNames);
|
||||
const activeNameSet = new Set(validToolNames);
|
||||
for (const name of Array.from(this.#selectedDiscoveredToolNames)) {
|
||||
if (!activeNameSet.has(name) || isMCPToolName(name) || !this.#toolRegistry.has(name)) {
|
||||
@@ -6403,6 +6459,7 @@ export class AgentSession {
|
||||
async refreshBaseSystemPrompt(): Promise<void> {
|
||||
if (!this.#rebuildSystemPrompt) return;
|
||||
const activeToolNames = this.getActiveToolNames();
|
||||
this.#setActiveToolNames?.(activeToolNames);
|
||||
const built = await this.#rebuildSystemPrompt(activeToolNames, this.#toolRegistry);
|
||||
this.#baseSystemPrompt = built.systemPrompt;
|
||||
this.#baseSystemPromptBeforeMemoryPromotion = undefined;
|
||||
@@ -6525,7 +6582,7 @@ export class AgentSession {
|
||||
entries.sort();
|
||||
instructionsSegment = entries.join("\u0006");
|
||||
}
|
||||
const date = new Date().toISOString().slice(0, 10);
|
||||
const date = this.#getLocalCalendarDate();
|
||||
return `${nameSegment}\u0003${descriptionSegment}\u0005${registrySegment}\u0007${instructionsSegment}|${date}`;
|
||||
}
|
||||
|
||||
@@ -7342,11 +7399,15 @@ export class AgentSession {
|
||||
timestamp,
|
||||
});
|
||||
}
|
||||
if (this.#magicKeywordEnabled("workflow") && containsWorkflow(text)) {
|
||||
if (
|
||||
this.#magicKeywordEnabled("workflow") &&
|
||||
containsWorkflow(text) &&
|
||||
this.getActiveToolNames().includes("task")
|
||||
) {
|
||||
keywordNotices.push({
|
||||
role: "custom",
|
||||
customType: "workflow-notice",
|
||||
content: WORKFLOW_NOTICE,
|
||||
content: renderWorkflowNotice({ taskBatch: this.settings.get("task.batch") }),
|
||||
display: false,
|
||||
attribution: "user",
|
||||
timestamp,
|
||||
@@ -8787,6 +8848,7 @@ export class AgentSession {
|
||||
|
||||
const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model);
|
||||
|
||||
this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel));
|
||||
this.#clearActiveRetryFallback();
|
||||
this.#setModelWithProviderSessionReset(targetModel);
|
||||
this.sessionManager.appendModelChange(`${targetModel.provider}/${targetModel.id}`, role);
|
||||
@@ -8824,6 +8886,7 @@ export class AgentSession {
|
||||
|
||||
const targetModel = await this.#modelRegistry.refreshSelectedModelMetadata(model);
|
||||
|
||||
this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(targetModel));
|
||||
this.#clearActiveRetryFallback();
|
||||
this.#setModelWithProviderSessionReset(targetModel);
|
||||
this.sessionManager.appendModelChange(
|
||||
@@ -8983,6 +9046,7 @@ export class AgentSession {
|
||||
const next = scopedModels[nextIndex];
|
||||
|
||||
// Apply model
|
||||
this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(next.model));
|
||||
this.#clearActiveRetryFallback();
|
||||
this.#setModelWithProviderSessionReset(next.model);
|
||||
this.sessionManager.appendModelChange(`${next.model.provider}/${next.model.id}`);
|
||||
@@ -9013,6 +9077,7 @@ export class AgentSession {
|
||||
throw new Error(`No API key for ${nextModel.provider}/${nextModel.id}`);
|
||||
}
|
||||
|
||||
this.#modelRegistry.clearSuppressedSelector(formatModelStringWithRouting(nextModel));
|
||||
this.#clearActiveRetryFallback();
|
||||
this.#setModelWithProviderSessionReset(nextModel);
|
||||
this.sessionManager.appendModelChange(`${nextModel.provider}/${nextModel.id}`);
|
||||
@@ -10041,6 +10106,17 @@ export class AgentSession {
|
||||
|
||||
// Start a new session
|
||||
const previousSessionFile = this.sessionFile;
|
||||
if (this.#extensionRunner?.hasHandlers("session_before_switch")) {
|
||||
const result = (await this.#extensionRunner.emit({
|
||||
type: "session_before_switch",
|
||||
reason: "handoff",
|
||||
})) as SessionBeforeSwitchResult | undefined;
|
||||
|
||||
if (result?.cancel) {
|
||||
options?.onSwitchCancelled?.();
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
await this.sessionManager.flush();
|
||||
this.#cancelOwnAsyncJobs();
|
||||
await this.sessionManager.newSession(previousSessionFile ? { parentSession: previousSessionFile } : undefined);
|
||||
@@ -10097,6 +10173,13 @@ export class AgentSession {
|
||||
this.agent.replaceMessages(sessionContext.messages);
|
||||
this.#resetAllAdvisorRuntimes();
|
||||
this.#syncTodoPhasesFromBranch();
|
||||
if (this.#extensionRunner) {
|
||||
await this.#extensionRunner.emit({
|
||||
type: "session_switch",
|
||||
reason: "handoff",
|
||||
previousSessionFile,
|
||||
});
|
||||
}
|
||||
|
||||
return { document: handoffText, savedPath };
|
||||
} catch (error) {
|
||||
@@ -12313,13 +12396,17 @@ export class AgentSession {
|
||||
// queue, not the core steering queue (which handoff's agent.reset() would wipe).
|
||||
await this.#emitSessionEvent({ type: "auto_compaction_start", reason, action });
|
||||
if (action === "handoff") {
|
||||
let handoffSwitchCancelled = false;
|
||||
const handoffFocus = AUTO_HANDOFF_THRESHOLD_FOCUS;
|
||||
const handoffResult = await this.handoff(handoffFocus, {
|
||||
autoTriggered: true,
|
||||
signal: autoCompactionSignal,
|
||||
onSwitchCancelled: () => {
|
||||
handoffSwitchCancelled = true;
|
||||
},
|
||||
});
|
||||
if (!handoffResult) {
|
||||
const aborted = autoCompactionSignal.aborted;
|
||||
const aborted = autoCompactionSignal.aborted || handoffSwitchCancelled;
|
||||
if (aborted) {
|
||||
await this.#emitSessionEvent({
|
||||
type: "auto_compaction_end",
|
||||
@@ -13220,6 +13307,7 @@ export class AgentSession {
|
||||
#resolveRetryFallbackRole(currentSelector: string): string | undefined {
|
||||
const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry);
|
||||
if (!parsedCurrent) return undefined;
|
||||
const chains = this.#getRetryFallbackChains();
|
||||
const currentBaseSelector = formatRetryFallbackBaseSelector(parsedCurrent);
|
||||
const currentPlainSelector = this.model
|
||||
? formatModelSelectorValue(formatModelString(this.model), parsedCurrent.thinkingLevel)
|
||||
@@ -13229,11 +13317,11 @@ export class AgentSession {
|
||||
? formatRetryFallbackBaseSelector(parseRetryFallbackSelector(currentPlainSelector) ?? parsedCurrent)
|
||||
: undefined;
|
||||
|
||||
for (const role of Object.keys(this.#getRetryFallbackChains())) {
|
||||
for (const role of Object.keys(chains)) {
|
||||
const primarySelector = this.#getRetryFallbackPrimarySelector(role);
|
||||
if (primarySelector?.raw === currentSelector) return role;
|
||||
}
|
||||
for (const role of Object.keys(this.#getRetryFallbackChains())) {
|
||||
for (const role of Object.keys(chains)) {
|
||||
const primarySelector = this.#getRetryFallbackPrimarySelector(role);
|
||||
if (!primarySelector) continue;
|
||||
if (currentPlainSelector && primarySelector.raw === currentPlainSelector) return role;
|
||||
@@ -13241,6 +13329,14 @@ export class AgentSession {
|
||||
if (primaryBaseSelector === currentBaseSelector) return role;
|
||||
if (currentPlainBaseSelector && primaryBaseSelector === currentPlainBaseSelector) return role;
|
||||
}
|
||||
const defaultChain = chains.default;
|
||||
if (
|
||||
Array.isArray(defaultChain) &&
|
||||
defaultChain.length > 0 &&
|
||||
this.#getRetryFallbackPrimarySelector("default") === undefined
|
||||
) {
|
||||
return "default";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
@@ -13259,9 +13355,27 @@ export class AgentSession {
|
||||
}
|
||||
|
||||
#findRetryFallbackCandidates(role: string, currentSelector: string): RetryFallbackSelector[] {
|
||||
const chain = this.#getRetryFallbackEffectiveChain(role);
|
||||
if (chain.length <= 1) return [];
|
||||
let chain = this.#getRetryFallbackEffectiveChain(role);
|
||||
const parsedCurrent = parseRetryFallbackSelector(currentSelector, this.#modelRegistry);
|
||||
if (chain.length === 0 && role === "default" && parsedCurrent) {
|
||||
const chains = this.#getRetryFallbackChains();
|
||||
const defaultChain = chains.default;
|
||||
if (
|
||||
Array.isArray(defaultChain) &&
|
||||
defaultChain.length > 0 &&
|
||||
this.#getRetryFallbackPrimarySelector("default") === undefined
|
||||
) {
|
||||
const seen = new Set<string>([parsedCurrent.raw]);
|
||||
chain = [parsedCurrent];
|
||||
for (const selector of defaultChain) {
|
||||
const parsed = parseRetryFallbackSelector(selector, this.#modelRegistry);
|
||||
if (!parsed || seen.has(parsed.raw)) continue;
|
||||
seen.add(parsed.raw);
|
||||
chain.push(parsed);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (chain.length <= 1) return [];
|
||||
const currentBaseSelector = parsedCurrent ? formatRetryFallbackBaseSelector(parsedCurrent) : undefined;
|
||||
const currentPlainSelector =
|
||||
this.model && parsedCurrent
|
||||
@@ -15535,7 +15649,7 @@ export class AgentSession {
|
||||
lastAttemptAtByAccount: coordinator.lastAttemptAtByAccount,
|
||||
});
|
||||
if (!decision.redeem) {
|
||||
logger.debug("codex-auto-reset: skipped", { reason: decision.reason });
|
||||
logger.debug("codex-auto-reset: skipped", { reason: decision.reason, account: accountKey });
|
||||
return false;
|
||||
}
|
||||
if (shouldPromptCodexAutoRedeem(cfg.autoRedeem) && !(await this.#confirmCodexAutoRedeem(decision))) {
|
||||
|
||||
@@ -24,6 +24,7 @@ import projectPromptTemplate from "./prompts/system/project-prompt.md" with { ty
|
||||
import systemPromptTemplate from "./prompts/system/system-prompt.md" with { type: "text" };
|
||||
import { shortenPath } from "./tools/render-utils";
|
||||
import { type ActiveRepoContext, resolveActiveRepoContext } from "./utils/active-repo-context";
|
||||
import { formatLocalCalendarDate } from "./utils/local-date";
|
||||
import { normalizePromptPath } from "./utils/prompt-path";
|
||||
import { AGENTS_MD_LIMIT, buildWorkspaceTree, type WorkspaceTree } from "./workspace-tree";
|
||||
|
||||
@@ -400,7 +401,7 @@ export async function loadSystemPromptFiles(options: LoadContextFilesOptions = {
|
||||
return userLevel?.content ?? null;
|
||||
}
|
||||
|
||||
export const DEFAULT_SYSTEM_PROMPT_TOOL_NAMES = ["read", "bash", "eval", "edit", "write"] as const;
|
||||
export const DEFAULT_SYSTEM_PROMPT_TOOL_NAMES = ["read", "bash", "edit", "write"] as const;
|
||||
|
||||
export interface SystemPromptToolMetadata {
|
||||
label: string;
|
||||
@@ -693,7 +694,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}):
|
||||
}
|
||||
}
|
||||
|
||||
const date = new Date().toISOString().slice(0, 10);
|
||||
const date = formatLocalCalendarDate();
|
||||
const dateTime = date;
|
||||
const promptCwd = shortenPath(normalizePromptPath(resolvedCwd));
|
||||
const activeRepoContextPrompt = renderActiveRepoContextPrompt(activeRepoContext);
|
||||
|
||||
@@ -300,7 +300,7 @@ export async function runInteractiveBashPty(
|
||||
options: {
|
||||
command: string;
|
||||
cwd: string;
|
||||
timeoutMs: number;
|
||||
timeoutMs?: number;
|
||||
signal?: AbortSignal;
|
||||
env?: Record<string, string>;
|
||||
artifactPath?: string;
|
||||
|
||||
@@ -140,6 +140,30 @@ function unquoteToken(token: string): string {
|
||||
return token;
|
||||
}
|
||||
|
||||
function isInsideShellQuote(command: string, index: number): boolean {
|
||||
let quote: "'" | '"' | undefined;
|
||||
for (let i = 0; i < index; i++) {
|
||||
const char = command[i];
|
||||
if (char === "\\" && quote !== "'") {
|
||||
i++;
|
||||
continue;
|
||||
}
|
||||
if (char === "'" && quote !== '"') {
|
||||
quote = quote === "'" ? undefined : "'";
|
||||
continue;
|
||||
}
|
||||
if (char === '"' && quote !== "'") {
|
||||
quote = quote === '"' ? undefined : '"';
|
||||
}
|
||||
}
|
||||
return quote !== undefined;
|
||||
}
|
||||
|
||||
function isEmbeddedInQuotedText(command: string, token: string, index: number): boolean {
|
||||
if (token.startsWith("'") || token.startsWith('"')) return false;
|
||||
return isInsideShellQuote(command, index);
|
||||
}
|
||||
|
||||
/** Shell-escape a path using single quotes. */
|
||||
function shellEscape(p: string): string {
|
||||
return `'${p.replace(/'/g, "'\\''")}'`;
|
||||
@@ -216,6 +240,7 @@ export function expandSkillUrls(command: string, skills: readonly Skill[]): stri
|
||||
|
||||
/**
|
||||
* Expand supported internal URLs in a bash command string to shell-escaped absolute paths.
|
||||
* Unresolvable URLs and literal mentions inside larger quoted text are left unchanged.
|
||||
* Supported schemes: skill://, agent://, artifact://, memory://, rule://, local://
|
||||
*/
|
||||
export async function expandInternalUrls(command: string, options: InternalUrlExpansionOptions): Promise<string> {
|
||||
@@ -231,15 +256,22 @@ export async function expandInternalUrls(command: string, options: InternalUrlEx
|
||||
const index = match.index;
|
||||
if (index === undefined) continue;
|
||||
|
||||
if (isEmbeddedInQuotedText(command, token, index)) continue;
|
||||
|
||||
const rawUrl = unquoteToken(token);
|
||||
const url = normalizeLocalScheme(rawUrl);
|
||||
const resolvedPath = await resolveInternalUrlToPath(
|
||||
url,
|
||||
options.skills,
|
||||
options.internalRouter,
|
||||
options.localOptions,
|
||||
options.ensureLocalParentDirs,
|
||||
);
|
||||
let resolvedPath: string;
|
||||
try {
|
||||
resolvedPath = await resolveInternalUrlToPath(
|
||||
url,
|
||||
options.skills,
|
||||
options.internalRouter,
|
||||
options.localOptions,
|
||||
options.ensureLocalParentDirs,
|
||||
);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
const replacement = options.noEscape ? resolvedPath : shellEscape(resolvedPath);
|
||||
expanded = `${expanded.slice(0, index)}${replacement}${expanded.slice(index + token.length)}`;
|
||||
}
|
||||
|
||||
@@ -27,6 +27,7 @@ import { type BashInteractiveResult, runInteractiveBashPty } from "./bash-intera
|
||||
import { checkBashInterception } from "./bash-interceptor";
|
||||
import { canUseInteractiveBashPty } from "./bash-pty-selection";
|
||||
import { expandInternalUrls, type InternalUrlExpansionOptions } from "./bash-skill-urls";
|
||||
import { resolveEvalBackends } from "./eval-backends";
|
||||
import { invalidateGithubCacheForBashCommand } from "./gh-cache-invalidation";
|
||||
import {
|
||||
formatStyledTruncationWarning,
|
||||
@@ -131,7 +132,7 @@ async function saveBashOriginalArtifact(session: ToolSession, originalText: stri
|
||||
}
|
||||
}
|
||||
|
||||
const BASH_TIMEOUT_DESCRIPTION = `timeout in seconds; clamped to ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}`;
|
||||
const BASH_TIMEOUT_DESCRIPTION = `timeout in seconds; 0 disables the command deadline; nonzero values are clamped to ${TOOL_TIMEOUTS.bash.min}-${TOOL_TIMEOUTS.bash.max}`;
|
||||
|
||||
const bashSchemaBase = type({
|
||||
command: type("string").describe("command to execute"),
|
||||
@@ -166,6 +167,7 @@ export interface BashToolDetails {
|
||||
meta?: OutputMeta;
|
||||
timeoutSeconds?: number;
|
||||
requestedTimeoutSeconds?: number;
|
||||
timeoutDisabled?: boolean;
|
||||
wallTimeMs?: number;
|
||||
/** Exit code of a command that ran to completion but failed (non-zero). */
|
||||
exitCode?: number;
|
||||
@@ -375,7 +377,24 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
};
|
||||
readonly label = "Bash";
|
||||
readonly loadMode = "essential";
|
||||
readonly description: string;
|
||||
get description(): string {
|
||||
const evalBackends = resolveEvalBackends(this.session);
|
||||
const isToolActive = (name: string, fallback: boolean): boolean => this.session.isToolActive?.(name) ?? fallback;
|
||||
return prompt.render(bashDescription, {
|
||||
asyncEnabled: this.#asyncEnabled,
|
||||
autoBackgroundEnabled: this.#autoBackgroundEnabled,
|
||||
autoBackgroundThresholdSeconds: Math.max(0, Math.floor(this.#autoBackgroundThresholdMs / 1000)),
|
||||
hasAstGrep: isToolActive("ast_grep", this.session.settings.get("astGrep.enabled")),
|
||||
hasAstEdit: isToolActive("ast_edit", this.session.settings.get("astEdit.enabled")),
|
||||
hasGrep: isToolActive("grep", this.session.settings.get("grep.enabled")),
|
||||
hasGlob: isToolActive("glob", this.session.settings.get("glob.enabled")),
|
||||
hasRead: isToolActive("read", true),
|
||||
hasEval: isToolActive(
|
||||
"eval",
|
||||
evalBackends.python || evalBackends.js || evalBackends.ruby || evalBackends.julia,
|
||||
),
|
||||
});
|
||||
}
|
||||
readonly parameters: BashToolSchema;
|
||||
// Non-pty calls run alongside each other (the executor isolates overlapping
|
||||
// runs on the same shell session); pty takes over the terminal UI and must
|
||||
@@ -397,15 +416,6 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
),
|
||||
);
|
||||
this.parameters = this.#asyncEnabled ? bashSchemaWithAsync : bashSchemaBase;
|
||||
this.description = prompt.render(bashDescription, {
|
||||
asyncEnabled: this.#asyncEnabled,
|
||||
autoBackgroundEnabled: this.#autoBackgroundEnabled,
|
||||
autoBackgroundThresholdSeconds: Math.max(0, Math.floor(this.#autoBackgroundThresholdMs / 1000)),
|
||||
hasAstGrep: this.session.settings.get("astGrep.enabled"),
|
||||
hasAstEdit: this.session.settings.get("astEdit.enabled"),
|
||||
hasGrep: this.session.settings.get("grep.enabled"),
|
||||
hasGlob: this.session.settings.get("glob.enabled"),
|
||||
});
|
||||
}
|
||||
|
||||
#formatResultOutput(result: BashResult | BashInteractiveResult): string {
|
||||
@@ -421,7 +431,11 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
* completed command that failed; #buildCompletedResult surfaces it as an
|
||||
* error *result* (carrying execution details) rather than a throw.
|
||||
*/
|
||||
#throwIfUnfinished(result: BashResult | BashInteractiveResult, timeoutSec: number, outputText: string): void {
|
||||
#throwIfUnfinished(
|
||||
result: BashResult | BashInteractiveResult,
|
||||
timeoutSec: number | undefined,
|
||||
outputText: string,
|
||||
): void {
|
||||
if (result.cancelled) {
|
||||
// executeBash output already carries a `[Command cancelled]` notice from
|
||||
// the sink; PTY/bridge interactive output does not, so annotate it here.
|
||||
@@ -431,11 +445,9 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
}
|
||||
if (isInteractiveResult(result) && result.timedOut) {
|
||||
const out = normalizeResultOutput(result);
|
||||
throw new ToolError(
|
||||
out
|
||||
? `${out}\n\n[Command timed out after ${timeoutSec} seconds]`
|
||||
: `Command timed out after ${timeoutSec} seconds`,
|
||||
);
|
||||
const message =
|
||||
timeoutSec === undefined ? "Command timed out" : `Command timed out after ${timeoutSec} seconds`;
|
||||
throw new ToolError(out ? `${out}\n\n[${message}]` : message);
|
||||
}
|
||||
if (result.exitCode === undefined) {
|
||||
throw new ToolError(`${outputText}\n\nCommand failed: missing exit status`);
|
||||
@@ -444,7 +456,7 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
|
||||
async #buildCompletedResult(
|
||||
result: BashResult | BashInteractiveResult,
|
||||
timeoutSec: number,
|
||||
timeoutSec: number | undefined,
|
||||
options: {
|
||||
requestedTimeoutSec?: number;
|
||||
notices?: readonly string[];
|
||||
@@ -472,7 +484,12 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
// Aborts / timeouts / missing-status still propagate as thrown errors.
|
||||
this.#throwIfUnfinished(result, timeoutSec, outputText);
|
||||
|
||||
const details: BashToolDetails = { timeoutSeconds: timeoutSec };
|
||||
const details: BashToolDetails = {};
|
||||
if (timeoutSec === undefined) {
|
||||
details.timeoutDisabled = true;
|
||||
} else {
|
||||
details.timeoutSeconds = timeoutSec;
|
||||
}
|
||||
if (options.requestedTimeoutSec !== undefined && options.requestedTimeoutSec !== timeoutSec) {
|
||||
details.requestedTimeoutSeconds = options.requestedTimeoutSec;
|
||||
}
|
||||
@@ -503,13 +520,17 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
jobId: string,
|
||||
label: string,
|
||||
previewText: string,
|
||||
timeoutSec: number,
|
||||
timeoutSec: number | undefined,
|
||||
options: { requestedTimeoutSec?: number; notices?: readonly string[] } = {},
|
||||
): AgentToolResult<BashToolDetails> {
|
||||
const details: BashToolDetails = {
|
||||
timeoutSeconds: timeoutSec,
|
||||
async: { state: "running", jobId, type: "bash" },
|
||||
};
|
||||
if (timeoutSec === undefined) {
|
||||
details.timeoutDisabled = true;
|
||||
} else {
|
||||
details.timeoutSeconds = timeoutSec;
|
||||
}
|
||||
if (options.requestedTimeoutSec !== undefined && options.requestedTimeoutSec !== timeoutSec) {
|
||||
details.requestedTimeoutSeconds = options.requestedTimeoutSec;
|
||||
}
|
||||
@@ -539,8 +560,8 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
#startManagedBashJob(options: {
|
||||
command: string;
|
||||
commandCwd: string;
|
||||
timeoutMs: number;
|
||||
timeoutSec: number;
|
||||
timeoutMs: number | undefined;
|
||||
timeoutSec: number | undefined;
|
||||
requestedTimeoutSec?: number;
|
||||
notices?: readonly string[];
|
||||
|
||||
@@ -569,7 +590,7 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
const result = await executeBash(options.command, {
|
||||
cwd: options.commandCwd,
|
||||
sessionKey: `${this.session.getSessionId?.() ?? ""}:async:${jobId}`,
|
||||
timeout: options.timeoutMs,
|
||||
timeout: options.timeoutMs ?? 0,
|
||||
signal: runSignal,
|
||||
env: options.resolvedEnv,
|
||||
artifactPath,
|
||||
@@ -661,8 +682,9 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
}
|
||||
}
|
||||
|
||||
#resolveAutoBackgroundWaitMs(timeoutMs: number): number {
|
||||
#resolveAutoBackgroundWaitMs(timeoutMs: number | undefined): number {
|
||||
if (this.#autoBackgroundThresholdMs <= 0) return 0;
|
||||
if (timeoutMs === undefined) return this.#autoBackgroundThresholdMs;
|
||||
const timeoutBufferMs = 1_000;
|
||||
return Math.max(0, Math.min(this.#autoBackgroundThresholdMs, timeoutMs - timeoutBufferMs));
|
||||
}
|
||||
@@ -765,13 +787,17 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
throw new ToolError(`Working directory is not a directory: ${commandCwd}`);
|
||||
}
|
||||
|
||||
// Clamp to reasonable range: 1s - 3600s (1 hour)
|
||||
// A timeout of 0 is an explicit long-running-command contract: the user
|
||||
// must still cancel the call or job, but OMP does not impose a deadline.
|
||||
const requestedTimeoutSec = rawTimeout;
|
||||
const timeoutSec = clampTimeout("bash", requestedTimeoutSec);
|
||||
const timeoutMs = timeoutSec * 1000;
|
||||
const timeoutDisabled = requestedTimeoutSec === 0;
|
||||
const timeoutSec = timeoutDisabled ? undefined : clampTimeout("bash", requestedTimeoutSec);
|
||||
const timeoutMs = timeoutSec === undefined ? undefined : timeoutSec * 1000;
|
||||
const pendingNotices: string[] = [];
|
||||
const timeoutClampNotice = formatTimeoutClampNotice(requestedTimeoutSec, timeoutSec);
|
||||
if (timeoutClampNotice) pendingNotices.push(timeoutClampNotice);
|
||||
if (timeoutSec !== undefined) {
|
||||
const timeoutClampNotice = formatTimeoutClampNotice(requestedTimeoutSec, timeoutSec);
|
||||
if (timeoutClampNotice) pendingNotices.push(timeoutClampNotice);
|
||||
}
|
||||
|
||||
if (asyncRequested) {
|
||||
if (!this.session.asyncJobManager) {
|
||||
@@ -909,14 +935,16 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
throw new ToolAbortError("Command aborted");
|
||||
}
|
||||
|
||||
const timeoutPromise = Bun.sleep(timeoutMs).then(() => ({ kind: "timeout" as const }));
|
||||
const timeoutPromise = timeoutMs
|
||||
? Bun.sleep(timeoutMs).then(() => ({ kind: "timeout" as const }))
|
||||
: undefined;
|
||||
// Poll until the process exits, times out, or the caller aborts.
|
||||
for (;;) {
|
||||
const racers: Array<Promise<BridgeRaceResult>> = [
|
||||
exitPromise.then(s => ({ kind: "exit" as const, status: s })),
|
||||
timeoutPromise,
|
||||
Bun.sleep(250).then(() => ({ kind: "poll" as const })),
|
||||
];
|
||||
if (timeoutPromise) racers.push(timeoutPromise);
|
||||
if (signal) {
|
||||
racers.push(abortedP.then(() => ({ kind: "aborted" as const })));
|
||||
}
|
||||
@@ -1053,7 +1081,7 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
: await executeBash(command, {
|
||||
cwd: commandCwd,
|
||||
sessionKey: this.session.getSessionId?.() ?? undefined,
|
||||
timeout: timeoutMs,
|
||||
timeout: timeoutMs ?? 0,
|
||||
signal,
|
||||
env: resolvedEnv,
|
||||
artifactPath,
|
||||
@@ -1074,11 +1102,9 @@ export class BashTool implements AgentTool<typeof bashSchemaBase | typeof bashSc
|
||||
}
|
||||
if (isInteractiveResult(result) && result.timedOut) {
|
||||
const out = normalizeResultOutput(result);
|
||||
throw new ToolError(
|
||||
out
|
||||
? `${out}\n\n[Command timed out after ${timeoutSec} seconds]`
|
||||
: `Command timed out after ${timeoutSec} seconds`,
|
||||
);
|
||||
const message =
|
||||
timeoutSec === undefined ? "Command timed out" : `Command timed out after ${timeoutSec} seconds`;
|
||||
throw new ToolError(out ? `${out}\n\n[${message}]` : message);
|
||||
}
|
||||
return this.#buildCompletedResult(result, timeoutSec, {
|
||||
requestedTimeoutSec,
|
||||
@@ -1286,13 +1312,17 @@ export function createShellRenderer<TArgs>(config: ShellRendererConfig<TArgs>) {
|
||||
const showingFullOutput = expanded && renderContext?.isFullOutput === true;
|
||||
|
||||
// Build truncation warning
|
||||
const timeoutSeconds = details?.timeoutSeconds ?? renderContext?.timeout;
|
||||
const timeoutDisabled = details?.timeoutDisabled === true || renderContext?.timeout === 0;
|
||||
const timeoutSeconds = timeoutDisabled ? undefined : (details?.timeoutSeconds ?? renderContext?.timeout);
|
||||
const requestedTimeoutSeconds = details?.requestedTimeoutSeconds;
|
||||
const wallTimeMs = details?.wallTimeMs;
|
||||
const statsParts: string[] = [];
|
||||
if (wallTimeMs !== undefined) {
|
||||
statsParts.push(`Wall: ${formatWallTimeSeconds(wallTimeMs)}s`);
|
||||
}
|
||||
if (timeoutDisabled) {
|
||||
statsParts.push("Timeout: disabled");
|
||||
}
|
||||
if (typeof timeoutSeconds === "number") {
|
||||
statsParts.push(
|
||||
requestedTimeoutSeconds !== undefined && requestedTimeoutSeconds !== timeoutSeconds
|
||||
|
||||
@@ -29,15 +29,20 @@ export const DEFAULT_VIEWPORT = { width: 1365, height: 768, deviceScaleFactor: 1
|
||||
* connection dropped, etc.).
|
||||
*/
|
||||
export const BROWSER_PROTOCOL_TIMEOUT_MS = 60_000;
|
||||
const ENABLE_AUTOMATION_FLAG = "--enable-automation";
|
||||
// Automation-tell launch flags that puppeteer-core adds by default. We suppress
|
||||
// them via `ignoreDefaultArgs` (the supported escape hatch) to mirror xxxx's
|
||||
// chromiumSwitches patch. `--enable-automation` is the loudest: it sets
|
||||
// chromiumSwitches patch. `--enable-automation` is the loudest: it normally sets
|
||||
// navigator.webdriver=true and shows the "controlled by automated software" infobar.
|
||||
// Edge is the launch-stability exception: it can exit before CDP opens when this
|
||||
// default flag is stripped, so Edge keeps Puppeteer's flag while our explicit
|
||||
// `--disable-blink-features=AutomationControlled` launch arg still handles
|
||||
// navigator.webdriver.
|
||||
// `ignoreDefaultArgs` does exact-string matching, so each entry must be a flag that
|
||||
// puppeteer emits verbatim. The default `--disable-features=...` string can't be
|
||||
// matched this way; it is neutralized in the puppeteer-core patch (ChromeLauncher).
|
||||
const STEALTH_IGNORE_DEFAULT_ARGS = [
|
||||
"--enable-automation",
|
||||
ENABLE_AUTOMATION_FLAG,
|
||||
"--disable-extensions",
|
||||
"--disable-default-apps",
|
||||
"--disable-component-extensions-with-background-pages",
|
||||
@@ -47,6 +52,23 @@ const STEALTH_IGNORE_DEFAULT_ARGS = [
|
||||
"--disable-ipc-flooding-protection",
|
||||
"--metrics-recording-only",
|
||||
];
|
||||
|
||||
function isMicrosoftEdgeExecutable(executablePath: string | undefined): boolean {
|
||||
if (!executablePath) return false;
|
||||
const normalizedPath = executablePath.replaceAll("\\", "/").toLowerCase();
|
||||
const executableName = normalizedPath.slice(normalizedPath.lastIndexOf("/") + 1);
|
||||
return (
|
||||
executableName === "msedge.exe" ||
|
||||
executableName === "microsoft edge" ||
|
||||
executableName.startsWith("microsoft-edge")
|
||||
);
|
||||
}
|
||||
|
||||
function stealthIgnoreDefaultArgs(executablePath: string | undefined): string[] {
|
||||
if (!isMicrosoftEdgeExecutable(executablePath)) return [...STEALTH_IGNORE_DEFAULT_ARGS];
|
||||
return STEALTH_IGNORE_DEFAULT_ARGS.filter(arg => arg !== ENABLE_AUTOMATION_FLAG);
|
||||
}
|
||||
|
||||
const STEALTH_ACCEPT_LANGUAGE = "en-US,en";
|
||||
|
||||
const USER_AGENT_TARGET_TIMEOUT_MS = 5_000;
|
||||
@@ -282,12 +304,13 @@ export async function launchHeadlessBrowser(opts: LaunchHeadlessOptions): Promis
|
||||
if (ignoreCert === "true" || ignoreCert === "1" || ignoreCert === "yes" || ignoreCert === "on") {
|
||||
launchArgs.push("--ignore-certificate-errors");
|
||||
}
|
||||
const executablePath = await ensureChromiumExecutable();
|
||||
return await puppeteer.launch({
|
||||
headless: opts.headless,
|
||||
defaultViewport: opts.headless ? initialViewport : null,
|
||||
executablePath: await ensureChromiumExecutable(),
|
||||
executablePath,
|
||||
args: launchArgs,
|
||||
ignoreDefaultArgs: [...STEALTH_IGNORE_DEFAULT_ARGS],
|
||||
ignoreDefaultArgs: stealthIgnoreDefaultArgs(executablePath),
|
||||
protocolTimeout: BROWSER_PROTOCOL_TIMEOUT_MS,
|
||||
});
|
||||
}
|
||||
@@ -737,6 +760,10 @@ export async function applyStealthPatches(
|
||||
await injectStealthScripts(page);
|
||||
}
|
||||
|
||||
export function stealthIgnoreDefaultArgsForTest(executablePath: string | undefined): string[] {
|
||||
return stealthIgnoreDefaultArgs(executablePath);
|
||||
}
|
||||
|
||||
export function targetSupportsUserAgentOverrideForTest(target: Target): boolean {
|
||||
return targetSupportsUserAgentOverride(target);
|
||||
}
|
||||
|
||||
@@ -48,12 +48,14 @@ import {
|
||||
type LineRange,
|
||||
parseLineRanges,
|
||||
pathTargetsSsh,
|
||||
probeLiteralPathExists,
|
||||
type ResolvedSearchTarget,
|
||||
resolveReadPath,
|
||||
resolveToolSearchScope,
|
||||
selectorLineRanges,
|
||||
splitInternalUrlSel,
|
||||
splitPathAndSel,
|
||||
splitPathAndSelPreferringLiteral,
|
||||
toPathList,
|
||||
} from "./path-utils";
|
||||
import {
|
||||
@@ -77,6 +79,9 @@ const searchSchema = type({
|
||||
"path?": searchPathEntry.describe(
|
||||
'file, directory, glob, internal URL, or "<file>:<lines>" selector to search; pass several as a semicolon-delimited list ("src; tests"). Omitted -> searches the workspace root (".")',
|
||||
),
|
||||
"selector?": type("string").describe(
|
||||
'line selector without a leading colon (e.g. "50-100", "50+10", "50-100,200-300"); keeps `path` literal when filenames contain colons',
|
||||
),
|
||||
"case?": type("boolean").describe("case-sensitive search"),
|
||||
"gitignore?": type("boolean").describe("respect gitignore"),
|
||||
"skip?": type("number")
|
||||
@@ -119,6 +124,7 @@ const SEARCH_GREP_TIMEOUT_MS = 30_000;
|
||||
interface GrepPathSpec {
|
||||
original: string;
|
||||
clean: string;
|
||||
literalFilesystemMatch?: boolean;
|
||||
ranges?: [LineRange, ...LineRange[]];
|
||||
}
|
||||
|
||||
@@ -147,9 +153,38 @@ function isReadSelectorGrammar(sel: string): boolean {
|
||||
return lower === "raw" || lower === "conflicts" || parseLineRanges(sel) !== null;
|
||||
}
|
||||
|
||||
function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] {
|
||||
async function parsePathSpecs(
|
||||
rawEntries: readonly string[],
|
||||
cwd: string,
|
||||
explicitSelector?: string,
|
||||
): Promise<GrepPathSpec[]> {
|
||||
const explicitRanges =
|
||||
explicitSelector === undefined || explicitSelector.length === 0 ? undefined : parseLineRanges(explicitSelector);
|
||||
if (explicitSelector !== undefined && !explicitRanges) {
|
||||
throw new ToolError(
|
||||
`selector "${explicitSelector}" is invalid — use line ranges like "50-100", "50+10", or "50-100,200-300" without a leading colon`,
|
||||
);
|
||||
}
|
||||
const specs: GrepPathSpec[] = [];
|
||||
for (const entry of rawEntries) {
|
||||
if (explicitRanges) {
|
||||
// Separate selector parameter makes `path` deterministic: first try the
|
||||
// exact local filesystem path (with read-path normalization), then let
|
||||
// archive/internal/URL resolution handle non-literal structured paths.
|
||||
const rawPathHasScheme = /^[a-z][a-z0-9+.-]*:\/\//i.test(entry);
|
||||
const probe = rawPathHasScheme ? "missing" : await probeLiteralPathExists(entry, cwd);
|
||||
// `"unknown"` covers EACCES/IO where we cannot confirm existence — treat
|
||||
// it as a literal so a real file such as `test:1-2` under an unreadable
|
||||
// parent is never silently reinterpreted as `test` + selector.
|
||||
const literalMatch = probe !== "missing";
|
||||
specs.push({
|
||||
original: entry,
|
||||
clean: literalMatch && !rawPathHasScheme ? resolveReadPath(entry, cwd) : entry,
|
||||
literalFilesystemMatch: literalMatch,
|
||||
ranges: explicitRanges,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
// Internal URLs (`artifact://`, `skill://`, …) use the URL-aware splitter,
|
||||
// which peels selector-shaped tails only for selector-capable schemes and
|
||||
// leaves opaque ones (`mcp://`) intact. Unlike filesystem paths, their
|
||||
@@ -168,10 +203,14 @@ function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] {
|
||||
specs.push({ original: entry, clean: internalSplit.path, ranges: selectorLineRanges(internalSplit.sel) });
|
||||
continue;
|
||||
}
|
||||
const split = splitPathAndSel(entry);
|
||||
let clean = entry;
|
||||
// Prefer a literal filesystem match when one exists — a real file named
|
||||
// `test:1-2` outranks the `:1-2` selector interpretation (issue #4618).
|
||||
const strictSplit = splitPathAndSel(entry);
|
||||
const split = await splitPathAndSelPreferringLiteral(entry, cwd);
|
||||
const literalFilesystemMatch = strictSplit.sel !== undefined && split.sel === undefined;
|
||||
let clean = literalFilesystemMatch ? resolveReadPath(entry, cwd) : entry;
|
||||
let ranges: [LineRange, ...LineRange[]] | undefined;
|
||||
if (split.sel) {
|
||||
if (!literalFilesystemMatch && split.sel) {
|
||||
const parsed = parseLineRanges(split.sel);
|
||||
if (!parsed) {
|
||||
throw new ToolError(
|
||||
@@ -184,7 +223,7 @@ function parsePathSpecs(rawEntries: readonly string[]): GrepPathSpec[] {
|
||||
clean = split.path;
|
||||
ranges = parsed;
|
||||
}
|
||||
specs.push({ original: entry, clean, ranges });
|
||||
specs.push({ original: entry, clean, literalFilesystemMatch, ranges });
|
||||
}
|
||||
return specs;
|
||||
}
|
||||
@@ -220,7 +259,7 @@ function matchAbsolutePath(matchPath: string, searchPath: string): string {
|
||||
* cleanup hook the caller MUST invoke in a `finally`.
|
||||
*/
|
||||
async function resolveArchiveSearchPaths(
|
||||
paths: string[],
|
||||
pathSpecs: readonly GrepPathSpec[],
|
||||
cwd: string,
|
||||
): Promise<{
|
||||
resolvedPaths: string[];
|
||||
@@ -229,17 +268,18 @@ async function resolveArchiveSearchPaths(
|
||||
unreadable: string[];
|
||||
cleanup: () => Promise<void>;
|
||||
}> {
|
||||
const resolvedPaths = paths.slice();
|
||||
const resolvedPaths = pathSpecs.map(spec => spec.clean);
|
||||
const displayMap = new Map<string, string>();
|
||||
const displaySet = new Set<string>();
|
||||
const unreadable: string[] = [];
|
||||
let tempDir: string | undefined;
|
||||
const archiveCache = new Map<string, ArchiveReader>();
|
||||
|
||||
for (let idx = 0; idx < paths.length; idx++) {
|
||||
const entry = paths[idx];
|
||||
for (let idx = 0; idx < pathSpecs.length; idx++) {
|
||||
const spec = pathSpecs[idx];
|
||||
if (!spec || spec.literalFilesystemMatch) continue;
|
||||
const entry = spec.clean;
|
||||
const candidates = parseArchivePathCandidates(entry);
|
||||
// Longest archive prefix first; we want the one whose member portion is non-empty.
|
||||
const member = candidates.find(c => c.subPath !== "" && c.archivePath !== entry);
|
||||
if (!member) continue;
|
||||
|
||||
@@ -879,7 +919,7 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
_onUpdate?: AgentToolUpdateCallback<GrepToolDetails>,
|
||||
_toolContext?: AgentToolContext,
|
||||
): Promise<AgentToolResult<GrepToolDetails>> {
|
||||
const { pattern, path: rawPath, case: caseSensitive, gitignore, skip } = params;
|
||||
const { pattern, path: rawPath, selector, case: caseSensitive, gitignore, skip } = params;
|
||||
|
||||
return untilAborted(signal, async () => {
|
||||
// Preserve the pattern verbatim — leading/trailing whitespace is
|
||||
@@ -897,8 +937,7 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
const scopedPaths = toPathList(rawPath);
|
||||
const effectivePaths = scopedPaths.length > 0 ? scopedPaths : ["."];
|
||||
const rawEntries = await expandDelimitedPathEntries(effectivePaths, this.session.cwd);
|
||||
const pathSpecs = parsePathSpecs(rawEntries);
|
||||
const paths = pathSpecs.map(spec => spec.clean);
|
||||
const pathSpecs = await parsePathSpecs(rawEntries, this.session.cwd, selector);
|
||||
const materializedExternalPaths = new Map<string, string>();
|
||||
const materializeExternalUrlForSearch = async (rawPath: string) => {
|
||||
const target = parseReadUrlTarget(rawPath);
|
||||
@@ -917,7 +956,7 @@ export class GrepTool implements AgentTool<typeof searchSchema, GrepToolDetails>
|
||||
displaySet: archiveDisplaySet,
|
||||
unreadable: archiveUnreadable,
|
||||
cleanup: cleanupArchiveScratch,
|
||||
} = await resolveArchiveSearchPaths(paths, this.session.cwd);
|
||||
} = await resolveArchiveSearchPaths(pathSpecs, this.session.cwd);
|
||||
try {
|
||||
const internalResolution = await resolveInternalSearchInputs({
|
||||
pathSpecs,
|
||||
|
||||
@@ -224,6 +224,10 @@ export interface ToolSession {
|
||||
getAgentId?: () => string | null;
|
||||
/** Look up a registered tool by name (used by the eval js backend's tool bridge). */
|
||||
getToolByName?: (name: string) => AgentTool | undefined;
|
||||
/** Return whether a built-in tool is active in this turn's tool set. */
|
||||
isToolActive?: (name: string) => boolean;
|
||||
/** Update the active built-in tool predicate when a session changes tools mid-run. */
|
||||
setActiveToolNames?: (names: Iterable<string>) => void;
|
||||
/** Agent registry for IRC routing across live sessions. */
|
||||
agentRegistry?: AgentRegistry;
|
||||
/** Get artifacts directory for artifact:// URLs */
|
||||
@@ -647,6 +651,13 @@ export async function createTools(session: ToolSession, toolNames?: string[]): P
|
||||
...(goalModeActive ? ([["goal", HIDDEN_TOOLS.goal]] as const) : []),
|
||||
];
|
||||
|
||||
const activeToolNames = new Set(baseEntries.map(([name]) => name));
|
||||
if (session.setActiveToolNames) {
|
||||
session.setActiveToolNames(activeToolNames);
|
||||
} else {
|
||||
session.isToolActive = name => activeToolNames.has(name);
|
||||
}
|
||||
|
||||
const baseResults = await Promise.all(
|
||||
baseEntries.map(async ([name, factory]) => {
|
||||
const tool = await logger.time(`createTools:${name}`, factory as ToolFactory, session);
|
||||
|
||||
@@ -315,6 +315,46 @@ export function splitPathAndSel(rawPath: string): { path: string; sel?: string }
|
||||
return { path: basePath, sel };
|
||||
}
|
||||
|
||||
/**
|
||||
* Three-way probe for whether the exact filesystem entry named by `filePath`
|
||||
* exists. `stat` (used earlier) failed for reasons other than "no such file"
|
||||
* (dangling symlink, `EACCES` on a parent, transient I/O), and each of those
|
||||
* silently reinterpreted a real literal path such as `test:1-2` as `test`
|
||||
* plus selector `1-2` (issue #4618). `lstat` inspects the entry itself, so a
|
||||
* dangling symlink is still detected as present; ambiguous errors resolve to
|
||||
* `"unknown"` so callers keep the raw path instead of guessing.
|
||||
*/
|
||||
export async function probeLiteralPathExists(filePath: string, cwd: string): Promise<"exists" | "missing" | "unknown"> {
|
||||
const resolved = resolveReadPath(filePath, cwd);
|
||||
try {
|
||||
await fs.promises.lstat(resolved);
|
||||
return "exists";
|
||||
} catch (err) {
|
||||
if (isEnoent(err) || isEnotdir(err)) return "missing";
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Async sibling of {@link splitPathAndSel} that prefers a literal filesystem
|
||||
* path over selector interpretation. Filenames whose tail matches the selector
|
||||
* grammar (e.g. `test:1-2`, `log:raw`) are legal on POSIX; without this the
|
||||
* strict splitter peels the tail and both `read` and `grep` refuse to open the
|
||||
* real file (issue #4618). The literal wins on a confirmed `lstat`, and also
|
||||
* on `"unknown"` (`EACCES` on a parent, transient I/O), so an unreachable
|
||||
* literal is never silently reinterpreted as `path + selector`. Only a
|
||||
* definitive `ENOENT`/`ENOTDIR` falls back to the strict split.
|
||||
*/
|
||||
export async function splitPathAndSelPreferringLiteral(
|
||||
rawPath: string,
|
||||
cwd: string,
|
||||
): Promise<{ path: string; sel?: string }> {
|
||||
const strict = splitPathAndSel(rawPath);
|
||||
if (strict.sel === undefined) return strict;
|
||||
const probe = await probeLiteralPathExists(rawPath, cwd);
|
||||
return probe === "missing" ? strict : { path: rawPath };
|
||||
}
|
||||
|
||||
/**
|
||||
* Variant of {@link splitPathAndSel} for internal URLs (`scheme://...`).
|
||||
*
|
||||
@@ -669,7 +709,12 @@ export async function splitDelimitedPathEntry(
|
||||
const normalizedEntry = normalizePathLikeInput(entry);
|
||||
if (!hasTopLevelPathDelimiter(normalizedEntry)) return null;
|
||||
if (isInternalUrlPath(normalizedEntry)) return null;
|
||||
|
||||
// A real POSIX file may contain the delimiter and a selector-shaped tail
|
||||
// (`a;b:1-2`, `a b:1-2`). Preserve the raw entry whenever the full literal
|
||||
// resolves — or is only ambiguous — so downstream literal-preferring
|
||||
// splitters see it before delimiter expansion peels or splits (issue #4618
|
||||
// reviewer feedback: delimited expansion ran before the literal check).
|
||||
if ((await probeLiteralPathExists(normalizedEntry, cwd)) !== "missing") return null;
|
||||
const splitter = options.splitter ?? parseSearchPath;
|
||||
const peeledEntry = splitPathAndSel(normalizedEntry).path;
|
||||
if (!hasGlobPathChars(peeledEntry) && (await delimitedPathPartResolves(normalizedEntry, cwd, splitter))) {
|
||||
|
||||
@@ -99,10 +99,12 @@ import {
|
||||
type LineRange,
|
||||
parseLineRanges,
|
||||
pathTargetsSsh,
|
||||
probeLiteralPathExists,
|
||||
resolveReadPath,
|
||||
splitDelimitedPathEntry,
|
||||
splitInternalUrlSel,
|
||||
splitPathAndSel,
|
||||
splitPathAndSelPreferringLiteral,
|
||||
} from "./path-utils";
|
||||
import { formatBytes, replaceTabs, shortenPath, wrapBrackets } from "./render-utils";
|
||||
import {
|
||||
@@ -746,7 +748,10 @@ function splitPdfImageMemberReadPath(readPath: string): { pdfPath: string; membe
|
||||
|
||||
const readSchema = type({
|
||||
path: type("string").describe(
|
||||
'Local path, internal URI (e.g. "omp://", "issue://123", "pr://123"), or URL; append :<sel> for line ranges or raw mode (e.g. "src/foo.ts:50-100")',
|
||||
'Local path, internal URI (e.g. "omp://", "issue://123", "pr://123"), or URL. Inline :<sel> is still accepted for compatibility.',
|
||||
),
|
||||
"selector?": type("string").describe(
|
||||
'selector without a leading colon (e.g. "50-100", "raw", "raw:50-100", "conflicts"); keeps `path` literal when filenames contain colons',
|
||||
),
|
||||
});
|
||||
|
||||
@@ -2113,6 +2118,14 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
_toolContext?: AgentToolContext,
|
||||
): Promise<AgentToolResult<ReadToolDetails>> {
|
||||
let { path: readPath } = params;
|
||||
let explicitSelector = params.selector?.trim();
|
||||
let explicitParsedSelector = explicitSelector === undefined ? undefined : parseSel(explicitSelector);
|
||||
if (
|
||||
params.selector !== undefined &&
|
||||
(explicitSelector === undefined || explicitSelector.length === 0 || explicitParsedSelector?.kind === "none")
|
||||
) {
|
||||
throw invalidSelector(params.selector);
|
||||
}
|
||||
if (readPath.startsWith("file://")) {
|
||||
readPath = expandPath(readPath);
|
||||
}
|
||||
@@ -2133,40 +2146,55 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
if (!this.session.settings.get("fetch.enabled")) {
|
||||
throw new ToolError("URL reads are disabled by settings.");
|
||||
}
|
||||
if (parsedUrlTarget.ranges !== undefined) {
|
||||
if (explicitParsedSelector?.kind === "conflicts") {
|
||||
throw new ToolError("The explicit read selector `conflicts` is only supported for local files.");
|
||||
}
|
||||
const urlRaw =
|
||||
explicitParsedSelector === undefined ? parsedUrlTarget.raw : isRawSelector(explicitParsedSelector);
|
||||
const urlRanges =
|
||||
explicitParsedSelector?.kind === "lines" ? explicitParsedSelector.ranges : parsedUrlTarget.ranges;
|
||||
if (urlRanges !== undefined && urlRanges.length > 1) {
|
||||
const cached = await loadReadUrlCacheEntry(
|
||||
this.session,
|
||||
{ path: parsedUrlTarget.path, raw: parsedUrlTarget.raw },
|
||||
{ path: parsedUrlTarget.path, raw: urlRaw },
|
||||
signal,
|
||||
{ ensureArtifact: true, preferCached: true },
|
||||
);
|
||||
return this.#buildInMemoryMultiRangeResult(cached.output, parsedUrlTarget.ranges, {
|
||||
return this.#buildInMemoryMultiRangeResult(cached.output, urlRanges, {
|
||||
details: { ...cached.details },
|
||||
sourceUrl: cached.details.finalUrl,
|
||||
entityLabel: "URL output",
|
||||
raw: parsedUrlTarget.raw,
|
||||
raw: urlRaw,
|
||||
immutable: true,
|
||||
});
|
||||
}
|
||||
if (parsedUrlTarget.offset !== undefined || parsedUrlTarget.limit !== undefined) {
|
||||
const urlRange = urlRanges?.[0];
|
||||
const urlOffset = explicitParsedSelector?.kind === "lines" ? urlRange?.startLine : parsedUrlTarget.offset;
|
||||
const urlLimit =
|
||||
explicitParsedSelector?.kind === "lines" && urlRange
|
||||
? urlRange.endLine !== undefined
|
||||
? urlRange.endLine - urlRange.startLine + 1
|
||||
: undefined
|
||||
: parsedUrlTarget.limit;
|
||||
if (urlOffset !== undefined || urlLimit !== undefined) {
|
||||
const cached = await loadReadUrlCacheEntry(
|
||||
this.session,
|
||||
{ path: parsedUrlTarget.path, raw: parsedUrlTarget.raw },
|
||||
{ path: parsedUrlTarget.path, raw: urlRaw },
|
||||
signal,
|
||||
{
|
||||
ensureArtifact: true,
|
||||
preferCached: true,
|
||||
},
|
||||
);
|
||||
return this.#buildInMemoryTextResult(cached.output, parsedUrlTarget.offset, parsedUrlTarget.limit, {
|
||||
return this.#buildInMemoryTextResult(cached.output, urlOffset, urlLimit, {
|
||||
details: { ...cached.details },
|
||||
sourceUrl: cached.details.finalUrl,
|
||||
entityLabel: "URL output",
|
||||
raw: parsedUrlTarget.raw,
|
||||
raw: urlRaw,
|
||||
immutable: true,
|
||||
});
|
||||
}
|
||||
return executeReadUrl(this.session, { path: parsedUrlTarget.path, raw: parsedUrlTarget.raw }, signal);
|
||||
return executeReadUrl(this.session, { path: parsedUrlTarget.path, raw: urlRaw }, signal);
|
||||
}
|
||||
|
||||
// Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://, omp://, issue://, pr://).
|
||||
@@ -2174,8 +2202,9 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
// off the URL and surfaced via parseSel rather than confusing handlers.
|
||||
const internalRouter = InternalUrlRouter.instance();
|
||||
if (internalRouter.canHandle(readPath)) {
|
||||
const internalTarget = splitInternalUrlSel(readPath);
|
||||
const parsed = parseSel(internalTarget.sel);
|
||||
const internalTarget =
|
||||
explicitSelector === undefined ? splitInternalUrlSel(readPath) : { path: readPath, sel: explicitSelector };
|
||||
const parsed = explicitParsedSelector ?? parseSel(internalTarget.sel);
|
||||
if (internalTarget.sel !== undefined && parsed.kind === "none") {
|
||||
throw new ToolError(
|
||||
`Invalid selector ':${internalTarget.sel}' on '${internalTarget.path}'. Use :N, :N-M, :N+K, :N- (open-ended), a comma-separated list of ranges, :raw, or a range combined with raw (e.g. :raw:50-100).`,
|
||||
@@ -2192,7 +2221,16 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
skills: this.session.skills,
|
||||
});
|
||||
if (localFile) {
|
||||
readPath = internalTarget.sel === undefined ? localFile.path : `${localFile.path}:${internalTarget.sel}`;
|
||||
readPath = localFile.path;
|
||||
// Promote the URL-embedded selector into the explicit-selector state so
|
||||
// downstream literal-preferring routing does NOT re-split the synthesized
|
||||
// `${localFile.path}:${sel}` string — a sibling literal file at that name
|
||||
// would otherwise shadow the intended local:// URL selector semantics
|
||||
// (issue #4618 reviewer feedback on c493d12).
|
||||
if (explicitSelector === undefined && internalTarget.sel !== undefined) {
|
||||
explicitSelector = internalTarget.sel;
|
||||
explicitParsedSelector = parsed;
|
||||
}
|
||||
} else {
|
||||
return this.#handleInternalUrl(internalTarget.path, parsed, signal);
|
||||
}
|
||||
@@ -2205,48 +2243,68 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
// resolution share misses instead of re-globbing the workspace.
|
||||
const suffixCache: SuffixMatchCache = new Map();
|
||||
|
||||
const archivePath = await this.#resolveArchiveReadPath(readPath, suffixCache, signal);
|
||||
if (archivePath) {
|
||||
const archiveSubPath = splitPathAndSel(archivePath.archiveSubPath);
|
||||
const archiveParsed = parseSel(archiveSubPath.sel);
|
||||
return this.#readArchive(
|
||||
readPath,
|
||||
archiveParsed,
|
||||
{ ...archivePath, archiveSubPath: archiveSubPath.path },
|
||||
signal,
|
||||
);
|
||||
}
|
||||
// Prefer a literal filesystem match over selector interpretation so real
|
||||
// POSIX filenames containing selector-looking suffixes win over structured
|
||||
// archive / sqlite / pdf-image dispatch. With explicit `selector`, `path`
|
||||
// is exact: `path: "test:1-2", selector: "1-2"` means "lines 1-2 from
|
||||
// the literal file test:1-2", without recursively depending on whether a
|
||||
// longer `test:1-2:1-2` filename also exists (issue #4618).
|
||||
const literalSplit =
|
||||
explicitSelector === undefined
|
||||
? await splitPathAndSelPreferringLiteral(readPath, this.session.cwd)
|
||||
: { path: readPath, sel: explicitSelector };
|
||||
const rawPathIsLiteral =
|
||||
explicitSelector !== undefined
|
||||
? readPath.includes(":") && (await probeLiteralPathExists(readPath, this.session.cwd)) !== "missing"
|
||||
: literalSplit.sel === undefined && splitPathAndSel(readPath).sel !== undefined;
|
||||
|
||||
const sqlitePath = await this.#resolveSqliteReadPath(readPath, suffixCache, signal);
|
||||
if (sqlitePath) {
|
||||
return this.#readSqlite(sqlitePath, signal);
|
||||
}
|
||||
|
||||
const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath);
|
||||
if (pdfImageMemberPath) {
|
||||
let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd);
|
||||
let suffixResolution: { from: string; to: string } | undefined;
|
||||
try {
|
||||
const stat = await Bun.file(absolutePdfPath).stat();
|
||||
if (stat.isDirectory())
|
||||
throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`);
|
||||
} catch (error) {
|
||||
if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error;
|
||||
const suffixMatch = await this.#findSuffixMatchCached(suffixCache, pdfImageMemberPath.pdfPath, signal);
|
||||
if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`);
|
||||
absolutePdfPath = suffixMatch.absolutePath;
|
||||
suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath };
|
||||
if (!rawPathIsLiteral) {
|
||||
const archivePath = await this.#resolveArchiveReadPath(readPath, suffixCache, signal);
|
||||
if (archivePath) {
|
||||
const archiveSubPath =
|
||||
explicitSelector === undefined
|
||||
? splitPathAndSel(archivePath.archiveSubPath)
|
||||
: { path: archivePath.archiveSubPath, sel: explicitSelector };
|
||||
const archiveParsed = parseSel(archiveSubPath.sel);
|
||||
return this.#readArchive(
|
||||
readPath,
|
||||
archiveParsed,
|
||||
{ ...archivePath, archiveSubPath: archiveSubPath.path },
|
||||
signal,
|
||||
);
|
||||
}
|
||||
|
||||
const sqlitePath = await this.#resolveSqliteReadPath(readPath, suffixCache, signal);
|
||||
if (sqlitePath) {
|
||||
return this.#readSqlite(sqlitePath, signal);
|
||||
}
|
||||
|
||||
const pdfImageMemberPath = splitPdfImageMemberReadPath(readPath);
|
||||
if (pdfImageMemberPath) {
|
||||
let absolutePdfPath = resolveReadPath(pdfImageMemberPath.pdfPath, this.session.cwd);
|
||||
let suffixResolution: { from: string; to: string } | undefined;
|
||||
try {
|
||||
const stat = await Bun.file(absolutePdfPath).stat();
|
||||
if (stat.isDirectory())
|
||||
throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' is a directory, not a PDF file`);
|
||||
} catch (error) {
|
||||
if (!isNotFoundError(error) || isRemoteMountPath(absolutePdfPath)) throw error;
|
||||
const suffixMatch = await this.#findSuffixMatchCached(suffixCache, pdfImageMemberPath.pdfPath, signal);
|
||||
if (!suffixMatch) throw new ToolError(`Path '${pdfImageMemberPath.pdfPath}' not found`);
|
||||
absolutePdfPath = suffixMatch.absolutePath;
|
||||
suffixResolution = { from: pdfImageMemberPath.pdfPath, to: suffixMatch.displayPath };
|
||||
}
|
||||
return this.#readPdfImageMember(
|
||||
absolutePdfPath,
|
||||
pdfImageMemberPath.pdfPath,
|
||||
pdfImageMemberPath.member,
|
||||
suffixResolution,
|
||||
signal,
|
||||
);
|
||||
}
|
||||
return this.#readPdfImageMember(
|
||||
absolutePdfPath,
|
||||
pdfImageMemberPath.pdfPath,
|
||||
pdfImageMemberPath.member,
|
||||
suffixResolution,
|
||||
signal,
|
||||
);
|
||||
}
|
||||
|
||||
const localTarget = splitPathAndSel(readPath);
|
||||
const localTarget = literalSplit;
|
||||
const localReadPath = localTarget.path;
|
||||
const parsed = parseSel(localTarget.sel);
|
||||
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
/** formatLocalCalendarDate formats a Date as YYYY-MM-DD in the host local timezone. */
|
||||
export function formatLocalCalendarDate(date: Date = new Date()): string {
|
||||
const year = date.getFullYear();
|
||||
const month = String(date.getMonth() + 1).padStart(2, "0");
|
||||
const day = String(date.getDate()).padStart(2, "0");
|
||||
return `${year}-${month}-${day}`;
|
||||
}
|
||||
@@ -32,22 +32,48 @@ function getExistingWslLocalPath(urlOrPath: string): string | undefined {
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the Windows `rundll32.exe` command used to hand a URL/path to the
|
||||
* user's registered protocol handler. Anchoring to `%SystemRoot%\System32`
|
||||
* (rather than relying on `rundll32` being on `PATH`) survives environments
|
||||
* where the machine `PATH` no longer references `System32` — a common
|
||||
* real-world misconfiguration where `System32\Wbem` / `WindowsPowerShell` /
|
||||
* `OpenSSH` survive but `System32` itself is dropped. Bare `rundll32` on
|
||||
* such boxes throws `Executable not found in $PATH: "rundll32"` from
|
||||
* `Bun.spawn` before ShellExecute ever sees the URL.
|
||||
* Resolve the Windows opener used to hand a URL/path to the user's registered
|
||||
* protocol handler. PowerShell's `Start-Process` goes through ShellExecute
|
||||
* like the previous `rundll32 url.dll,FileProtocolHandler`, with two
|
||||
* advantages that make the delayed-failure telemetry in {@link openPath}
|
||||
* actually observable on Windows:
|
||||
*
|
||||
* - `rundll32` exits 0 unconditionally, so no launch failure ever reaches the
|
||||
* non-zero-exit logging below. `Start-Process` surfaces the failures
|
||||
* ShellExecute itself reports — missing target file, no handler executable,
|
||||
* access denied — as exit code 1 (verified live: a nonexistent file path
|
||||
* exits 1; `$ErrorActionPreference='Stop'` additionally promotes any
|
||||
* non-terminating error classes). Known limitation shared by every opener:
|
||||
* an unregistered URL scheme exits 0 because Windows "handles" it by
|
||||
* offering the app-picker.
|
||||
* - `-EncodedCommand` carries the target as a UTF-16LE/base64 payload, so no
|
||||
* cmd/PowerShell metacharacter parsing ever sees it (OAuth authorize URLs
|
||||
* carry `&`); inside the decoded script the target is a single-quoted
|
||||
* literal (no `$` expansion) with embedded quotes doubled.
|
||||
*
|
||||
* PowerShell is anchored to `%SystemRoot%\System32` for the same reason the
|
||||
* previous revision anchored `rundll32`: machine PATHs that dropped
|
||||
* `System32` are a real-world occurrence, and bare names throw
|
||||
* `Executable not found in $PATH` from `Bun.spawn`. A bare-name fallback
|
||||
* remains for exotic SystemRoot layouts.
|
||||
*/
|
||||
function windowsOpenerCommand(target: string): string[] {
|
||||
const systemRoot = process.env.SystemRoot?.trim() || process.env.SYSTEMROOT?.trim() || "C:\\Windows";
|
||||
// `path.win32` (not the platform-adaptive `path.join`) keeps Windows path
|
||||
// separators when tests run under a POSIX host and matches Windows call
|
||||
// conventions on the real target.
|
||||
const rundll32 = path.win32.join(systemRoot, "System32", "rundll32.exe");
|
||||
return [rundll32, "url.dll,FileProtocolHandler", target];
|
||||
const absolute = path.win32.join(systemRoot, "System32", "WindowsPowerShell", "v1.0", "powershell.exe");
|
||||
const powershell = fs.existsSync(absolute) ? absolute : "powershell.exe";
|
||||
const script = `$ErrorActionPreference='Stop';Start-Process '${target.replaceAll("'", "''")}'`;
|
||||
return [
|
||||
powershell,
|
||||
"-NoProfile",
|
||||
"-NonInteractive",
|
||||
"-WindowStyle",
|
||||
"Hidden",
|
||||
"-EncodedCommand",
|
||||
Buffer.from(script, "utf16le").toString("base64"),
|
||||
];
|
||||
}
|
||||
/** Open a URL or file path in the default browser/application. Best-effort, never throws. */
|
||||
export function openPath(urlOrPath: string): void {
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import { Agent, type AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Model } from "@oh-my-pi/pi-ai";
|
||||
@@ -6,9 +7,10 @@ import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { getProjectAgentDir, TempDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
describe("AgentSession advisor toggle", () => {
|
||||
let sharedDir: TempDir;
|
||||
@@ -22,6 +24,7 @@ describe("AgentSession advisor toggle", () => {
|
||||
authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db"));
|
||||
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
||||
authStorage.setRuntimeApiKey("openai", "test-key");
|
||||
authStorage.setRuntimeApiKey("openrouter", "test-key");
|
||||
modelRegistry = new ModelRegistry(authStorage);
|
||||
const bundled = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
const replacement = getBundledModel("openai", "gpt-4o-mini");
|
||||
@@ -98,6 +101,89 @@ describe("AgentSession advisor toggle", () => {
|
||||
expect(session.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id);
|
||||
});
|
||||
|
||||
it("refreshes the live advisor when the advisor role setting changes", () => {
|
||||
session.settings.setModelRole("advisor", `${model.provider}/${model.id}`);
|
||||
expect(session.setAdvisorEnabled(true)).toBe(true);
|
||||
expect(session.getAdvisorAgent()?.state.model.provider).toBe(model.provider);
|
||||
expect(session.getAdvisorAgent()?.state.model.id).toBe(model.id);
|
||||
|
||||
session.settings.setModelRole("advisor", `${replacementModel.provider}/${replacementModel.id}`);
|
||||
|
||||
expect(session.getAdvisorAgent()?.state.model.provider).toBe(replacementModel.provider);
|
||||
expect(session.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id);
|
||||
});
|
||||
|
||||
it("refreshes the live advisor when only the advisor route changes", () => {
|
||||
session.settings.setModelRole("advisor", "openrouter/z-ai/glm-4.7@cerebras");
|
||||
expect(session.setAdvisorEnabled(true)).toBe(true);
|
||||
expect(session.getAdvisorAgent()?.state.model.provider).toBe("openrouter");
|
||||
expect(session.getAdvisorAgent()?.state.model.id).toBe("z-ai/glm-4.7");
|
||||
expect(
|
||||
(session.getAdvisorAgent()?.state.model.compat as { openRouterRouting?: { only?: string[] } } | undefined)
|
||||
?.openRouterRouting?.only,
|
||||
).toEqual(["cerebras"]);
|
||||
|
||||
session.settings.setModelRole("advisor", "openrouter/z-ai/glm-4.7@fireworks");
|
||||
|
||||
expect(session.getAdvisorAgent()?.state.model.provider).toBe("openrouter");
|
||||
expect(session.getAdvisorAgent()?.state.model.id).toBe("z-ai/glm-4.7");
|
||||
expect(
|
||||
(session.getAdvisorAgent()?.state.model.compat as { openRouterRouting?: { only?: string[] } } | undefined)
|
||||
?.openRouterRouting?.only,
|
||||
).toEqual(["fireworks"]);
|
||||
});
|
||||
|
||||
it("refreshes the live advisor after project model-role reloads", async () => {
|
||||
const projectA = path.join(tempDir.path(), "project-a");
|
||||
const projectB = path.join(tempDir.path(), "project-b");
|
||||
const agentDir = path.join(tempDir.path(), "agent");
|
||||
fs.mkdirSync(getProjectAgentDir(projectA), { recursive: true });
|
||||
fs.mkdirSync(getProjectAgentDir(projectB), { recursive: true });
|
||||
fs.mkdirSync(agentDir, { recursive: true });
|
||||
await Bun.write(
|
||||
path.join(getProjectAgentDir(projectA), "settings.json"),
|
||||
JSON.stringify({ modelRoles: { advisor: `${model.provider}/${model.id}` } }),
|
||||
);
|
||||
await Bun.write(
|
||||
path.join(getProjectAgentDir(projectB), "settings.json"),
|
||||
JSON.stringify({ modelRoles: { advisor: `${replacementModel.provider}/${replacementModel.id}` } }),
|
||||
);
|
||||
|
||||
const settings = await Settings.loadIsolated({
|
||||
cwd: projectA,
|
||||
agentDir,
|
||||
overrides: { "compaction.enabled": false },
|
||||
});
|
||||
const customSession = new AgentSession({
|
||||
agent: new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
}),
|
||||
sessionManager: SessionManager.create(tempDir.path(), tempDir.path()),
|
||||
settings,
|
||||
modelRegistry,
|
||||
advisorTools: [],
|
||||
});
|
||||
|
||||
try {
|
||||
expect(customSession.setAdvisorEnabled(true)).toBe(true);
|
||||
expect(customSession.getAdvisorAgent()?.state.model.provider).toBe(model.provider);
|
||||
expect(customSession.getAdvisorAgent()?.state.model.id).toBe(model.id);
|
||||
|
||||
await settings.reloadForCwd(projectB);
|
||||
|
||||
expect(customSession.getAdvisorAgent()?.state.model.provider).toBe(replacementModel.provider);
|
||||
expect(customSession.getAdvisorAgent()?.state.model.id).toBe(replacementModel.id);
|
||||
} finally {
|
||||
await customSession.dispose();
|
||||
AgentStorage.resetInstance();
|
||||
}
|
||||
});
|
||||
|
||||
it("keeps explicit enable idempotent when the advisor config is unchanged", () => {
|
||||
session.settings.setModelRole("advisor", `${model.provider}/${model.id}`);
|
||||
expect(session.setAdvisorEnabled(true)).toBe(true);
|
||||
|
||||
@@ -1235,6 +1235,126 @@ describe("AgentSession TTSR resume gate", () => {
|
||||
expect(text).not.toContain("Request was aborted");
|
||||
});
|
||||
|
||||
it("labels only the matching aborted tool placeholder with the TTSR rule reason", async () => {
|
||||
collapseSchedulerSettleDelays();
|
||||
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
|
||||
let streamCallCount = 0;
|
||||
|
||||
const ttsrManager = new TtsrManager({
|
||||
enabled: true,
|
||||
contextMode: "discard",
|
||||
interruptMode: "always",
|
||||
repeatMode: "once",
|
||||
repeatGap: 10,
|
||||
});
|
||||
ttsrManager.addRule(testRule);
|
||||
|
||||
const readToolCallContent: ToolCall = {
|
||||
type: "toolCall",
|
||||
id: "call_innocent_read",
|
||||
name: "read",
|
||||
arguments: { path: "history://Eval1WithSkill" },
|
||||
};
|
||||
const matchedToolCallContent: ToolCall = {
|
||||
type: "toolCall",
|
||||
id: "call_ttsr_abort_reason",
|
||||
name: "mock_edit",
|
||||
arguments: { snippet: "let val = result.unwrap(" },
|
||||
};
|
||||
|
||||
const makeToolCallMsg = (stopReason: "toolUse" | "aborted" = "toolUse"): AssistantMessage => ({
|
||||
role: "assistant",
|
||||
content: [readToolCallContent, matchedToolCallContent],
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
model: "mock",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason,
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
|
||||
const agent = new Agent({
|
||||
getApiKey: () => "test-key",
|
||||
initialState: { model, systemPrompt: ["Test"], tools: [] },
|
||||
streamFn: (_model, _context, options) => {
|
||||
streamCallCount++;
|
||||
const stream = new AssistantMessageEventStream();
|
||||
const signal = options?.signal;
|
||||
if (streamCallCount === 1) {
|
||||
queueMicrotask(() => {
|
||||
const partial = makeToolCallMsg();
|
||||
if (signal) {
|
||||
signal.addEventListener(
|
||||
"abort",
|
||||
() => {
|
||||
stream.push({
|
||||
type: "error",
|
||||
reason: "aborted",
|
||||
error: makeToolCallMsg("aborted"),
|
||||
});
|
||||
},
|
||||
{ once: true },
|
||||
);
|
||||
}
|
||||
stream.push({ type: "start", partial });
|
||||
stream.push({ type: "toolcall_start", contentIndex: 1, partial });
|
||||
stream.push({
|
||||
type: "toolcall_delta",
|
||||
contentIndex: 1,
|
||||
delta: 'let val = result.unwrap("oops")',
|
||||
partial,
|
||||
});
|
||||
// The abort placeholder is only minted for tool calls that reached
|
||||
// `toolcall_end`: the agent loop drops incomplete tool calls from an
|
||||
// aborted turn (partial args are unsafe to replay). Complete the
|
||||
// innocent read before the rule-driven abort fires so its placeholder
|
||||
// survives and can carry the neutral sibling label.
|
||||
stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: readToolCallContent, partial });
|
||||
});
|
||||
} else {
|
||||
pushContinuationStream(stream, () => {});
|
||||
}
|
||||
return stream;
|
||||
},
|
||||
});
|
||||
|
||||
const sessionManager = SessionManager.inMemory();
|
||||
const settings = Settings.isolated();
|
||||
const authStorage = await AuthStorage.create(path.join(tempDir, "testauth-abort-reason.db"));
|
||||
authStorages.push(authStorage);
|
||||
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml"));
|
||||
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
||||
session = new AgentSession({ agent, sessionManager, settings, modelRegistry, ttsrManager });
|
||||
|
||||
await session.prompt("Write some Rust code");
|
||||
|
||||
const toolResults = sessionManager
|
||||
.getEntries()
|
||||
.filter(entry => entry.type === "message" && entry.message.role === "toolResult")
|
||||
.map(entry => (entry.type === "message" && entry.message.role === "toolResult" ? entry.message : undefined))
|
||||
.filter(message => message !== undefined);
|
||||
const toolResultText = (toolCallId: string): string =>
|
||||
toolResults
|
||||
.find(message => message.toolCallId === toolCallId)
|
||||
?.content.find((part): part is { type: "text"; text: string } => part.type === "text")?.text ?? "";
|
||||
|
||||
const readText = toolResultText(readToolCallContent.id);
|
||||
expect(readText).toContain("Tool execution was aborted: TTSR interrupt on another tool call");
|
||||
expect(readText).not.toContain("TTSR matched rule: no-unwrap");
|
||||
// The matching call never reached `toolcall_end`, so the loop drops it from
|
||||
// the aborted turn (partial args are unsafe to replay) and no placeholder is
|
||||
// minted. The rule label for a completed matching call is covered by the
|
||||
// single-call test above.
|
||||
expect(toolResultText(matchedToolCallContent.id)).toBe("");
|
||||
});
|
||||
|
||||
it("relativizes the rule file path in the TTSR interrupt injection (no absolute leak)", async () => {
|
||||
collapseSchedulerSettleDelays();
|
||||
const model = getBundledModel("anthropic", "claude-sonnet-4-5")!;
|
||||
|
||||
@@ -158,6 +158,90 @@ describe("AgentSession handoff", () => {
|
||||
expect(sessionManager.getEntries().filter(entry => entry.type === "compaction")).toHaveLength(0);
|
||||
});
|
||||
|
||||
it("emits handoff lifecycle hooks on the outgoing and replacement sessions", async () => {
|
||||
const extensionsResult = await loadExtensions([], tempDir.path());
|
||||
const extensionRunner = new ExtensionRunner(
|
||||
extensionsResult.extensions,
|
||||
extensionsResult.runtime,
|
||||
tempDir.path(),
|
||||
sessionManager,
|
||||
modelRegistry,
|
||||
);
|
||||
const observedEvents: Array<{
|
||||
type: "session_before_switch" | "session_switch";
|
||||
reason: string;
|
||||
previousSessionFile: string | undefined;
|
||||
activeSessionFile: string | undefined;
|
||||
messageCount: number;
|
||||
handoffEntryCount: number;
|
||||
}> = [];
|
||||
vi.spyOn(extensionRunner, "hasHandlers").mockImplementation(eventName => eventName === "session_before_switch");
|
||||
const emit = extensionRunner.emit.bind(extensionRunner);
|
||||
vi.spyOn(extensionRunner, "emit").mockImplementation(event => {
|
||||
if (event.type === "session_before_switch" || event.type === "session_switch") {
|
||||
observedEvents.push({
|
||||
type: event.type,
|
||||
reason: event.reason,
|
||||
previousSessionFile: event.type === "session_switch" ? event.previousSessionFile : undefined,
|
||||
activeSessionFile: session.sessionFile,
|
||||
messageCount: sessionManager.getBranch().filter(entry => entry.type === "message").length,
|
||||
handoffEntryCount: sessionManager
|
||||
.getBranch()
|
||||
.filter(entry => entry.type === "custom_message" && entry.customType === "handoff").length,
|
||||
});
|
||||
}
|
||||
return emit(event);
|
||||
});
|
||||
|
||||
await session.dispose();
|
||||
session = new AgentSession({
|
||||
agent: new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
}),
|
||||
sessionManager,
|
||||
settings: Settings.isolated({
|
||||
"compaction.enabled": true,
|
||||
"compaction.autoContinue": false,
|
||||
}),
|
||||
modelRegistry,
|
||||
extensionRunner,
|
||||
obfuscator,
|
||||
});
|
||||
const previousSessionFile = session.sessionFile;
|
||||
const generateHandoffSpy = vi
|
||||
.spyOn(compactionModule, "generateHandoffFromContext")
|
||||
.mockResolvedValue("## Goal\nContinue from here");
|
||||
|
||||
await session.handoff();
|
||||
|
||||
const nextSessionFile = session.sessionFile;
|
||||
expect(generateHandoffSpy).toHaveBeenCalledTimes(1);
|
||||
expect(nextSessionFile).not.toBe(previousSessionFile);
|
||||
expect(observedEvents).toEqual([
|
||||
{
|
||||
type: "session_before_switch",
|
||||
reason: "handoff",
|
||||
previousSessionFile: undefined,
|
||||
activeSessionFile: previousSessionFile,
|
||||
messageCount: 2,
|
||||
handoffEntryCount: 0,
|
||||
},
|
||||
{
|
||||
type: "session_switch",
|
||||
reason: "handoff",
|
||||
previousSessionFile,
|
||||
activeSessionFile: nextSessionFile,
|
||||
messageCount: 0,
|
||||
handoffEntryCount: 1,
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("runs handoff generation through the configured side stream function", async () => {
|
||||
const handoffText = "## Goal\nContinue via side stream";
|
||||
let sideStreamCalls = 0;
|
||||
@@ -1336,6 +1420,7 @@ describe("AgentSession handoff", () => {
|
||||
expect(handoffSpy).toHaveBeenCalledWith(expect.stringContaining("Threshold-triggered maintenance"), {
|
||||
autoTriggered: true,
|
||||
signal: expect.anything(),
|
||||
onSwitchCancelled: expect.any(Function),
|
||||
});
|
||||
expect(events.filter(event => event.type === "auto_compaction_start")).toHaveLength(1);
|
||||
const endEvents = events.filter(event => event.type === "auto_compaction_end");
|
||||
@@ -1598,6 +1683,89 @@ describe("AgentSession handoff", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("treats a vetoed auto-handoff switch as cancelled instead of falling back", async () => {
|
||||
session.settings.set("compaction.strategy", "handoff");
|
||||
session.settings.set("compaction.thresholdPercent", 1);
|
||||
session.settings.set("contextPromotion.enabled", false);
|
||||
|
||||
const model = session.model;
|
||||
if (!model) {
|
||||
throw new Error("Expected model to be set");
|
||||
}
|
||||
|
||||
const extensionsResult = await loadExtensions([], tempDir.path());
|
||||
const extensionRunner = new ExtensionRunner(
|
||||
extensionsResult.extensions,
|
||||
extensionsResult.runtime,
|
||||
tempDir.path(),
|
||||
sessionManager,
|
||||
modelRegistry,
|
||||
);
|
||||
vi.spyOn(extensionRunner, "hasHandlers").mockImplementation(eventName => eventName === "session_before_switch");
|
||||
const emitSpy = vi.spyOn(extensionRunner, "emit").mockImplementation((async () => ({
|
||||
cancel: true,
|
||||
})) as ExtensionRunner["emit"]);
|
||||
|
||||
await session.dispose();
|
||||
session = new AgentSession({
|
||||
agent: new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
}),
|
||||
sessionManager,
|
||||
settings: session.settings,
|
||||
modelRegistry,
|
||||
extensionRunner,
|
||||
obfuscator,
|
||||
});
|
||||
session.subscribe(event => {
|
||||
events.push(event);
|
||||
});
|
||||
const previousSessionFile = session.sessionFile;
|
||||
const generateHandoffSpy = vi
|
||||
.spyOn(compactionModule, "generateHandoffFromContext")
|
||||
.mockResolvedValue("## Goal\nContinue from here");
|
||||
const assistantMessage: AssistantMessage = {
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text: "maintenance trigger" }],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
stopReason: "stop",
|
||||
usage: {
|
||||
input: 10_000,
|
||||
output: 1_000,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 11_000,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: assistantMessage });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMessage] });
|
||||
await waitFor(() => events.filter(event => event.type === "auto_compaction_end").length === 1);
|
||||
|
||||
expect(generateHandoffSpy).toHaveBeenCalledTimes(1);
|
||||
expect(emitSpy).toHaveBeenCalledWith({ type: "session_before_switch", reason: "handoff" });
|
||||
expect(emitSpy).not.toHaveBeenCalledWith(expect.objectContaining({ type: "session_switch" }));
|
||||
expect(session.sessionFile).toBe(previousSessionFile);
|
||||
expect(sessionManager.getEntries().filter(entry => entry.type === "compaction")).toHaveLength(0);
|
||||
const endEvents = events.filter(event => event.type === "auto_compaction_end");
|
||||
expect(endEvents).toHaveLength(1);
|
||||
expect(endEvents[0]).toMatchObject({
|
||||
type: "auto_compaction_end",
|
||||
action: "handoff",
|
||||
aborted: true,
|
||||
willRetry: false,
|
||||
});
|
||||
});
|
||||
|
||||
it("resets to the base system prompt before generating a handoff", async () => {
|
||||
const model = session.model;
|
||||
if (!model) {
|
||||
|
||||
@@ -2,7 +2,7 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { Agent } from "@oh-my-pi/pi-agent-core";
|
||||
import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import { Effort } from "@oh-my-pi/pi-ai";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import * as autoThinkingClassifier from "@oh-my-pi/pi-coding-agent/auto-thinking/classifier";
|
||||
@@ -13,8 +13,20 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { AUTO_THINKING } from "@oh-my-pi/pi-coding-agent/thinking";
|
||||
import { removeWithRetries } from "@oh-my-pi/pi-utils";
|
||||
import { type } from "arktype";
|
||||
|
||||
async function createMagicKeywordSession(root: string): Promise<{
|
||||
const mockTaskTool: AgentTool = {
|
||||
name: "task",
|
||||
label: "Task",
|
||||
description: "Mock task tool",
|
||||
parameters: type({}),
|
||||
execute: async () => ({ content: [{ type: "text" as const, text: "ok" }] }),
|
||||
};
|
||||
|
||||
async function createMagicKeywordSession(
|
||||
root: string,
|
||||
tools: AgentTool[] = [mockTaskTool],
|
||||
): Promise<{
|
||||
session: AgentSession;
|
||||
settings: Settings;
|
||||
authStorage: AuthStorage;
|
||||
@@ -25,7 +37,7 @@ async function createMagicKeywordSession(root: string): Promise<{
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
tools,
|
||||
messages: [],
|
||||
thinkingLevel: Effort.High,
|
||||
},
|
||||
@@ -103,6 +115,34 @@ describe("AgentSession magic keyword settings", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("renders workflowz notice for the active task schema", async () => {
|
||||
const created = await createMagicKeywordSession(root);
|
||||
session = created.session;
|
||||
authStorage = created.authStorage;
|
||||
created.settings.set("task.batch", false);
|
||||
const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
||||
|
||||
await session.prompt("please workflowz this");
|
||||
|
||||
const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ content?: string; customType?: string }>;
|
||||
const notice = promptMessages.find(message => message.customType === "workflow-notice")?.content ?? "";
|
||||
expect(notice).toContain("once per independent subagent");
|
||||
expect(notice).toContain("Do not pass `context` or `tasks[]`");
|
||||
expect(notice).not.toContain("Call `task` once per independent fan-out batch");
|
||||
});
|
||||
|
||||
it("skips workflowz notice when the task tool is inactive", async () => {
|
||||
const created = await createMagicKeywordSession(root, []);
|
||||
session = created.session;
|
||||
authStorage = created.authStorage;
|
||||
const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
||||
|
||||
await session.prompt("please workflowz this");
|
||||
|
||||
const promptMessages = promptSpy.mock.calls[0]![0] as unknown as Array<{ customType?: string }>;
|
||||
expect(promptMessages.map(message => message.customType).filter(Boolean)).toEqual([]);
|
||||
});
|
||||
|
||||
it("does not use a disabled ultrathink keyword to force auto thinking", async () => {
|
||||
const created = await createMagicKeywordSession(root);
|
||||
session = created.session;
|
||||
|
||||
@@ -282,6 +282,79 @@ describe("AgentSession retry delay cap", () => {
|
||||
expect(last.content).toContainEqual({ type: "text", text: "recovered after credential switch" });
|
||||
});
|
||||
|
||||
it("switches same-provider credentials before model fallback on ChatGPT usage limits", async () => {
|
||||
const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
const fallbackModel = getBundledModel("openai", "gpt-5.5");
|
||||
if (!primaryModel || !fallbackModel) {
|
||||
throw new Error("Expected bundled primary and fallback test models to exist");
|
||||
}
|
||||
|
||||
authStorage.removeRuntimeApiKey("anthropic");
|
||||
authStorage.setRuntimeApiKey("openai", "openai-fallback-key");
|
||||
await authStorage.set("anthropic", [
|
||||
{ type: "api_key", key: "anthropic-key-1" },
|
||||
{ type: "api_key", key: "anthropic-key-2" },
|
||||
]);
|
||||
|
||||
const usageLimitError = "Error: You have hit your ChatGPT usage limit (k12 plan). Try again in ~231 min.";
|
||||
const mock = createMockModel();
|
||||
const requestedModels: string[] = [];
|
||||
const requestedKeys: string[] = [];
|
||||
let agent!: Agent;
|
||||
agent = new Agent({
|
||||
getApiKey: model => modelRegistry.resolver(model, agent.sessionId),
|
||||
initialState: {
|
||||
model: primaryModel,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
streamFn: (requestedModel, context, options) => {
|
||||
requestedModels.push(`${requestedModel.provider}/${requestedModel.id}`);
|
||||
const apiKey = resolveInitialApiKey(options?.apiKey);
|
||||
requestedKeys.push(apiKey);
|
||||
if (requestedKeys.length === 1) {
|
||||
mock.push({ throw: usageLimitError });
|
||||
} else {
|
||||
mock.push({ content: ["recovered after sibling account"] });
|
||||
}
|
||||
return mock.stream(requestedModel, context, options);
|
||||
},
|
||||
});
|
||||
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
"retry.baseDelayMs": 5,
|
||||
"retry.maxDelayMs": 100,
|
||||
"retry.maxRetries": 1,
|
||||
"retry.modelFallback": true,
|
||||
"retry.fallbackChains": {
|
||||
default: [`${fallbackModel.provider}/${fallbackModel.id}`],
|
||||
},
|
||||
});
|
||||
settings.setModelRole("default", `${primaryModel.provider}/${primaryModel.id}`);
|
||||
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
modelRegistry,
|
||||
});
|
||||
|
||||
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
|
||||
await session.prompt("Trigger k12 usage limit");
|
||||
await session.waitForIdle();
|
||||
|
||||
expect(requestedModels).toEqual([
|
||||
`${primaryModel.provider}/${primaryModel.id}`,
|
||||
`${primaryModel.provider}/${primaryModel.id}`,
|
||||
]);
|
||||
expect([...requestedKeys].sort()).toEqual(["anthropic-key-1", "anthropic-key-2"]);
|
||||
const last = lastAssistant(session);
|
||||
expect(last.stopReason).toBe("stop");
|
||||
expect(last.content).toContainEqual({ type: "text", text: "recovered after sibling account" });
|
||||
});
|
||||
|
||||
it("waits for the earliest sibling unblock instead of failing the delay cap", async () => {
|
||||
// Regression: with every sibling credential momentarily blocked (e.g. a
|
||||
// short post-401 or usage-probe block), a usage-limit 429 with a
|
||||
|
||||
@@ -219,6 +219,59 @@ describe("AgentSession retry fallback", () => {
|
||||
]);
|
||||
});
|
||||
|
||||
it("uses the active initial model as the default fallback primary when other role fallback chains are configured", async () => {
|
||||
const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
const fallbackModel = getBundledModel("openai", "gpt-4o-mini");
|
||||
const otherRoleFallbackModel = getBundledModel("openai", "gpt-4o");
|
||||
if (!primaryModel || !fallbackModel || !otherRoleFallbackModel) {
|
||||
throw new Error("Expected bundled test models to exist");
|
||||
}
|
||||
|
||||
const requestedModels: string[] = [];
|
||||
const fallbackAppliedEvents: Array<Extract<AgentSessionEvent, { type: "retry_fallback_applied" }>> = [];
|
||||
const agent = createFallbackAgent(primaryModel, requestedModels);
|
||||
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
"retry.maxRetries": 1,
|
||||
"retry.fallbackChains": {
|
||||
default: [`${fallbackModel.provider}/${fallbackModel.id}`],
|
||||
smol: [`${otherRoleFallbackModel.provider}/${otherRoleFallbackModel.id}`],
|
||||
},
|
||||
});
|
||||
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
modelRegistry,
|
||||
});
|
||||
|
||||
session.subscribe(event => {
|
||||
if (event.type === "retry_fallback_applied") {
|
||||
fallbackAppliedEvents.push(event);
|
||||
}
|
||||
});
|
||||
|
||||
await session.prompt("Recover using implicit default primary");
|
||||
await session.waitForIdle();
|
||||
|
||||
expect(requestedModels).toEqual([
|
||||
`${primaryModel.provider}/${primaryModel.id}`,
|
||||
`${fallbackModel.provider}/${fallbackModel.id}`,
|
||||
]);
|
||||
expect(session.model?.provider).toBe(fallbackModel.provider);
|
||||
expect(session.model?.id).toBe(fallbackModel.id);
|
||||
expect(fallbackAppliedEvents).toEqual([
|
||||
{
|
||||
type: "retry_fallback_applied",
|
||||
from: `${primaryModel.provider}/${primaryModel.id}`,
|
||||
to: `${fallbackModel.provider}/${fallbackModel.id}`,
|
||||
role: "default",
|
||||
},
|
||||
]);
|
||||
});
|
||||
|
||||
it("falls back on structured classifier refusals and pins the fallback", async () => {
|
||||
const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
const fallbackModel = getBundledModel("openai", "gpt-4o-mini");
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { afterEach, describe, expect, it, setSystemTime } from "bun:test";
|
||||
import { afterEach, describe, expect, it } from "bun:test";
|
||||
import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Model } from "@oh-my-pi/pi-ai";
|
||||
import { buildModel } from "@oh-my-pi/pi-catalog/build";
|
||||
@@ -68,6 +68,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => {
|
||||
interface NewSessionOptions {
|
||||
mcpDiscoveryEnabled?: boolean;
|
||||
getMcpServerInstructions?: () => Map<string, string> | undefined;
|
||||
getLocalCalendarDate?: () => string;
|
||||
}
|
||||
|
||||
function newSession(
|
||||
@@ -101,6 +102,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => {
|
||||
}),
|
||||
mcpDiscoveryEnabled: options.mcpDiscoveryEnabled,
|
||||
getMcpServerInstructions: options.getMcpServerInstructions,
|
||||
getLocalCalendarDate: options.getLocalCalendarDate,
|
||||
});
|
||||
sessions.push(session);
|
||||
return { session };
|
||||
@@ -176,6 +178,52 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => {
|
||||
expect(rebuildCount).toBe(baseline + 2);
|
||||
});
|
||||
|
||||
it("updates live active-tool predicates before rebuilding the prompt", async () => {
|
||||
const activeToolNames = new Set(["read", "bash", "grep"]);
|
||||
const readTool = createBasicTool("read", "Read");
|
||||
const bashTool = createBasicTool("bash", "Bash");
|
||||
const grepTool = createBasicTool("grep", "Grep");
|
||||
Object.defineProperty(bashTool, "description", {
|
||||
get: () => (activeToolNames.has("grep") ? "bash sees grep" : "bash hides grep"),
|
||||
enumerable: true,
|
||||
configurable: true,
|
||||
});
|
||||
const toolRegistry = new Map<string, AgentTool>([
|
||||
[readTool.name, readTool],
|
||||
[bashTool.name, bashTool],
|
||||
[grepTool.name, grepTool],
|
||||
]);
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model: createModel(),
|
||||
systemPrompt: ["initial"],
|
||||
tools: [readTool, bashTool, grepTool],
|
||||
messages: [],
|
||||
},
|
||||
});
|
||||
const session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry: {} as never,
|
||||
toolRegistry,
|
||||
setActiveToolNames: names => {
|
||||
activeToolNames.clear();
|
||||
for (const name of names) {
|
||||
activeToolNames.add(name);
|
||||
}
|
||||
},
|
||||
rebuildSystemPrompt: async (_toolNames, tools) => ({
|
||||
systemPrompt: [tools.get("bash")?.description ?? "missing bash"],
|
||||
}),
|
||||
});
|
||||
sessions.push(session);
|
||||
|
||||
await session.setActiveToolsByName(["read", "bash"]);
|
||||
|
||||
expect(agent.state.systemPrompt).toEqual(["bash hides grep"]);
|
||||
});
|
||||
|
||||
it("does not skip when refreshBaseSystemPrompt is called explicitly", async () => {
|
||||
let rebuildCount = 0;
|
||||
const { session } = newSession(async toolNames => {
|
||||
@@ -388,40 +436,38 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => {
|
||||
await session.refreshMCPTools([dynamicTool]);
|
||||
expect(rebuildCount).toBe(baseline + 1);
|
||||
});
|
||||
it("rebuilds when the calendar date rolls over between tool-stable MCP refreshes", async () => {
|
||||
// `buildSystemPrompt` injects today's date into the prompt body.
|
||||
// A session spanning midnight must not serve yesterday's date after an MCP
|
||||
// reconnect that happens to bring an identical tool set.
|
||||
setSystemTime(new Date("2025-01-01T23:59:58Z"));
|
||||
try {
|
||||
let rebuildCount = 0;
|
||||
const { session } = newSession(async toolNames => {
|
||||
it("rebuilds when the local calendar date rolls over between tool-stable MCP refreshes", async () => {
|
||||
// `buildSystemPrompt` injects today's local date into the prompt body. The
|
||||
// signature reads the same date provider so a session spanning local midnight
|
||||
// must rebuild after an MCP reconnect with an otherwise identical tool set.
|
||||
let currentDate = "2026-06-30";
|
||||
let rebuildCount = 0;
|
||||
const { session } = newSession(
|
||||
async toolNames => {
|
||||
rebuildCount++;
|
||||
return `tools:${toolNames.join(",")}`;
|
||||
});
|
||||
const tool = createMcpCustomTool("mcp__nucleus_search", "nucleus", "search", "Search");
|
||||
},
|
||||
{ getLocalCalendarDate: () => currentDate },
|
||||
);
|
||||
const tool = createMcpCustomTool("mcp__nucleus_search", "nucleus", "search", "Search");
|
||||
|
||||
// First refresh: no signature yet, must rebuild.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(1);
|
||||
// First refresh: no signature yet, must rebuild.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(1);
|
||||
|
||||
// Same tools, same day: signature matches, skip.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(1);
|
||||
// Same tools, same local day: signature matches, skip.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(1);
|
||||
|
||||
// Advance past midnight.
|
||||
setSystemTime(new Date("2025-01-02T00:00:01Z"));
|
||||
currentDate = "2026-07-01";
|
||||
|
||||
// Same tools, new calendar day: date segment changed, must rebuild.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(2);
|
||||
// Same tools, new local calendar day: date segment changed, must rebuild.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(2);
|
||||
|
||||
// Same tools, same new day: skip again.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(2);
|
||||
} finally {
|
||||
setSystemTime(); // restore real time
|
||||
}
|
||||
// Same tools, same new local day: skip again.
|
||||
await session.refreshMCPTools([tool]);
|
||||
expect(rebuildCount).toBe(2);
|
||||
});
|
||||
it("does not rebuild when MCP server instructions change only beyond the 4000-char truncation boundary", async () => {
|
||||
// `rebuildSystemPrompt` (sdk.ts) truncates each server instruction to 4000 chars
|
||||
|
||||
@@ -95,4 +95,54 @@ describe("AuthStorage account rotation", () => {
|
||||
const exhaustedFallbackKey = await authStorage.getApiKey("openai-codex", sessionId);
|
||||
expect(exhaustedFallbackKey).toMatch(/^api-acct-/);
|
||||
});
|
||||
|
||||
test("usage-limit rotation can match the failed bearer when session stickiness is missing", async () => {
|
||||
await authStorage.set("openai-codex", [
|
||||
{
|
||||
type: "oauth",
|
||||
access: "access-1",
|
||||
refresh: "refresh-1",
|
||||
expires: Date.now() + 60_000,
|
||||
accountId: "acct-1",
|
||||
},
|
||||
{
|
||||
type: "oauth",
|
||||
access: "access-2",
|
||||
refresh: "refresh-2",
|
||||
expires: Date.now() + 60_000,
|
||||
accountId: "acct-2",
|
||||
},
|
||||
]);
|
||||
|
||||
const sessionId = "missing-sticky-session";
|
||||
const result = await authStorage.markUsageLimitReached("openai-codex", sessionId, { apiKey: "access-1" });
|
||||
expect(result.switched).toBe(true);
|
||||
expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe("api-acct-2");
|
||||
});
|
||||
|
||||
test("usage-limit rotation trusts the failed bearer over stale session stickiness", async () => {
|
||||
await authStorage.set("openai-codex", [
|
||||
{
|
||||
type: "oauth",
|
||||
access: "plus-access",
|
||||
refresh: "plus-refresh",
|
||||
expires: Date.now() + 60_000,
|
||||
accountId: "plus-acct",
|
||||
},
|
||||
{
|
||||
type: "oauth",
|
||||
access: "k12-access",
|
||||
refresh: "k12-refresh",
|
||||
expires: Date.now() + 60_000,
|
||||
accountId: "k12-acct",
|
||||
},
|
||||
]);
|
||||
|
||||
const sessionId = "stale-sticky-session";
|
||||
const stickyKey = await authStorage.getApiKey("openai-codex", sessionId);
|
||||
const failedKey = stickyKey === "api-plus-acct" ? "k12-access" : "plus-access";
|
||||
const result = await authStorage.markUsageLimitReached("openai-codex", sessionId, { apiKey: failedKey });
|
||||
expect(result.switched).toBe(true);
|
||||
expect(await authStorage.getApiKey("openai-codex", sessionId)).toBe(stickyKey);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -402,6 +402,15 @@ exit 64
|
||||
expect(result.output).not.toContain("done");
|
||||
});
|
||||
|
||||
it("does not arm a deadline when timeout is zero", async () => {
|
||||
if (process.platform === "win32") {
|
||||
return;
|
||||
}
|
||||
const result = await executeBash("sleep 1.2; echo done", { cwd: tempDir, timeout: 0 });
|
||||
expect(result.cancelled).toBe(false);
|
||||
expect(result.output.trim()).toBe("done");
|
||||
});
|
||||
|
||||
it("aborts commands", async () => {
|
||||
if (process.platform === "win32") {
|
||||
return;
|
||||
|
||||
@@ -1,11 +1,9 @@
|
||||
/**
|
||||
* Regression test for #1266:
|
||||
* Regression tests for top-level `RULES.md` sticky rules.
|
||||
*
|
||||
* `RULES.md` (singular, top-level) MUST be loaded as a sticky always-apply rule
|
||||
* from both `~/.omp/agent/RULES.md` (user) and the nearest `.omp/RULES.md`
|
||||
* (project, walked up from cwd to repoRoot).
|
||||
*
|
||||
* Calls the native provider's `load` directly with the agent dir pointed at a
|
||||
* tempdir (via setAgentDir) so the user scope can be staged in isolation.
|
||||
*/
|
||||
import { afterEach, beforeEach, expect, test } from "bun:test";
|
||||
import * as fs from "node:fs";
|
||||
@@ -15,8 +13,8 @@ import { getCapability } from "@oh-my-pi/pi-coding-agent/capability";
|
||||
import { clearCache } from "@oh-my-pi/pi-coding-agent/capability/fs";
|
||||
import { type Rule, ruleCapability } from "@oh-my-pi/pi-coding-agent/capability/rule";
|
||||
import type { LoadContext } from "@oh-my-pi/pi-coding-agent/capability/types";
|
||||
// Register all discovery providers as a side effect.
|
||||
import "@oh-my-pi/pi-coding-agent/discovery";
|
||||
// Importing discovery registers all providers as a side effect.
|
||||
import { loadCapability } from "@oh-my-pi/pi-coding-agent/discovery";
|
||||
import { getConfigRootDir, removeSyncWithRetries, setAgentDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
let tempDir: string;
|
||||
@@ -40,6 +38,11 @@ async function loadNativeRules(ctx: LoadContext): Promise<Rule[]> {
|
||||
return result.items;
|
||||
}
|
||||
|
||||
async function loadRulesCapability(cwd: string): Promise<Rule[]> {
|
||||
const result = await loadCapability<Rule>(ruleCapability.id, { cwd, providers: ["native"] });
|
||||
return result.items;
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
clearCache();
|
||||
tempDir = fs.mkdtempSync(path.join(os.tmpdir(), "pi-rules-md-"));
|
||||
@@ -81,7 +84,7 @@ test("project .omp/RULES.md becomes an alwaysApply rule", async () => {
|
||||
|
||||
const rules = await loadNativeRules({ cwd: project, home, repoRoot: project });
|
||||
|
||||
const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES");
|
||||
const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES@project");
|
||||
expect(projectRule).toBeDefined();
|
||||
expect(projectRule?.alwaysApply).toBe(true);
|
||||
expect(projectRule?.content).toContain("Always say hi.");
|
||||
@@ -94,12 +97,48 @@ test("project RULES.md is found walking up from a sub-package cwd", async () =>
|
||||
|
||||
const rules = await loadNativeRules({ cwd: subPkg, home, repoRoot: project });
|
||||
|
||||
const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES");
|
||||
const projectRule = rules.find(r => r._source.level === "project" && r.name === "RULES@project");
|
||||
expect(projectRule).toBeDefined();
|
||||
expect(projectRule?.alwaysApply).toBe(true);
|
||||
expect(projectRule?.path).toBe(path.join(project, ".omp", "RULES.md"));
|
||||
});
|
||||
|
||||
test("user and project sticky RULES.md both survive public capability dedup", async () => {
|
||||
const userRulesPath = path.join(home, ".omp", "agent", "RULES.md");
|
||||
const projectRulesPath = path.join(project, ".omp", "RULES.md");
|
||||
const userRuleText = "User sticky rule: keep the personal safety checklist active.\n";
|
||||
const projectRuleText = "Project sticky rule: require repo-local release notes.\n";
|
||||
writeFile(userRulesPath, userRuleText);
|
||||
writeFile(projectRulesPath, projectRuleText);
|
||||
|
||||
const rules = await loadRulesCapability(project);
|
||||
|
||||
const stickyRules = rules.filter(rule => rule.path === userRulesPath || rule.path === projectRulesPath);
|
||||
expect(stickyRules).toHaveLength(2);
|
||||
|
||||
const userRule = stickyRules.find(rule => rule._source.level === "user");
|
||||
const projectRule = stickyRules.find(rule => rule._source.level === "project");
|
||||
|
||||
if (!userRule) throw new Error("user sticky rule missing");
|
||||
expect(userRule.name).toBe("RULES");
|
||||
expect(userRule.path).toBe(userRulesPath);
|
||||
expect(userRule._source.path).toBe(userRulesPath);
|
||||
expect(userRule.alwaysApply).toBe(true);
|
||||
expect(userRule.content).toContain(userRuleText.trim());
|
||||
expect("_shadowed" in userRule).toBe(false);
|
||||
|
||||
if (!projectRule) throw new Error("project sticky rule missing");
|
||||
expect(projectRule.name).toBe("RULES@project");
|
||||
expect(projectRule.path).toBe(projectRulesPath);
|
||||
expect(projectRule._source.path).toBe(projectRulesPath);
|
||||
expect(projectRule.alwaysApply).toBe(true);
|
||||
expect(projectRule.content).toContain(projectRuleText.trim());
|
||||
expect("_shadowed" in projectRule).toBe(false);
|
||||
|
||||
expect(userRule.name).not.toBe(projectRule.name);
|
||||
expect(userRule.content).not.toBe(projectRule.content);
|
||||
});
|
||||
|
||||
test("alwaysApply is forced even when frontmatter says false", async () => {
|
||||
writeFile(path.join(home, ".omp", "agent", "RULES.md"), "---\nalwaysApply: false\n---\nStick around anyway.\n");
|
||||
|
||||
|
||||
@@ -645,6 +645,354 @@ describe("listClaudePluginRoots", () => {
|
||||
|
||||
expect(found).toBeUndefined();
|
||||
});
|
||||
|
||||
test("reads slash commands from array-form commands manifest field (Claude plugin path-behavior rules)", async () => {
|
||||
// Mirrors real-world plugins such as addyosmani/agent-skills whose plugin.json
|
||||
// declares `"commands": ["./.claude/commands", "./commands"]`. Both directories
|
||||
// contribute; each command lands under the plugin's namespace.
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-commands-array");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude", "commands"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true });
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-commands-array@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ commands: ["./.claude/commands", "./commands"] }),
|
||||
);
|
||||
await fs.writeFile(path.join(pluginPath, ".claude", "commands", "spec.md"), "Spec\n");
|
||||
await fs.writeFile(path.join(pluginPath, ".claude", "commands", "plan.md"), "Plan\n");
|
||||
await fs.writeFile(path.join(pluginPath, "commands", "review.md"), "Review\n");
|
||||
|
||||
const result = await loadCapability<SlashCommand>("slash-commands", { cwd: tempDir });
|
||||
expect(result.warnings).toEqual([]);
|
||||
const names = result.all
|
||||
.filter(command => command.name.startsWith("manifest-commands-array:"))
|
||||
.map(command => command.name)
|
||||
.sort();
|
||||
expect(names).toEqual([
|
||||
"manifest-commands-array:plan",
|
||||
"manifest-commands-array:review",
|
||||
"manifest-commands-array:spec",
|
||||
]);
|
||||
});
|
||||
|
||||
test("reads slash commands from array-form manifest file entries", async () => {
|
||||
// Claude plugins reference allows command paths to be either flat `.md`
|
||||
// files or directories. A manifest-declared commands field still replaces
|
||||
// default `commands/`; plugins that want defaults must list `./commands`.
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-commands-files");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "custom"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "ops"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true });
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-commands-files@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ commands: ["./custom/deploy.md", "./ops"] }),
|
||||
);
|
||||
await fs.writeFile(path.join(pluginPath, "custom", "deploy.md"), "Deploy\n");
|
||||
await fs.writeFile(path.join(pluginPath, "ops", "rollback.md"), "Rollback\n");
|
||||
await fs.writeFile(path.join(pluginPath, "commands", "default.md"), "Default\n");
|
||||
|
||||
const result = await loadCapability<SlashCommand>("slash-commands", { cwd: tempDir });
|
||||
expect(result.warnings).toEqual([]);
|
||||
expect(result.all.find(c => c.name === "manifest-commands-files:deploy")?.content).toBe("Deploy\n");
|
||||
expect(result.all.find(c => c.name === "manifest-commands-files:rollback")?.content).toBe("Rollback\n");
|
||||
expect(result.all.find(c => c.name === "manifest-commands-files:default")).toBeUndefined();
|
||||
});
|
||||
|
||||
test("array-form commands warns on out-of-root entries while loading valid ones", async () => {
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-commands-mixed");
|
||||
const outsideDir = path.join(tempDir, "outside-commands");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude", "commands"), { recursive: true });
|
||||
await fs.mkdir(outsideDir, { recursive: true });
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-commands-mixed@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ commands: ["./.claude/commands", "../../outside-commands"] }),
|
||||
);
|
||||
await fs.writeFile(path.join(pluginPath, ".claude", "commands", "spec.md"), "Spec\n");
|
||||
await fs.writeFile(path.join(outsideDir, "escape.md"), "Escape\n");
|
||||
|
||||
const result = await loadCapability<SlashCommand>("slash-commands", { cwd: tempDir });
|
||||
expect(result.warnings.some(w => w.includes("Ignoring commands path outside plugin root"))).toBe(true);
|
||||
expect(result.all.find(c => c.name === "manifest-commands-mixed:spec")).toBeDefined();
|
||||
expect(result.all.find(c => c.name === "manifest-commands-mixed:escape")).toBeUndefined();
|
||||
});
|
||||
|
||||
test("reads skills from array-form skills manifest field", async () => {
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-skills-array");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "extra-skills", "alpha"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "more-skills", "beta"), { recursive: true });
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-skills-array@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ skills: ["./extra-skills", "./more-skills"] }),
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "extra-skills", "alpha", "SKILL.md"),
|
||||
"---\nname: alpha\ndescription: Alpha skill\n---\nBody\n",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "more-skills", "beta", "SKILL.md"),
|
||||
"---\nname: beta\ndescription: Beta skill\n---\nBody\n",
|
||||
);
|
||||
|
||||
const result = await loadCapability<Skill>("skills", { cwd: tempDir });
|
||||
expect(result.warnings).toEqual([]);
|
||||
expect(result.all.find(s => s.name === "alpha")).toBeDefined();
|
||||
expect(result.all.find(s => s.name === "beta")).toBeDefined();
|
||||
});
|
||||
|
||||
test("manifest skills field merges with default skills/ directory (adds, not replaces)", async () => {
|
||||
// Per Claude plugins reference "Path behavior rules":
|
||||
// `skills` adds to the default `skills/` scan; the default is always loaded
|
||||
// alongside any manifest-declared directories.
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-skills-merge");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "skills", "default-skill"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "extra-skills", "extra-skill"), { recursive: true });
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-skills-merge@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ skills: ["./extra-skills"] }),
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "skills", "default-skill", "SKILL.md"),
|
||||
"---\nname: default-skill\ndescription: Default skill\n---\nBody\n",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "extra-skills", "extra-skill", "SKILL.md"),
|
||||
"---\nname: extra-skill\ndescription: Extra skill\n---\nBody\n",
|
||||
);
|
||||
|
||||
const result = await loadCapability<Skill>("skills", { cwd: tempDir });
|
||||
expect(result.warnings).toEqual([]);
|
||||
expect(result.all.find(s => s.name === "default-skill")).toBeDefined();
|
||||
expect(result.all.find(s => s.name === "extra-skill")).toBeDefined();
|
||||
});
|
||||
|
||||
test("marketplace-root skills manifest field replaces default skills directory", async () => {
|
||||
// Claude path-behavior rules carve out marketplace entries whose source is the
|
||||
// marketplace root: their manifest `skills` field selects the published
|
||||
// subdirectories instead of also loading the root `skills/` directory.
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-skills-marketplace-root");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "skills", "unpublished-root-skill"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "plugins", "published", "skills", "published-skill"), {
|
||||
recursive: true,
|
||||
});
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-skills-marketplace-root@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "marketplace.json"),
|
||||
JSON.stringify({
|
||||
name: "market",
|
||||
owner: { name: "Market" },
|
||||
plugins: [{ name: "manifest-skills-marketplace-root", source: "./" }],
|
||||
}),
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ skills: ["./plugins/published/skills"] }),
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "skills", "unpublished-root-skill", "SKILL.md"),
|
||||
"---\nname: unpublished-root-skill\ndescription: Unpublished root skill\n---\nBody\n",
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "plugins", "published", "skills", "published-skill", "SKILL.md"),
|
||||
"---\nname: published-skill\ndescription: Published skill\n---\nBody\n",
|
||||
);
|
||||
|
||||
const result = await loadCapability<Skill>("skills", { cwd: tempDir });
|
||||
expect(result.warnings).toEqual([]);
|
||||
expect(result.all.find(s => s.name === "published-skill")).toBeDefined();
|
||||
expect(result.all.find(s => s.name === "unpublished-root-skill")).toBeUndefined();
|
||||
});
|
||||
|
||||
test("array-form skills entry pointing at a directory containing SKILL.md loads the single skill", async () => {
|
||||
// Per Claude plugins reference: a skills path may point directly at a directory whose
|
||||
// SKILL.md is the skill (frontmatter name → invocation, directory basename → fallback).
|
||||
// Real plugins use `"skills": ["./"]` — that entry must not silently drop the skill.
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-skills-self");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "single"), { recursive: true });
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-skills-self@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ skills: ["./single"] }),
|
||||
);
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, "single", "SKILL.md"),
|
||||
"---\nname: solo-skill\ndescription: Solo skill\n---\nBody\n",
|
||||
);
|
||||
|
||||
const result = await loadCapability<Skill>("skills", { cwd: tempDir });
|
||||
expect(result.warnings).toEqual([]);
|
||||
expect(result.all.find(s => s.name === "solo-skill")).toBeDefined();
|
||||
});
|
||||
|
||||
test("manifest commands field replaces default commands/ directory (Claude replace semantics)", async () => {
|
||||
// Per Claude plugins reference "Path behavior rules":
|
||||
// `commands` REPLACES the default `commands/` scan when the manifest key is set.
|
||||
// A plugin that wants both must list `./commands` explicitly.
|
||||
const pluginsDir = path.join(tempDir, ".claude", "plugins");
|
||||
const pluginPath = path.join(tempDir, "plugins", "manifest-commands-replace");
|
||||
await fs.mkdir(pluginsDir, { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, ".claude-plugin"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "commands"), { recursive: true });
|
||||
await fs.mkdir(path.join(pluginPath, "admin-commands"), { recursive: true });
|
||||
|
||||
const registry = {
|
||||
version: 2,
|
||||
plugins: {
|
||||
"manifest-commands-replace@market": [
|
||||
{
|
||||
scope: "user",
|
||||
installPath: pluginPath,
|
||||
version: "1.0.0",
|
||||
installedAt: "2025-01-01T00:00:00Z",
|
||||
lastUpdated: "2025-01-01T00:00:00Z",
|
||||
},
|
||||
],
|
||||
},
|
||||
};
|
||||
await fs.writeFile(path.join(pluginsDir, "installed_plugins.json"), JSON.stringify(registry));
|
||||
await fs.writeFile(
|
||||
path.join(pluginPath, ".claude-plugin", "plugin.json"),
|
||||
JSON.stringify({ commands: ["./admin-commands"] }),
|
||||
);
|
||||
// This file lives under the default commands/ dir and MUST NOT load once the
|
||||
// manifest declares `commands` (Claude's documented "replaces default" semantic).
|
||||
await fs.writeFile(path.join(pluginPath, "commands", "default.md"), "Default\n");
|
||||
await fs.writeFile(path.join(pluginPath, "admin-commands", "admin.md"), "Admin\n");
|
||||
|
||||
const result = await loadCapability<SlashCommand>("slash-commands", { cwd: tempDir });
|
||||
expect(result.warnings).toEqual([]);
|
||||
expect(result.all.find(c => c.name === "manifest-commands-replace:admin")).toBeDefined();
|
||||
expect(result.all.find(c => c.name === "manifest-commands-replace:default")).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe("discoverAgents plugin precedence", () => {
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { WorkerCore } from "@oh-my-pi/pi-coding-agent/eval/js/worker-core";
|
||||
import type {
|
||||
SessionSnapshot,
|
||||
Transport,
|
||||
WorkerInbound,
|
||||
WorkerOutbound,
|
||||
} from "@oh-my-pi/pi-coding-agent/eval/js/worker-protocol";
|
||||
|
||||
interface WorkerHarness {
|
||||
send(message: WorkerInbound): void;
|
||||
onMessage(handler: (message: WorkerOutbound) => void): () => void;
|
||||
}
|
||||
|
||||
function createWorkerHarness(): WorkerHarness {
|
||||
const hostListeners = new Set<(message: WorkerOutbound) => void>();
|
||||
const workerListeners = new Set<(message: WorkerInbound) => void>();
|
||||
const transport: Transport = {
|
||||
send: message => {
|
||||
queueMicrotask(() => {
|
||||
for (const listener of hostListeners) listener(message);
|
||||
});
|
||||
},
|
||||
onMessage: handler => {
|
||||
workerListeners.add(handler);
|
||||
return () => workerListeners.delete(handler);
|
||||
},
|
||||
close: () => {},
|
||||
};
|
||||
new WorkerCore(transport);
|
||||
return {
|
||||
send(message) {
|
||||
queueMicrotask(() => {
|
||||
for (const listener of workerListeners) listener(message);
|
||||
});
|
||||
},
|
||||
onMessage(handler) {
|
||||
hostListeners.add(handler);
|
||||
return () => hostListeners.delete(handler);
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function waitForMessage(
|
||||
harness: WorkerHarness,
|
||||
predicate: (message: WorkerOutbound) => boolean,
|
||||
): Promise<WorkerOutbound> {
|
||||
const { promise, resolve } = Promise.withResolvers<WorkerOutbound>();
|
||||
let unsubscribe = (): void => {};
|
||||
unsubscribe = harness.onMessage(message => {
|
||||
if (!predicate(message)) return;
|
||||
unsubscribe();
|
||||
resolve(message);
|
||||
});
|
||||
return promise;
|
||||
}
|
||||
|
||||
async function initializeWorker(harness: WorkerHarness, snapshot: SessionSnapshot): Promise<void> {
|
||||
const ready = waitForMessage(harness, message => message.type === "ready");
|
||||
harness.send({ type: "init", snapshot });
|
||||
expect((await ready).type).toBe("ready");
|
||||
}
|
||||
|
||||
describe("WorkerCore", () => {
|
||||
it("reports same-realm cwd conflicts through the worker protocol", async () => {
|
||||
const first = createWorkerHarness();
|
||||
const second = createWorkerHarness();
|
||||
const cwd = process.cwd();
|
||||
await initializeWorker(first, { cwd, sessionId: "same-realm-first", localRoots: {} });
|
||||
await initializeWorker(second, { cwd, sessionId: "same-realm-second", localRoots: {} });
|
||||
|
||||
const gate = Promise.withResolvers<void>();
|
||||
const entered = Promise.withResolvers<void>();
|
||||
(globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise<void> } }).__omp_worker_core_gate = {
|
||||
entered: () => entered.resolve(),
|
||||
wait: gate.promise,
|
||||
};
|
||||
try {
|
||||
first.send({
|
||||
type: "run",
|
||||
runId: "hold-first-runtime",
|
||||
code: "globalThis.__omp_worker_core_gate.entered(); await globalThis.__omp_worker_core_gate.wait;",
|
||||
filename: "[same-realm-first].js",
|
||||
snapshot: { cwd, sessionId: "same-realm-first", localRoots: {} },
|
||||
});
|
||||
await entered.promise;
|
||||
|
||||
const result = waitForMessage(
|
||||
second,
|
||||
message => message.type === "result" && message.runId === "overlap-second-runtime",
|
||||
);
|
||||
second.send({
|
||||
type: "run",
|
||||
runId: "overlap-second-runtime",
|
||||
code: "1 + 1;",
|
||||
filename: "[same-realm-second].js",
|
||||
snapshot: { cwd, sessionId: "same-realm-second", localRoots: {} },
|
||||
});
|
||||
|
||||
expect(await result).toMatchObject({
|
||||
type: "result",
|
||||
runId: "overlap-second-runtime",
|
||||
ok: false,
|
||||
error: { message: "Cannot set cwd while another same-realm JS runtime is running" },
|
||||
});
|
||||
} finally {
|
||||
gate.resolve();
|
||||
delete (globalThis as { __omp_worker_core_gate?: { entered(): void; wait: Promise<void> } })
|
||||
.__omp_worker_core_gate;
|
||||
first.send({ type: "close" });
|
||||
second.send({ type: "close" });
|
||||
}
|
||||
});
|
||||
});
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user