diff --git a/.omp/commands/fix-issues.md b/.omp/commands/fix-issues.md new file mode 100644 index 000000000..9e9191375 --- /dev/null +++ b/.omp/commands/fix-issues.md @@ -0,0 +1,148 @@ +# Fix Issues Command + +Diagnose, reproduce, and (when reproducible) fix open GitHub issues in parallel — each in its own clean worktree, with build artifacts symlinked so nothing recompiles. + +## Arguments + +- `$ARGUMENTS` — optional. Either: + - a space- or comma-separated list of issue numbers / URLs, OR + - GitHub-search qualifiers (`is:open`, `label:bug`, `author:foo`, ...) and/or a relative time window like `3d`, `2w`, `12h`. + +If no issues and no flags are passed, default to **all open issues opened in the last 3 days**. + +## Steps + +### 1. Resolve the issue set + +Parse `$ARGUMENTS`. + +- If explicit issue numbers/URLs given, use them verbatim. +- Otherwise call the `github` tool with `op: search_issues`. Default (no args): + + ``` + github { op: "search_issues", query: "is:open", since: "3d", limit: 50 } + ``` + + Pass any user-supplied qualifiers verbatim through `query` (combine with `is:open` if not already present). Use `since` for the time window (`3d`, `2w`, `12h`, ISO date — see the `github` tool docs); set `dateField: "updated"` instead of the `created` default only when the user explicitly asks for recently-touched issues. + +Print the resolved set before fanning out so the user can confirm scope. + +### 2. Fan out one subagent per issue + +Use **`task` with parallel subagents** — one task per issue. Pass the issue number, title, body summary, and the workflow below as the assignment. Subagents work in isolation; coordinate via `irc` only when two issues clearly touch the same file. + +Each subagent **MUST** follow this exact workflow: + +#### a. Read everything + +1. `github issue_view` (with comments) — comments often carry the real repro and fix hints. +2. `gh search prs` for the issue number to see if a fix is already in flight. + - If a PR exists and looks reasonable → switch tracks: review that PR per `.omp/commands/review-prs.md` instead, and report back as `existing-pr`. Do **not** open a competing fix. + +#### b. Diagnose & try to reproduce — **in the current cwd, on `main`** + +Reproduce **here first**, before touching any worktree. The point is to confirm the bug is real on current main before investing in a fix branch. + +1. Read the relevant source paths in this checkout. Form a concrete hypothesis (one or two sentences) about the failure. +2. Write a focused test file under the package the bug lives in. Naming: `repro-issue--.test.ts` (or `.rs`, etc.) — unique, greppable, deletable. +3. Run **only that test file**, not the suite. Confirm it fails for the reason in the issue. + +Outcomes: +- **Reproduced** → continue to (c). +- **Not reproduced** → stop. Delete the test file. Report `unreproduced` with: hypothesis tried, evidence it doesn't fail, and what info would unblock (versions, OS, config, repro snippet from author). Do **not** create a worktree or commit. +- **Out of scope / not a bug** (e.g. user config error, intended behavior, dup) → stop. Report `not-a-bug` with the explanation suitable for posting to the issue. + +#### c. Create a worktree off main + +Only after a confirmed local repro: + +```bash +MAIN="$(git rev-parse --show-toplevel)" +ENC="$(printf '%s' "$MAIN" | sed 's|[/\\:]|-|g')" +WT="$HOME/.omp/wt/${ENC}/fix-issue-" + +git -C "$MAIN" fetch origin main +git -C "$MAIN" worktree add -B "fix/issue-" "$WT" origin/main +``` + +Branch naming: `fix/issue-` (or `fix/issue--` if you'll open multiple). Path under `~/.omp/wt//...` matches the convention `pr_checkout` uses. + +#### d. Symlink build artifacts + +From the new worktree, link build outputs from `$MAIN` so `bun check` / `cargo build` / native loaders skip rebuilds: + +```bash +cd "$WT" +ln -snf "$MAIN/target" "$WT/target" +ln -snf "$MAIN/node_modules" "$WT/node_modules" + +# Only the .node binaries are expensive to rebuild. The rest of +# packages/natives/native/ is tracked by git, so folder-level symlinks would +# shadow real source files and break the fix. +for f in "$MAIN"/packages/natives/native/*.node; do + [ -e "$f" ] && ln -snf "$f" "$WT/packages/natives/native/" +done +``` + +Use absolute paths — the worktree lives outside the main checkout. + +#### e. Move the repro test in & fix + +1. Move (don't copy) the failing test file from the main checkout into the same path inside the worktree. Delete it from main so the original cwd is left clean. +2. Confirm it still fails inside the worktree on the current branch. +3. Implement the fix in source. Match existing patterns (see `AGENTS.md`); fix at the source, not at the symptom; no stubs, no mocks added to product code. +4. Re-run the repro test until it passes. +5. Add or adjust adjacent unit/contract tests where the fix changes a real contract — not just plumbing. Run **only** the affected test files; no full-suite runs from subagents. +6. Run `bun fmt` over the union of files edited. + +#### f. Commit + +Conventional commit, one logical change per commit, with `Fixes #`: + +```bash +git add -A +git commit -m "fix(): + + + +Fixes #." +``` + +Do **not** push. The human pushes / opens the PR. + +#### g. Report back + +Each subagent returns a short structured report: + +``` +Issue # +Status: fixed | unreproduced | not-a-bug | existing-pr (#<M>) +Repro: <test path inside worktree> (if applicable) +Worktree: ~/.omp/wt/.../fix-issue-<N> (if created) +Branch: fix/issue-<N> (if created) +Commits: <shas + one-liners> (if any) +Notes: <root cause in one sentence; or what info is missing> +``` + +### 3. Aggregate + +After all subagents finish, print a single summary table: + +``` +| # | Title | Status | Branch / Notes | +|---|-------|--------|----------------| +``` + +Group worktree paths by status (`fixed` first), so the user can `cd` and push the ready ones in one pass. + +## Rules + +- **MUST** reproduce on `main` in the current cwd **before** creating any worktree. No worktree until repro is confirmed. +- **MUST** use parallel subagents — one per issue. +- **MUST** check for an existing PR first; if one exists and is reasonable, divert to `review-prs` flow instead of duplicating work. +- **MUST** symlink `target`, `node_modules`, and the native `*.node` binaries before any build/test runs in the worktree. **MUST NOT** symlink the whole `packages/natives/native/` directory that would shadow tracked source files. +- **MUST** use conventional commits with `Fixes #<N>` in the body. +- **MUST NOT** push, open PRs, or comment on issues. Human handles delivery. +- **MUST NOT** ship stubs, mocks-as-product-code, or "TODO: implement" placeholders as a fix. +- **MUST NOT** expand scope: fix the reported bug, not adjacent code smells. +- If repro fails, delete the temporary test file from cwd before yielding — leave the original checkout clean. diff --git a/.omp/commands/review-prs.md b/.omp/commands/review-prs.md new file mode 100644 index 000000000..73ab65c31 --- /dev/null +++ b/.omp/commands/review-prs.md @@ -0,0 +1,147 @@ +# Review PRs Command + +Triage incoming pull requests in parallel: decide what's worth merging, prep clean rebased worktrees, fix any blockers, and hand them back ready for human merge. + +## Arguments + +- `$ARGUMENTS` — optional. Either: + - a space- or comma-separated list of PR numbers / URLs, OR + - GitHub-search qualifiers (`is:open`, `author:foo`, `label:bug`, `draft:false`, ...) and/or a relative time window like `3d`, `2w`, `12h`. + +If no PRs and no flags are passed, default to **all open PRs opened in the last 3 days**. + +## Steps + +### 1. Resolve the PR set + +Parse `$ARGUMENTS`. + +- If explicit PR numbers/URLs given, use them verbatim. +- Otherwise call the `github` tool with `op: search_prs`. Default (no args): + + ``` + github { op: "search_prs", query: "is:open", since: "3d", limit: 50 } + ``` + + Pass any user-supplied qualifiers verbatim through `query` (combine with `is:open` if not already present). Use `since` for the time window (`3d`, `2w`, `12h`, ISO date — see the `github` tool docs); set `dateField: "updated"` instead of the `created` default only when the user explicitly asks for recently-touched PRs. + +Print the resolved set before fanning out so the user can confirm scope. + +### 2. Fan out one subagent per PR + +Use **`task` with parallel subagents** — one task per PR. Pass the PR number, head ref, author, and the workflow below as the assignment. Each subagent works in isolation; they coordinate via `irc` only if a fix on PR A would obviously conflict with PR B. + +Each subagent **MUST** follow this exact workflow: + +#### a. Read & decide + +1. `github pr_view` (with comments) and `github pr_diff` for the PR. +2. Check `git log origin/main` and `gh search prs` for whether the same change already landed. +3. Classify into one of: + - **slop** — AI-generated noise, broken, off-spec, or net-negative. Drop, write a 1–2 line justification, do not check out. + - **superseded** — already fixed/merged in main or by a newer PR. Drop with a pointer. + - **worthy** — proceed. + +Anything ambiguous defaults to `worthy` — let the human decide on a real branch. + +#### b. Check out into a worktree + +```bash +gh_PR=<NUMBER> +# pr_checkout creates ~/.omp/wt/<encoded-repo>/pr-<N>/ and configures push remote +``` + +Use the `github pr_checkout` tool, **not** raw `gh pr checkout`. That gives a dedicated worktree wired up for `pr_push` later. + +#### c. Symlink build artifacts (skip native rebuilds) + +From inside the new worktree, link the heavy build outputs from the main checkout so `bun check` / `cargo build` / native loaders do not recompile: + +```bash +MAIN="<absolute path to main worktree, e.g. ~/Projects/pi>" +WT="$(pwd)" + +# Rust target dir + JS deps (root-level in this monorepo) +ln -snf "$MAIN/target" "$WT/target" +ln -snf "$MAIN/node_modules" "$WT/node_modules" + +# Prebuilt native addon (avoids 30s+ napi-rs rebuild). Link only the .node +# binaries — the rest of packages/natives/native/ is tracked by git, so +# folder-level symlinks would shadow PR-modified files and break review. +for f in "$MAIN"/packages/natives/native/*.node; do + [ -e "$f" ] && ln -snf "$f" "$WT/packages/natives/native/" +done +``` + +Resolve `$MAIN` from the original cwd before `pr_checkout` (`git rev-parse --show-toplevel`). Use absolute paths in symlinks; the worktree lives outside the main repo so relative paths break. + +#### d. Rebase onto main + +```bash +git fetch origin main +git rebase origin/main +``` + +If the rebase conflicts: +- Resolve trivially mechanical conflicts (formatting, import order, adjacent-line edits) and continue. +- Anything semantic → abort the rebase, leave a note in the final report, do not commit. + +#### e. Review & fix critical issues + +Inside the worktree, review the diff with the lens of: correctness, security, regressions, breaking-change impact, test coverage of the new path. + +Only fix things that **block merge**: build/test breakage, obvious bugs introduced by the PR, missing edge-case handling the PR's own goal demands. Do **not** rewrite for taste, refactor unrelated code, or expand scope. + +For every fix: +- Read existing patterns first; match repo conventions (see `AGENTS.md`). +- Add or update tests for the actual behavior change. +- Run only the targeted test file(s) for the area touched. No project-wide test runs from subagents. + +Format/lint at the end with `bun fmt` over the union of files you edited. + +#### f. Commit + +One conventional commit per logical fix on top of the rebased PR branch: + +```bash +git add -A +git commit -m "fix(<scope>): <what & why> + +Addresses review feedback on #<PR>." +``` + +Do **not** amend the PR author's commits. Do **not** push — the human merges. + +#### g. Report back + +Each subagent returns a short structured report: + +``` +PR #<N> <title> +Decision: worthy | slop | superseded +Worktree: ~/.omp/wt/.../pr-<N> (or: not checked out) +Rebase: clean | conflicts (resolved | aborted: <reason>) +Fixes: <commit shas + one-liners> (or: none needed) +Blockers: <anything the human must decide> +``` + +### 3. Aggregate + +After all subagents finish, print a single summary table: + +``` +| PR | Title | Decision | Rebase | Fixes | Blockers | +|----|-------|----------|--------|-------|----------| +``` + +Followed by the worktree paths grouped by decision, so the user can `cd` and merge in one go. + +## Rules + +- **MUST** use parallel subagents — one per PR — not a serial loop. +- **MUST** use `github pr_checkout` (carries push metadata) — not raw `gh pr checkout`. +- **MUST** symlink `target`, `node_modules`, and the native `*.node` binaries before any build/test runs in the worktree. **MUST NOT** symlink the whole `packages/natives/native/` directory that would shadow tracked PR changes. +- **MUST NOT** push or merge. Human reviews and merges. +- **MUST NOT** expand scope: fixes are limited to merge blockers on this PR's diff. +- **MUST NOT** force-push over the PR author's history. +- If a PR is `slop`/`superseded`, skip checkout entirely — just record the decision. diff --git a/Cargo.lock b/Cargo.lock index 7a2de8389..c987b3a88 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2370,47 +2370,15 @@ dependencies = [ ] [[package]] -name = "pi-natives" +name = "pi-ast" version = "14.9.3" dependencies = [ "anyhow", - "arboard", "ast-grep-core", - "brush-builtins", - "brush-core", - "brush-parser 0.3.0", - "clap", - "dashmap", "globset", - "grep-matcher", - "grep-regex", - "grep-searcher", - "html-to-markdown-rs", - "icy_sixel", "ignore", - "image", - "inferno", - "libc", - "memmap2", - "mimalloc", - "napi", - "napi-build", - "napi-derive", - "os_pipe", - "parking_lot", "phf 0.13.1", - "portable-pty", - "rayon", - "regex", "serde", - "serde_json", - "similar 3.1.0", - "smallvec", - "syntect", - "tiktoken-rs", - "tokio", - "tokio-util", - "toml", "tree-sitter", "tree-sitter-astro-next", "tree-sitter-bash", @@ -2467,6 +2435,48 @@ dependencies = [ "tree-sitter-xml", "tree-sitter-yaml", "tree-sitter-zig", +] + +[[package]] +name = "pi-natives" +version = "14.9.3" +dependencies = [ + "anyhow", + "arboard", + "ast-grep-core", + "clap", + "dashmap", + "globset", + "grep-matcher", + "grep-regex", + "grep-searcher", + "html-to-markdown-rs", + "icy_sixel", + "ignore", + "image", + "inferno", + "libc", + "memmap2", + "mimalloc", + "napi", + "napi-build", + "napi-derive", + "parking_lot", + "phf 0.13.1", + "pi-ast", + "pi-shell", + "portable-pty", + "rayon", + "regex", + "serde", + "serde_json", + "similar 3.1.0", + "smallvec", + "syntect", + "tiktoken-rs", + "tokio", + "tokio-util", + "toml", "unicode-segmentation", "unicode-width", "webp", @@ -2475,6 +2485,28 @@ dependencies = [ "xxhash-rust", ] +[[package]] +name = "pi-shell" +version = "14.9.3" +dependencies = [ + "anyhow", + "brush-builtins", + "brush-core", + "brush-parser 0.3.0", + "clap", + "libc", + "os_pipe", + "regex", + "serde", + "serde_json", + "tokio", + "tokio-util", + "toml", + "windows-sys 0.61.2", + "winreg 0.56.0", + "xxhash-rust", +] + [[package]] name = "pin-project-lite" version = "0.2.17" diff --git a/Cargo.toml b/Cargo.toml index 5bafe96ba..b6e6ef67c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -37,9 +37,17 @@ incremental = true strip = false [profile.dev] -opt-level = 3 -lto = "thin" -codegen-units = 16 +opt-level = 0 +lto = false +codegen-units = 256 +incremental = true +debug = "line-tables-only" +split-debuginfo = "unpacked" + +# Deps compile optimized once and cache; your own crates stay fast. +[profile.dev.package."*"] +opt-level = 2 +debug = false [workspace.lints.clippy] # Base Lint Levels diff --git a/bunfig.toml b/bunfig.toml index 6fcda289f..6bfbeeffa 100644 --- a/bunfig.toml +++ b/bunfig.toml @@ -1,6 +1,7 @@ telemetry = false [install] +minimumReleaseAge = 259200 # 3 days in seconds linker = "hoisted" exact = true saveTextLockfile = true diff --git a/crates/pi-ast/Cargo.toml b/crates/pi-ast/Cargo.toml new file mode 100644 index 000000000..0cf991a42 --- /dev/null +++ b/crates/pi-ast/Cargo.toml @@ -0,0 +1,74 @@ +[package] +name = "pi-ast" +version.workspace = true +edition.workspace = true +license.workspace = true +authors.workspace = true +repository.workspace = true + +[lints] +workspace = true + +[dependencies] +anyhow = "1.0" +ast-grep-core = { version = "0.39", default-features = false, features = ["tree-sitter"] } +globset = "0.4" +ignore = "0.4" +phf = { version = "0.13", features = ["macros"] } +serde = { version = "1.0", features = ["derive"] } +tree-sitter = "0.25" +tree-sitter-astro = { version = "0.1.1", package = "tree-sitter-astro-next" } +tree-sitter-bash = "0.25" +tree-sitter-c = "0.24" +tree-sitter-clojure = "0.1" +tree-sitter-cmake = "0.7.1" +tree-sitter-c-sharp = "0.23" +tree-sitter-cpp = "0.23" +tree-sitter-dart = "0.2" +tree-sitter-css = "0.25" +tree-sitter-diff = "0.1" +tree-sitter-dockerfile = { version = "0.2.0", package = "tree-sitter-dockerfile-updated" } +tree-sitter-elixir = "0.3" +tree-sitter-erlang = "0.16.0" +tree-sitter-go = "0.25" +tree-sitter-graphql = "0.1.0" +tree-sitter-haskell = "0.23" +tree-sitter-hcl = "1.1" +tree-sitter-html = "0.23" +tree-sitter-ini = "1.4.0" +tree-sitter-java = "0.23" +tree-sitter-javascript = "0.25" +tree-sitter-json = "0.24" +tree-sitter-just = "0.2.0" +tree-sitter-julia = "0.23" +tree-sitter-kotlin = { version = "0.4", package = "tree-sitter-kotlin-sg" } +tree-sitter-lua = "0.5" +tree-sitter-make = "1.1" +tree-sitter-md = "0.5" +tree-sitter-nix = "0.3" +tree-sitter-objc = "3.0" +tree-sitter-ocaml = "0.24.2" +tree-sitter-odin = "1.3" +tree-sitter-perl = { version = "0.1.0", package = "tree-sitter-perl-next" } +tree-sitter-php = "0.24" +tree-sitter-powershell = "0.26.4" +tree-sitter-proto = "0.4.0" +tree-sitter-python = "0.25" +tree-sitter-r = "1.2.0" +tree-sitter-regex = "0.25" +tree-sitter-ruby = "0.23" +tree-sitter-rust = "0.24" +tree-sitter-scala = "0.26" +tree-sitter-solidity = "1.2" +tree-sitter-sql = { version = "0.3.11", package = "tree-sitter-sequel" } +tree-sitter-starlark = "1.3" +tree-sitter-svelte = { version = "0.1.1", package = "tree-sitter-svelte-next" } +tree-sitter-swift = "0.7" +tree-sitter-toml-ng = "0.7" +tree-sitter-tlaplus = "1.5" +tree-sitter-typescript = "0.23" +tree-sitter-verilog = "1.0" +tree-sitter-vue = { version = "0.1.0", package = "tree-sitter-vue-next" } +tree-sitter-xml = "0.7" +tree-sitter-yaml = "0.7" +tree-sitter-zig = "1.1" diff --git a/crates/pi-natives/src/language/mod.rs b/crates/pi-ast/src/language/mod.rs similarity index 99% rename from crates/pi-natives/src/language/mod.rs rename to crates/pi-ast/src/language/mod.rs index 45cb63fc1..a7b93768b 100644 --- a/crates/pi-natives/src/language/mod.rs +++ b/crates/pi-ast/src/language/mod.rs @@ -411,6 +411,10 @@ impl SupportLang { LANG_ALIASES.get(lowered.as_str()).copied() } + pub fn from_path(path: &Path) -> Option<Self> { + from_extension(path) + } + pub fn sorted_aliases() -> &'static [&'static str] { &SORTED_ALIASES } diff --git a/crates/pi-natives/src/language/parsers.rs b/crates/pi-ast/src/language/parsers.rs similarity index 100% rename from crates/pi-natives/src/language/parsers.rs rename to crates/pi-ast/src/language/parsers.rs diff --git a/crates/pi-ast/src/lib.rs b/crates/pi-ast/src/lib.rs new file mode 100644 index 000000000..51f21642f --- /dev/null +++ b/crates/pi-ast/src/lib.rs @@ -0,0 +1,5 @@ +pub mod language; +pub mod ops; +pub mod summary; + +pub use language::SupportLang; diff --git a/crates/pi-ast/src/ops.rs b/crates/pi-ast/src/ops.rs new file mode 100644 index 000000000..bc0fce9c7 --- /dev/null +++ b/crates/pi-ast/src/ops.rs @@ -0,0 +1,305 @@ +use std::path::{Path, PathBuf}; + +use anyhow::{Result, anyhow}; +use ast_grep_core::{ + MatchStrictness, Position, + matcher::{Pattern, PatternError}, + source::Edit, + tree_sitter::{LanguageExt, StrDoc}, +}; +use globset::{Glob, GlobSet, GlobSetBuilder}; +use ignore::WalkBuilder; + +use crate::language::SupportLang; + +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum AstMatchStrictness { + Cst, + Smart, + Ast, + Relaxed, + Signature, + Template, +} + +impl From<AstMatchStrictness> for MatchStrictness { + fn from(value: AstMatchStrictness) -> Self { + match value { + AstMatchStrictness::Cst => Self::Cst, + AstMatchStrictness::Smart => Self::Smart, + AstMatchStrictness::Ast => Self::Ast, + AstMatchStrictness::Relaxed => Self::Relaxed, + AstMatchStrictness::Signature => Self::Signature, + AstMatchStrictness::Template => Self::Template, + } + } +} + +#[derive(Debug, Clone)] +pub struct AstMatch { + pub line: usize, + pub column: usize, + pub end_line: usize, + pub end_column: usize, + pub byte_start: usize, + pub byte_end: usize, + pub text: String, +} + +#[derive(Debug, Clone)] +pub struct MatchedFile { + pub absolute_path: PathBuf, + pub relative_path: String, +} + +#[derive(Debug, Clone)] +pub struct CompiledRewrite { + pub out: String, + pub patterns: Vec<Pattern>, +} + +#[must_use] +pub fn resolve_strictness(value: Option<AstMatchStrictness>) -> MatchStrictness { + value.map_or(MatchStrictness::Smart, Into::into) +} + +#[must_use] +pub fn supported_lang_list() -> String { + SupportLang::sorted_aliases().join(", ") +} + +pub fn resolve_supported_lang(value: &str) -> Result<SupportLang> { + SupportLang::from_alias(value).ok_or_else(|| { + anyhow!("Unsupported language '{value}'. Supported: {}", supported_lang_list()) + }) +} + +pub fn resolve_language(lang: Option<&str>, file_path: &Path) -> Result<SupportLang> { + if let Some(lang) = lang.map(str::trim).filter(|lang| !lang.is_empty()) { + return resolve_supported_lang(lang); + } + SupportLang::from_path(file_path).ok_or_else(|| { + anyhow!( + "Unable to infer language from file extension: {}. Specify `lang` explicitly.", + file_path.display() + ) + }) +} + +#[must_use] +pub fn is_supported_file(file_path: &Path, explicit_lang: Option<&str>) -> bool { + if explicit_lang.is_some() { + return true; + } + resolve_language(None, file_path).is_ok() +} + +pub fn compile_pattern( + pattern: &str, + selector: Option<&str>, + strictness: &MatchStrictness, + lang: SupportLang, +) -> Result<Pattern> { + let mut compiled = if let Some(selector) = selector.map(str::trim).filter(|s| !s.is_empty()) { + Pattern::contextual(pattern, selector, lang) + } else { + Pattern::try_new(pattern, lang) + } + .map_err(|err| anyhow!("Invalid pattern: {err}"))?; + compiled.strictness = strictness.clone(); + Ok(compiled) +} + +pub fn compile_search_patterns( + pattern: &str, + language: SupportLang, +) -> Result<Vec<Pattern>, PatternError> { + let mut compiled = vec![Pattern::try_new(pattern, language)?]; + if language == SupportLang::Rust { + let trimmed = pattern.trim_end(); + if let Some(contextual) = compile_rust_contextual_pattern(trimmed) { + compiled.push(contextual); + } + } + Ok(compiled) +} + +pub fn compile_rewrite_rules( + rules: &[(String, String)], + language: SupportLang, +) -> Result<Vec<CompiledRewrite>, (usize, PatternError)> { + rules + .iter() + .enumerate() + .map(|(index, (pattern, out))| { + compile_search_patterns(pattern, language) + .map(|patterns| CompiledRewrite { out: out.clone(), patterns }) + .map_err(|error| (index, error)) + }) + .collect() +} + +#[must_use] +pub fn collect_matches(source: &str, language: SupportLang, patterns: &[Pattern]) -> Vec<AstMatch> { + let ast = language.ast_grep(source); + let mut matches = Vec::new(); + for pattern in patterns { + for matched in ast.root().find_all(pattern.clone()) { + let start = matched.start_pos(); + let end = matched.end_pos(); + let range = matched.range(); + matches.push(AstMatch { + line: start.line() + 1, + column: char_column(start, matched.get_node()) + 1, + end_line: end.line() + 1, + end_column: char_column(end, matched.get_node()) + 1, + byte_start: range.start, + byte_end: range.end, + text: matched.text().into_owned(), + }); + } + } + matches +} + +pub fn rewrite_source( + source: &str, + language: SupportLang, + ops: &[CompiledRewrite], +) -> Result<(String, u32), String> { + let mut ast = language.ast_grep(source); + let mut replacements = 0_u32; + for op in ops { + for pattern in &op.patterns { + let edits = ast.root().replace_all(pattern.clone(), op.out.as_str()); + if edits.is_empty() { + continue; + } + replacements = replacements.saturating_add(edits.len() as u32); + let updated = + apply_edits(ast.root().text().as_ref(), &edits).map_err(|error| error.to_string())?; + ast = language.ast_grep(updated); + } + } + Ok((ast.root().text().into_owned(), replacements)) +} + +pub fn apply_edits(content: &str, edits: &[Edit<String>]) -> Result<String> { + let mut sorted: Vec<&Edit<String>> = edits.iter().collect(); + sorted.sort_by_key(|edit| edit.position); + let mut prev_end = 0usize; + for edit in &sorted { + if edit.position < prev_end { + return Err(anyhow!( + "Overlapping replacements detected; refine pattern to avoid ambiguous edits" + )); + } + prev_end = edit.position.saturating_add(edit.deleted_length); + } + + let mut output = content.to_string(); + for edit in sorted.into_iter().rev() { + let start = edit.position; + let end = edit.position.saturating_add(edit.deleted_length); + if end > output.len() || start > end { + return Err(anyhow!("Computed edit range is out of bounds")); + } + let replacement = String::from_utf8(edit.inserted_text.clone()) + .map_err(|err| anyhow!("Replacement text is not valid UTF-8: {err}"))?; + output.replace_range(start..end, &replacement); + } + Ok(output) +} + +pub fn collect_matched_files( + cwd: &Path, + patterns: &[String], +) -> Result<Vec<MatchedFile>, std::io::Error> { + let globset = build_globset(patterns)?; + let mut builder = WalkBuilder::new(cwd); + builder + .hidden(false) + .git_ignore(true) + .git_global(true) + .git_exclude(true); + let mut files = Vec::new(); + for entry in builder.build() { + let entry = match entry { + Ok(entry) => entry, + Err(error) => return Err(std::io::Error::other(error)), + }; + if !entry.file_type().is_some_and(|ft| ft.is_file()) { + continue; + } + let absolute_path = entry.into_path(); + let relative_path = absolute_path + .strip_prefix(cwd) + .unwrap_or(&absolute_path) + .to_string_lossy() + .replace('\\', "/"); + if globset.is_match(&relative_path) + || patterns.iter().any(|pattern| pattern == &relative_path) + { + files.push(MatchedFile { absolute_path, relative_path }); + } + } + files.sort_unstable_by(|left, right| left.relative_path.cmp(&right.relative_path)); + Ok(files) +} + +fn build_globset(patterns: &[String]) -> Result<GlobSet, std::io::Error> { + let mut builder = GlobSetBuilder::new(); + for pattern in patterns { + if has_glob_syntax(pattern) { + let glob = Glob::new(pattern).map_err(|error| { + std::io::Error::new( + std::io::ErrorKind::InvalidInput, + format!("invalid glob `{pattern}`: {error}"), + ) + })?; + builder.add(glob); + } + } + builder.build().map_err(std::io::Error::other) +} + +#[must_use] +pub fn has_glob_syntax(pattern: &str) -> bool { + pattern.contains('*') || pattern.contains('?') || pattern.contains('[') +} + +fn char_column(position: Position, node: &ast_grep_core::Node<'_, StrDoc<SupportLang>>) -> usize { + position.column(node) +} + +fn compile_rust_contextual_pattern(pattern: &str) -> Option<Pattern> { + let language = SupportLang::Rust; + let context = format!("fn __rwp_wrapper() {{ {pattern}; }}"); + let ast = language.ast_grep(&context); + let selector = ast.root().find("expression_statement")?; + Pattern::contextual(pattern, selector.kind().as_ref(), language).ok() +} + +#[cfg(test)] +mod tests { + use ast_grep_core::source::Edit; + + use super::{SupportLang, apply_edits, compile_search_patterns}; + + #[test] + fn compile_search_patterns_compiles_rust_patterns() { + let patterns = compile_search_patterns("foo($$$ARGS)", SupportLang::Rust) + .expect("rust pattern should compile"); + assert!(!patterns.is_empty()); + } + + #[test] + fn apply_edits_rejects_overlaps() { + let source = "abcdef"; + let edits = vec![ + Edit::<String> { position: 1, deleted_length: 3, inserted_text: b"x".to_vec() }, + Edit::<String> { position: 2, deleted_length: 1, inserted_text: b"y".to_vec() }, + ]; + assert!(apply_edits(source, &edits).is_err()); + } +} diff --git a/crates/pi-ast/src/summary.rs b/crates/pi-ast/src/summary.rs new file mode 100644 index 000000000..a917e1723 --- /dev/null +++ b/crates/pi-ast/src/summary.rs @@ -0,0 +1,1044 @@ +//! Structural source summaries powered by tree-sitter. + +use std::{collections::BTreeSet, path::Path}; + +use anyhow::{Result, anyhow}; +use ast_grep_core::tree_sitter::LanguageExt; +use serde::{Deserialize, Serialize}; +use tree_sitter::{Node, Parser}; + +use crate::language::SupportLang; + +const DEFAULT_MIN_BODY_LINES: u32 = 4; +const DEFAULT_MIN_COMMENT_LINES: u32 = 6; + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct SummaryOptions { + /// Source code to summarize. + pub code: String, + /// Language alias (e.g. "rust", "typescript") used before path inference. + pub lang: Option<String>, + /// File path used to infer language by extension when `lang` is omitted. + pub path: Option<String>, + /// Minimum total node lines before eliding a body/literal node. + pub min_body_lines: Option<u32>, + /// Minimum total comment lines before eliding a multiline block comment. + pub min_comment_lines: Option<u32>, +} + +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +pub struct SummarySegment { + /// "kept" or "elided". + pub kind: String, + /// 1-based inclusive start line. + pub start_line: u32, + /// 1-based inclusive end line. + pub end_line: u32, + /// Verbatim text for kept segments; absent for elided segments. + pub text: Option<String>, +} + +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +pub struct SummaryResult { + /// Canonical language name when parsing succeeded. + pub language: Option<String>, + /// True when tree-sitter parsed the source without syntax errors. + pub parsed: bool, + /// True when at least one elision span was emitted. + pub elided: bool, + /// Total source lines. + pub total_lines: u32, + /// Kept/elided segments in source order. + pub segments: Vec<SummarySegment>, +} + +#[derive(Clone, Copy, Debug, Eq, PartialEq)] +struct LineSpan { + start: u32, + end: u32, +} + +pub fn summarize_code(options: SummaryOptions) -> Result<SummaryResult> { + let source = options.code; + let total_lines = count_lines(&source); + if source.is_empty() { + return Ok(unparsed_result(source, total_lines)); + } + + let Some(language) = resolve_language(options.lang.as_deref(), options.path.as_deref()) else { + return Ok(unparsed_result(source, total_lines)); + }; + + let mut parser = Parser::new(); + parser + .set_language(&language.get_ts_language()) + .map_err(|err| anyhow!("Failed to load tree-sitter language: {err}"))?; + let Some(tree) = parser.parse(&source, None) else { + return Ok(unparsed_result(source, total_lines)); + }; + let root = tree.root_node(); + if root.has_error() { + return Ok(unparsed_result(source, total_lines)); + } + + let min_body_lines = options + .min_body_lines + .unwrap_or(DEFAULT_MIN_BODY_LINES) + .max(2); + let min_comment_lines = options + .min_comment_lines + .unwrap_or(DEFAULT_MIN_COMMENT_LINES) + .max(4); + let mut spans = Vec::new(); + collect_elisions(root, language, min_body_lines, min_comment_lines, &mut spans); + let spans = normalize_spans(spans, total_lines); + let segments = build_segments(&source, total_lines, &spans); + + Ok(SummaryResult { + language: Some(language.canonical_name().to_string()), + parsed: true, + elided: !spans.is_empty(), + total_lines, + segments, + }) +} + +fn resolve_language(lang: Option<&str>, path: Option<&str>) -> Option<SupportLang> { + if let Some(lang) = lang.map(str::trim).filter(|lang| !lang.is_empty()) { + return SupportLang::from_alias(lang); + } + let path = path?.trim(); + if path.is_empty() { + return None; + } + SupportLang::from_path(Path::new(path)) +} + +fn unparsed_result(source: String, total_lines: u32) -> SummaryResult { + let segments = if source.is_empty() { + Vec::new() + } else { + vec![SummarySegment { + kind: "kept".to_string(), + start_line: 1, + end_line: total_lines, + text: Some(source), + }] + }; + SummaryResult { language: None, parsed: false, elided: false, total_lines, segments } +} + +fn count_lines(source: &str) -> u32 { + if source.is_empty() { + 0 + } else { + source.lines().count().max(1).min(u32::MAX as usize) as u32 + } +} + +fn collect_elisions( + node: Node<'_>, + language: SupportLang, + min_body_lines: u32, + min_comment_lines: u32, + spans: &mut Vec<LineSpan>, +) { + let total_lines = node_line_count(node); + if is_comment_kind(language, node.kind()) { + if total_lines >= min_comment_lines { + let start_line = node_start_line(node) + 2; + let end_line = node_end_line(node).saturating_sub(1); + if start_line <= end_line { + spans.push(LineSpan { start: start_line, end: end_line }); + } + } + return; + } + + if is_elidable_kind(language, node.kind()) && total_lines >= min_body_lines { + let start_line = node_start_line(node) + 1; + let end_line = node_end_line(node).saturating_sub(1); + if start_line <= end_line { + spans.push(LineSpan { start: start_line, end: end_line }); + return; + } + } + + // Detect consecutive runs of groupable siblings (e.g. import statements). + // When the run's total line span meets `min_body_lines`, elide the lines + // strictly between the first and last sibling's content, leaving the + // boundary statements visible. + let child_count = node.child_count(); + let mut run_first: Option<Node<'_>> = None; + let mut run_last: Option<Node<'_>> = None; + let mut run_count: u32 = 0; + for index in 0..child_count { + let Some(child) = node.child(index) else { + continue; + }; + if is_groupable_kind(language, child.kind()) { + if run_first.is_none() { + run_first = Some(child); + } + run_last = Some(child); + run_count += 1; + } else { + flush_groupable_run(run_first, run_last, run_count, min_body_lines, spans); + run_first = None; + run_last = None; + run_count = 0; + } + } + flush_groupable_run(run_first, run_last, run_count, min_body_lines, spans); + + for index in 0..child_count { + if let Some(child) = node.child(index) { + collect_elisions(child, language, min_body_lines, min_comment_lines, spans); + } + } +} + +fn flush_groupable_run( + first: Option<Node<'_>>, + last: Option<Node<'_>>, + count: u32, + min_body_lines: u32, + spans: &mut Vec<LineSpan>, +) { + if count < 2 { + return; + } + let (Some(first), Some(last)) = (first, last) else { + return; + }; + let first_start = node_start_line(first); + let last_start = node_start_line(last); + let last_end = node_end_line(last); + let span_lines = last_end.saturating_sub(first_start).saturating_add(1); + if span_lines < min_body_lines { + return; + } + // Use the line of the first node's last visible content as the lower bound + // (some grammars include trailing newlines in the node range, which would + // otherwise place `end_line` on the next sibling's first line). + let first_content_end = node_content_end_line(first).min(last_start.saturating_sub(1)); + let start = first_content_end.saturating_add(1); + let end = last_start.saturating_sub(1); + if start <= end { + spans.push(LineSpan { start, end }); + } +} + +fn node_start_line(node: Node<'_>) -> u32 { + node + .start_position() + .row + .saturating_add(1) + .min(u32::MAX as usize) as u32 +} + +fn node_end_line(node: Node<'_>) -> u32 { + node + .end_position() + .row + .saturating_add(1) + .min(u32::MAX as usize) as u32 +} + +/// Last source line containing a content byte from `node`. +/// +/// Tree-sitter reports `end_position` as the position one past the last byte. +/// When that byte is a newline, the resulting position lands at column 0 of +/// the next row, which makes the naive `row + 1` answer one greater than the +/// row of the last visible content. This helper subtracts that off. +fn node_content_end_line(node: Node<'_>) -> u32 { + let pos = node.end_position(); + let row = if pos.column == 0 && pos.row > 0 { + pos.row - 1 + } else { + pos.row + }; + row.saturating_add(1).min(u32::MAX as usize) as u32 +} + +fn node_line_count(node: Node<'_>) -> u32 { + node_end_line(node) + .saturating_sub(node_start_line(node)) + .saturating_add(1) +} + +fn is_comment_kind(language: SupportLang, kind: &str) -> bool { + match language { + SupportLang::TypeScript | SupportLang::Tsx | SupportLang::JavaScript => kind == "comment", + SupportLang::Rust => kind == "block_comment", + SupportLang::Python => kind == "comment", + SupportLang::Go => kind == "comment", + SupportLang::Java => kind == "block_comment", + SupportLang::C | SupportLang::Cpp | SupportLang::ObjC => kind == "comment", + SupportLang::CSharp => kind == "comment", + SupportLang::Ruby => kind == "comment", + SupportLang::Php => kind == "comment", + SupportLang::Swift => kind == "comment", + SupportLang::Kotlin => kind == "block_comment", + SupportLang::Scala => kind == "block_comment", + SupportLang::Lua => kind == "comment", + _ => false, + } +} + +fn is_elidable_kind(language: SupportLang, kind: &str) -> bool { + match language { + SupportLang::TypeScript | SupportLang::Tsx | SupportLang::JavaScript => matches!( + kind, + "statement_block" + | "function_body" + | "object" + | "array" + | "template_string" + | "class_body" + | "interface_body" + | "enum_body" + | "object_type" + | "switch_body" + | "jsx_element" + | "jsx_self_closing_element" + ), + SupportLang::Rust => matches!( + kind, + "block" + | "array_expression" + | "tuple_expression" + | "struct_expression" + | "match_block" + | "raw_string_literal" + | "declaration_list" + | "field_declaration_list" + | "ordered_field_declaration_list" + | "enum_variant_list" + | "where_clause" + | "use_list" + | "macro_definition" + | "token_tree" + ), + SupportLang::Python => matches!( + kind, + "block" + | "dictionary" + | "list" | "set" + | "string" + | "tuple" + | "argument_list" + | "parameters" + | "parenthesized_expression" + | "list_comprehension" + | "set_comprehension" + | "dictionary_comprehension" + | "generator_expression" + | "import_from_statement" + | "subscript" + ), + SupportLang::Go => matches!( + kind, + "block" + | "composite_literal" + | "interpreted_string_literal" + | "raw_string_literal" + | "import_spec_list" + | "const_declaration" + | "var_declaration" + | "field_declaration_list" + | "interface_type" + | "expression_switch_statement" + | "type_switch_statement" + | "select_statement" + ), + SupportLang::Java => matches!( + kind, + "block" + | "array_initializer" + | "class_body" + | "interface_body" + | "enum_body" + | "annotation_type_body" + | "constructor_body" + | "switch_block" + | "string_literal" + ), + SupportLang::C => matches!( + kind, + "compound_statement" + | "initializer_list" + | "string_literal" + | "field_declaration_list" + | "enumerator_list" + | "concatenated_string" + ), + SupportLang::Cpp => matches!( + kind, + "compound_statement" + | "initializer_list" + | "string_literal" + | "field_declaration_list" + | "enumerator_list" + | "concatenated_string" + | "declaration_list" + | "raw_string_literal" + | "requires_clause" + ), + SupportLang::ObjC => matches!( + kind, + "compound_statement" + | "initializer_list" + | "string_literal" + | "protocol_declaration" + | "class_interface" + | "class_implementation" + | "instance_variables" + | "array_literal" + | "dictionary_literal" + ), + SupportLang::CSharp => matches!( + kind, + "block" + | "initializer_expression" + | "array_initializer_expression" + | "declaration_list" + | "enum_member_declaration_list" + | "switch_expression" + | "raw_string_literal" + | "interpolated_string_expression" + ), + SupportLang::Ruby => matches!( + kind, + "body_statement" + | "method" + | "do_block" + | "array" + | "hash" | "block" + | "case" | "heredoc_body" + ), + SupportLang::Php => matches!( + kind, + "compound_statement" + | "array_creation_expression" + | "declaration_list" + | "enum_declaration_list" + | "match_block" + | "heredoc" + | "nowdoc" + ), + SupportLang::Swift => matches!( + kind, + "function_body" + | "array_literal" + | "dictionary_literal" + | "multi_line_string_literal" + | "class_body" + | "protocol_body" + | "enum_class_body" + | "computed_property" + | "lambda_literal" + ), + SupportLang::Kotlin => matches!( + kind, + "function_body" + | "collection_literal" + | "multi_line_string_literal" + | "class_body" + | "enum_class_body" + | "when_expression" + | "import_list" + ), + SupportLang::Scala => matches!( + kind, + "block" + | "collection_literal" + | "template_body" + | "enum_body" + | "match_expression" + | "for_expression" + | "string" + ), + SupportLang::Lua => matches!(kind, "block" | "table_constructor" | "string"), + SupportLang::Perl => { + matches!(kind, "block" | "list_expression" | "heredoc_content" | "regexp_content") + }, + SupportLang::Dart => matches!( + kind, + "block" + | "function_expression_body" + | "class_body" + | "enum_body" + | "extension_body" + | "mixin_body" + | "list_literal" + | "set_or_map_literal" + | "string_literal" + ), + SupportLang::Bash => matches!( + kind, + "compound_statement" + | "if_statement" + | "case_statement" + | "do_group" + | "subshell" + | "array" + | "heredoc_body" + ), + SupportLang::Powershell => matches!( + kind, + "script_block" + | "statement_block" + | "class_statement" + | "param_block" + | "hash_literal_expression" + | "array_expression" + | "expandable_here_string_literal" + | "verbatim_here_string_characters" + ), + SupportLang::Haskell => matches!( + kind, + "imports" + | "data_type" + | "class" + | "instance" + | "function" + | "do" | "case" + | "let" | "local_binds" + | "list" | "tuple" + ), + SupportLang::Ocaml => matches!( + kind, + "structure" + | "signature" + | "variant_declaration" + | "record_declaration" + | "match_expression" + | "match_case" + | "let_expression" + | "value_definition" + | "list_expression" + ), + SupportLang::Elixir => matches!(kind, "do_block" | "list" | "map" | "string" | "sigil"), + SupportLang::Erlang => matches!( + kind, + "fun_decl" + | "case_expr" + | "if_expr" + | "receive_expr" + | "record_decl" + | "list" | "map_expr" + | "tuple" + ), + SupportLang::Clojure => { + matches!(kind, "list_lit" | "map_lit" | "vec_lit" | "set_lit" | "str_lit") + }, + SupportLang::Solidity => { + matches!(kind, "contract_body" | "function_body" | "struct_body" | "enum_body") + }, + SupportLang::Sql => matches!(kind, "column_definitions" | "case"), + SupportLang::Zig => matches!(kind, "Block" | "ContainerDecl" | "InitList"), + SupportLang::Odin => matches!( + kind, + "block" | "struct_declaration" | "enum_declaration" | "union_declaration" | "struct" + ), + SupportLang::Verilog => matches!( + kind, + "module_declaration" + | "seq_block" + | "case_statement" + | "function_declaration" + | "task_declaration" + | "list_of_port_declarations" + ), + SupportLang::Tlaplus => matches!(kind, "module" | "theorem" | "let_in"), + SupportLang::Nix => matches!( + kind, + "attrset_expression" | "list_expression" | "let_expression" | "indented_string_expression" + ), + SupportLang::Proto => matches!(kind, "message_body" | "enum_body" | "oneof" | "service"), + SupportLang::Julia => matches!( + kind, + "function_definition" + | "struct_definition" + | "module_definition" + | "do_clause" + | "vector_expression" + | "string_literal" + ), + SupportLang::R => matches!(kind, "braced_expression" | "call" | "string"), + SupportLang::Starlark => matches!(kind, "block" | "list" | "dictionary" | "string"), + SupportLang::Astro => { + matches!(kind, "frontmatter_js_block" | "script_element" | "style_element" | "element") + }, + SupportLang::Vue => { + matches!(kind, "template_element" | "script_element" | "style_element" | "element") + }, + SupportLang::Svelte => matches!(kind, "script_element" | "style_element" | "element"), + SupportLang::Html => matches!(kind, "element" | "script_element" | "style_element"), + SupportLang::Css => matches!(kind, "block" | "keyframe_block_list"), + SupportLang::Json => matches!(kind, "object" | "array"), + SupportLang::Xml => kind == "element", + SupportLang::Markdown => matches!(kind, "fenced_code_block" | "pipe_table" | "list"), + SupportLang::Graphql => matches!( + kind, + "fields_definition" + | "enum_values_definition" + | "input_fields_definition" + | "schema_definition" + ), + SupportLang::Hcl => matches!(kind, "body" | "object"), + SupportLang::Dockerfile => kind == "shell_command", + SupportLang::Cmake => matches!(kind, "argument_list" | "body"), + SupportLang::Make => kind == "recipe", + SupportLang::Just => kind == "recipe_body", + // Skip: data formats with no closing-token anchor (Yaml mappings, + // Toml tables, Ini sections), the diff format whose informational + // content IS the lines inside hunks, and the leaf-token-only Regex + // grammar. Eliding any of these deletes the only content worth + // reading. + SupportLang::Yaml + | SupportLang::Toml + | SupportLang::Ini + | SupportLang::Diff + | SupportLang::Regex => false, + } +} + +fn is_groupable_kind(language: SupportLang, kind: &str) -> bool { + match language { + SupportLang::TypeScript | SupportLang::Tsx | SupportLang::JavaScript => { + kind == "import_statement" + }, + SupportLang::Rust => matches!(kind, "use_declaration" | "extern_crate_declaration"), + SupportLang::Python => { + matches!(kind, "import_statement" | "import_from_statement" | "future_import_statement") + }, + SupportLang::Go => kind == "import_declaration", + SupportLang::Java => kind == "import_declaration", + SupportLang::C | SupportLang::Cpp => kind == "preproc_include", + SupportLang::ObjC => matches!(kind, "preproc_include" | "import_declaration"), + SupportLang::CSharp => kind == "using_directive", + SupportLang::Php => kind == "namespace_use_declaration", + SupportLang::Swift => kind == "import_declaration", + SupportLang::Scala => matches!(kind, "import_declaration" | "import"), + SupportLang::Dart => kind == "import_or_export", + SupportLang::Ocaml => kind == "open_module", + SupportLang::Solidity => kind == "import_directive", + SupportLang::Julia => matches!(kind, "import_statement" | "using_statement"), + SupportLang::Proto => kind == "import", + SupportLang::Perl => kind == "use_statement", + // Languages where imports either have no run pattern, are wrapped in a + // single AST node already covered by `is_elidable_kind` (Kotlin's + // `import_list`, Haskell's `imports`), or live inside a too-generic + // container (Powershell `statement_list`). + SupportLang::Kotlin + | SupportLang::Haskell + | SupportLang::Powershell + | SupportLang::Ruby + | SupportLang::Lua + | SupportLang::Elixir + | SupportLang::Erlang + | SupportLang::Clojure + | SupportLang::Sql + | SupportLang::Zig + | SupportLang::Odin + | SupportLang::Verilog + | SupportLang::Tlaplus + | SupportLang::Nix + | SupportLang::R + | SupportLang::Starlark + | SupportLang::Bash + | SupportLang::Astro + | SupportLang::Vue + | SupportLang::Svelte + | SupportLang::Html + | SupportLang::Css + | SupportLang::Json + | SupportLang::Xml + | SupportLang::Markdown + | SupportLang::Graphql + | SupportLang::Hcl + | SupportLang::Dockerfile + | SupportLang::Cmake + | SupportLang::Make + | SupportLang::Just + | SupportLang::Yaml + | SupportLang::Toml + | SupportLang::Ini + | SupportLang::Diff + | SupportLang::Regex => false, + } +} + +fn normalize_spans(mut spans: Vec<LineSpan>, total_lines: u32) -> Vec<LineSpan> { + if total_lines == 0 { + return Vec::new(); + } + spans.retain(|span| span.start <= span.end && span.start <= total_lines); + for span in &mut spans { + span.end = span.end.min(total_lines); + } + spans.sort_by_key(|span| (span.start, span.end)); + let mut merged: Vec<LineSpan> = Vec::new(); + for span in spans { + if let Some(last) = merged.last_mut() + && span.start <= last.end.saturating_add(1) + { + last.end = last.end.max(span.end); + continue; + } + merged.push(span); + } + merged +} + +fn build_segments(source: &str, total_lines: u32, spans: &[LineSpan]) -> Vec<SummarySegment> { + if total_lines == 0 { + return Vec::new(); + } + let source_lines: Vec<&str> = source.lines().collect(); + let elided_lines = spans + .iter() + .flat_map(|span| span.start..=span.end) + .collect::<BTreeSet<_>>(); + let mut segments = Vec::new(); + let mut current_kind: Option<&str> = None; + let mut current_start = 1; + let mut current_lines: Vec<&str> = Vec::new(); + + for line_number in 1..=total_lines { + let is_elided = elided_lines.contains(&line_number); + let kind = if is_elided { "elided" } else { "kept" }; + if current_kind.is_some_and(|existing| existing != kind) { + push_segment( + &mut segments, + current_kind.expect("kind set"), + current_start, + line_number - 1, + ¤t_lines, + ); + current_start = line_number; + current_lines.clear(); + } + current_kind = Some(kind); + if !is_elided { + let index = line_number.saturating_sub(1) as usize; + current_lines.push(source_lines.get(index).copied().unwrap_or_default()); + } + } + + if let Some(kind) = current_kind { + push_segment(&mut segments, kind, current_start, total_lines, ¤t_lines); + } + segments +} + +fn push_segment( + segments: &mut Vec<SummarySegment>, + kind: &str, + start_line: u32, + end_line: u32, + lines: &[&str], +) { + segments.push(SummarySegment { + kind: kind.to_string(), + start_line, + end_line, + text: (kind == "kept").then(|| lines.join("\n")), + }); +} + +#[cfg(test)] +mod tests { + use super::*; + + fn summarize(code: &str, path: &str) -> SummaryResult { + summarize_code(SummaryOptions { + code: code.to_string(), + lang: None, + path: Some(path.to_string()), + min_body_lines: None, + min_comment_lines: None, + }) + .expect("summary succeeds") + } + + fn segment_kinds(result: &SummaryResult) -> Vec<&str> { + result + .segments + .iter() + .map(|segment| segment.kind.as_str()) + .collect() + } + + #[test] + fn summarizes_typescript_function_body() { + let result = summarize( + "export function greet(name: string): string {\n\tconst clean = name.trim();\n\tconst \ + label = clean || 'world';\n\treturn `hello ${label}`;\n}\n", + "fixture.ts", + ); + + assert!(result.parsed); + assert!(result.elided); + assert_eq!(result.language.as_deref(), Some("typescript")); + assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); + assert_eq!( + result.segments[0].text.as_deref(), + Some("export function greet(name: string): string {") + ); + assert_eq!(result.segments[1].start_line, 2); + assert_eq!(result.segments[1].end_line, 4); + assert_eq!(result.segments[2].text.as_deref(), Some("}")); + } + + #[test] + fn summarizes_rust_method_body_but_keeps_impl_boundaries() { + let result = summarize( + "struct Greeter;\n\nimpl Greeter {\n\tfn greet(&self) -> String {\n\t\tlet name = \ + \"world\";\n\t\tlet label = name.to_uppercase();\n\t\tformat!(\"hello \ + {label}\")\n\t}\n}\n", + "fixture.rs", + ); + + assert!(result.parsed); + assert!(result.elided); + let rendered = result + .segments + .iter() + .map(|segment| segment.text.clone().unwrap_or_else(|| "...".to_string())) + .collect::<Vec<_>>() + .join("\n"); + assert!(rendered.contains("impl Greeter {\n...\n}")); + } + + #[test] + fn summarizes_python_function_body() { + let result = + summarize( + "class Greeter:\n def greet(self, name: str) -> str:\n clean = \ + name.strip()\n label = clean or 'world'\n return f'hello {label}'\n", + "fixture.py", + ); + + assert!(result.parsed); + assert!(result.elided); + assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); + assert!( + result.segments[0] + .text + .as_deref() + .unwrap_or_default() + .contains("def greet") + ); + assert!( + result.segments[2] + .text + .as_deref() + .unwrap_or_default() + .contains("return") + ); + } + + #[test] + fn min_body_lines_controls_short_body_elision() { + let code = "function small() {\n\treturn 1;\n}\n"; + let default_result = summarize(code, "fixture.ts"); + assert!(default_result.parsed); + assert!(!default_result.elided); + + let override_result = summarize_code(SummaryOptions { + code: code.to_string(), + lang: Some("typescript".to_string()), + path: None, + min_body_lines: Some(3), + min_comment_lines: None, + }) + .expect("summary succeeds"); + assert!(override_result.elided); + } + + #[test] + fn parse_failure_falls_back_to_unparsed() { + let result = summarize("export function broken( {\n", "fixture.ts"); + assert!(!result.parsed); + assert!(!result.elided); + assert_eq!(result.segments.len(), 1); + } + + #[test] + fn unsupported_language_is_unparsed() { + let result = summarize("plain text\nwith lines\n", "fixture.txt"); + assert!(!result.parsed); + assert_eq!(result.segments[0].text.as_deref(), Some("plain text\nwith lines\n")); + } + + #[test] + fn summarizes_typescript_interface_body() { + let result = summarize( + "export interface Args {\n\tcwd?: string;\n\tprovider?: string;\n\tmodel?: \ + string;\n\tapiKey?: string;\n}\n", + "fixture.ts", + ); + + assert!(result.parsed); + assert!(result.elided); + assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); + assert_eq!(result.segments[0].text.as_deref(), Some("export interface Args {")); + assert_eq!(result.segments[2].text.as_deref(), Some("}")); + } + + #[test] + fn summarizes_typescript_class_body() { + let result = summarize( + "export class Greeter {\n\tname: string = \"world\";\n\tlength(): number { return \ + this.name.length; }\n\tgreet(): string { return this.name; }\n\tshout(): string { \ + return this.name.toUpperCase(); }\n}\n", + "fixture.ts", + ); + + assert!(result.parsed); + assert!(result.elided); + assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); + assert!( + result.segments[0] + .text + .as_deref() + .unwrap_or_default() + .contains("class Greeter") + ); + assert_eq!(result.segments[2].text.as_deref(), Some("}")); + } + + #[test] + fn summarizes_rust_trait_declaration_list() { + let result = summarize( + "pub trait Greeter {\n\tfn greet(&self) -> String;\n\tfn length(&self) -> usize;\n\tfn \ + shout(&self) -> String;\n\tfn whisper(&self) -> String;\n}\n", + "fixture.rs", + ); + + assert!(result.parsed); + assert!(result.elided); + assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); + assert_eq!(result.segments[0].text.as_deref(), Some("pub trait Greeter {")); + assert_eq!(result.segments[2].text.as_deref(), Some("}")); + } + + #[test] + fn summarizes_java_class_body() { + let result = summarize( + "public class Greeter {\n\tprivate String name;\n\tpublic Greeter(String n) { this.name \ + = n; }\n\tpublic String greet() { return name; }\n\tpublic int length() { return \ + name.length(); }\n}\n", + "fixture.java", + ); + + assert!(result.parsed); + assert!(result.elided); + assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); + assert!( + result.segments[0] + .text + .as_deref() + .unwrap_or_default() + .contains("class Greeter") + ); + assert_eq!(result.segments[2].text.as_deref(), Some("}")); + } + + #[test] + fn summarizes_typescript_import_run() { + let code = "import a from \"a\";\nimport b from \"b\";\nimport c from \"c\";\nimport d from \ + \"d\";\nimport e from \"e\";\nimport f from \"f\";\n\nexport function main() \ + {}\n"; + let result = summarize(code, "fixture.ts"); + + assert!(result.parsed); + assert!(result.elided); + // Lines 2-5 are between the first and last imports and must be elided. + let elided = result + .segments + .iter() + .find(|seg| seg.kind == "elided") + .expect("elided segment"); + assert_eq!(elided.start_line, 2); + assert_eq!(elided.end_line, 5); + // First import line is kept. + assert!( + result.segments[0] + .text + .as_deref() + .unwrap_or_default() + .starts_with("import a from") + ); + } + + #[test] + fn does_not_elide_short_typescript_import_run() { + // 3 imports → total span 3 lines, below default min_body_lines (4). + let result = summarize( + "import a from \"a\";\nimport b from \"b\";\nimport c from \"c\";\n", + "fixture.ts", + ); + assert!(result.parsed); + assert!(!result.elided); + } + + #[test] + fn summarizes_python_import_run() { + let code = "import os\nimport sys\nfrom typing import List\nfrom pathlib import \ + Path\nimport json\nimport re\n\nprint('go')\n"; + let result = summarize(code, "fixture.py"); + + assert!(result.parsed); + assert!(result.elided); + let elided = result + .segments + .iter() + .find(|seg| seg.kind == "elided") + .expect("elided segment"); + assert_eq!(elided.start_line, 2); + assert_eq!(elided.end_line, 5); + } + + #[test] + fn summarizes_c_preproc_include_run() { + // C grammar puts each #include's `end_position` at column 0 of the next + // row (the trailing `\n`). Without `node_content_end_line`, the run + // elision would emit a span that starts past the second include and + // only collapse the third — verify the boundary statements stay + // visible and the middle is collapsed. + let code = "#include <stdio.h>\n#include \"a.h\"\n#include \"b.h\"\n#include \ + \"c.h\"\n#include <string.h>\nint main(void) { return 0; }\n"; + let result = summarize(code, "fixture.c"); + + assert!(result.parsed); + assert!(result.elided); + let elided = result + .segments + .iter() + .find(|seg| seg.kind == "elided") + .expect("elided segment"); + assert_eq!(elided.start_line, 2); + assert_eq!(elided.end_line, 4); + } + + #[test] + fn summarizes_rust_use_run() { + let code = "use std::fs;\nuse std::path::Path;\nuse std::collections::HashMap;\nuse \ + std::sync::Arc;\nuse std::io;\n\nfn main() {}\n"; + let result = summarize(code, "fixture.rs"); + + assert!(result.parsed); + assert!(result.elided); + let elided = result + .segments + .iter() + .find(|seg| seg.kind == "elided") + .expect("elided segment"); + assert_eq!(elided.start_line, 2); + assert_eq!(elided.end_line, 4); + } +} diff --git a/crates/pi-natives/Cargo.toml b/crates/pi-natives/Cargo.toml index 602dbd7d7..d40c64992 100644 --- a/crates/pi-natives/Cargo.toml +++ b/crates/pi-natives/Cargo.toml @@ -18,13 +18,10 @@ tokio = { version = "1", features = ["full"] } tokio-util = { version = "0.7", features = ["full"] } napi = { version = "3", features = ["napi10", "tokio_rt", "tokio_time"] } napi-derive = "3" -brush-core = { version = "0.5.0", path = "../brush-core-vendored" } -brush-builtins = { version = "0.2.0", path = "../brush-builtins-vendored" } -brush-parser = "0.3" +pi-shell = { path = "../pi-shell" } parking_lot = "0.12.5" dashmap = "6.1" clap = { version = "4", features = ["derive"] } -os_pipe = "1" portable-pty = "0.9" grep-regex = "0.1" grep-searcher = "0.1" @@ -35,62 +32,7 @@ rayon = "1.12" ast-grep-core = { version = "0.39", default-features = false, features = [ "tree-sitter", ] } -tree-sitter = "0.25" -tree-sitter-astro = { version = "0.1.1", package = "tree-sitter-astro-next" } -tree-sitter-bash = "0.25" -tree-sitter-c = "0.24" -tree-sitter-clojure = "0.1" -tree-sitter-cmake = "0.7.1" -tree-sitter-c-sharp = "0.23" -tree-sitter-cpp = "0.23" -tree-sitter-dart = "0.2" -tree-sitter-css = "0.25" -tree-sitter-diff = "0.1" -tree-sitter-dockerfile = { version = "0.2.0", package = "tree-sitter-dockerfile-updated" } -tree-sitter-elixir = "0.3" -tree-sitter-erlang = "0.16.0" -tree-sitter-go = "0.25" -tree-sitter-graphql = "0.1.0" -tree-sitter-haskell = "0.23" -tree-sitter-hcl = "1.1" -tree-sitter-html = "0.23" -tree-sitter-ini = "1.4.0" -tree-sitter-java = "0.23" -tree-sitter-javascript = "0.25" -tree-sitter-json = "0.24" -tree-sitter-just = "0.2.0" -tree-sitter-julia = "0.23" -tree-sitter-kotlin = { version = "0.4", package = "tree-sitter-kotlin-sg" } -tree-sitter-lua = "0.5" -tree-sitter-make = "1.1" -tree-sitter-md = "0.5" -tree-sitter-nix = "0.3" -tree-sitter-objc = "3.0" -tree-sitter-ocaml = "0.24.2" -tree-sitter-odin = "1.3" -tree-sitter-perl = { version = "0.1.0", package = "tree-sitter-perl-next" } -tree-sitter-php = "0.24" -tree-sitter-powershell = "0.26.4" -tree-sitter-proto = "0.4.0" -tree-sitter-python = "0.25" -tree-sitter-r = "1.2.0" -tree-sitter-regex = "0.25" -tree-sitter-ruby = "0.23" -tree-sitter-rust = "0.24" -tree-sitter-scala = "0.26" -tree-sitter-solidity = "1.2" -tree-sitter-sql = { version = "0.3.11", package = "tree-sitter-sequel" } -tree-sitter-starlark = "1.3" -tree-sitter-svelte = { version = "0.1.1", package = "tree-sitter-svelte-next" } -tree-sitter-swift = "0.7" -tree-sitter-toml-ng = "0.7" -tree-sitter-tlaplus = "1.5" -tree-sitter-typescript = "0.23" -tree-sitter-verilog = "1.0" -tree-sitter-vue = { version = "0.1.0", package = "tree-sitter-vue-next" } -tree-sitter-xml = "0.7" -tree-sitter-yaml = "0.7" -tree-sitter-zig = "1.1" +pi-ast = { path = "../pi-ast" } inferno = { version = "0.12", default-features = false } image = { version = "0.25", default-features = false, features = [ "png", diff --git a/crates/pi-natives/src/ast.rs b/crates/pi-natives/src/ast.rs index c45f90d89..5803e0aea 100644 --- a/crates/pi-natives/src/ast.rs +++ b/crates/pi-natives/src/ast.rs @@ -5,13 +5,15 @@ use std::{ path::{Path, PathBuf}, }; -use ast_grep_core::{ - Language, MatchStrictness, matcher::Pattern, source::Edit, tree_sitter::LanguageExt, -}; +use ast_grep_core::{MatchStrictness, matcher::Pattern, source::Edit, tree_sitter::LanguageExt}; use napi::bindgen_prelude::*; use napi_derive::napi; +use pi_ast::{ + SupportLang, + ops::{self as shared_ops}, +}; -use crate::{fs_cache, glob_util, language::SupportLang, task}; +use crate::{fs_cache, glob_util, task}; const DEFAULT_FIND_LIMIT: u32 = 50; @@ -227,29 +229,12 @@ fn to_u32(value: usize) -> u32 { value.min(u32::MAX as usize) as u32 } -fn supported_lang_list() -> String { - SupportLang::sorted_aliases().join(", ") -} - fn resolve_supported_lang(value: &str) -> Result<SupportLang> { - SupportLang::from_alias(value).ok_or_else(|| { - Error::from_reason(format!( - "Unsupported language '{value}'. Supported: {}", - supported_lang_list() - )) - }) + shared_ops::resolve_supported_lang(value).map_err(|err| Error::from_reason(err.to_string())) } fn resolve_language(lang: Option<&str>, file_path: &Path) -> Result<SupportLang> { - if let Some(lang) = lang.map(str::trim).filter(|lang| !lang.is_empty()) { - return resolve_supported_lang(lang); - } - SupportLang::from_path(file_path).ok_or_else(|| { - Error::from_reason(format!( - "Unable to infer language from file extension: {}. Specify `lang` explicitly.", - file_path.display() - )) - }) + shared_ops::resolve_language(lang, file_path).map_err(|err| Error::from_reason(err.to_string())) } /// Returns true if the file's extension resolves to a supported language. @@ -257,10 +242,7 @@ fn resolve_language(lang: Option<&str>, file_path: &Path) -> Result<SupportLang> /// (the user chose to treat them as that language). When `lang` is None, /// only files with recognizable code extensions are included. fn is_supported_file(file_path: &Path, explicit_lang: Option<&str>) -> bool { - if explicit_lang.is_some() { - return true; - } - resolve_language(None, file_path).is_ok() + shared_ops::is_supported_file(file_path, explicit_lang) } fn infer_single_replace_lang( @@ -414,43 +396,12 @@ fn compile_pattern( strictness: &MatchStrictness, lang: SupportLang, ) -> Result<Pattern> { - let mut compiled = if let Some(selector) = selector.map(str::trim).filter(|s| !s.is_empty()) { - Pattern::contextual(pattern, selector, lang) - } else { - Pattern::try_new(pattern, lang) - } - .map_err(|err| Error::from_reason(format!("Invalid pattern: {err}")))?; - compiled.strictness = strictness.clone(); - Ok(compiled) + shared_ops::compile_pattern(pattern, selector, strictness, lang) + .map_err(|err| Error::from_reason(err.to_string())) } fn apply_edits(content: &str, edits: &[Edit<String>]) -> Result<String> { - let mut sorted: Vec<&Edit<String>> = edits.iter().collect(); - sorted.sort_by_key(|edit| edit.position); - let mut prev_end = 0usize; - for edit in &sorted { - if edit.position < prev_end { - return Err(Error::from_reason( - "Overlapping replacements detected; refine pattern to avoid ambiguous edits" - .to_string(), - )); - } - prev_end = edit.position.saturating_add(edit.deleted_length); - } - - let mut output = content.to_string(); - for edit in sorted.into_iter().rev() { - let start = edit.position; - let end = edit.position.saturating_add(edit.deleted_length); - if end > output.len() || start > end { - return Err(Error::from_reason("Computed edit range is out of bounds".to_string())); - } - let replacement = String::from_utf8(edit.inserted_text.clone()).map_err(|err| { - Error::from_reason(format!("Replacement text is not valid UTF-8: {err}")) - })?; - output.replace_range(start..end, &replacement); - } - Ok(output) + shared_ops::apply_edits(content, edits).map_err(|err| Error::from_reason(err.to_string())) } fn normalize_pattern_list(patterns: Option<Vec<String>>) -> Result<Vec<String>> { diff --git a/crates/pi-natives/src/lib.rs b/crates/pi-natives/src/lib.rs index 5ef602144..da9a2295a 100644 --- a/crates/pi-natives/src/lib.rs +++ b/crates/pi-natives/src/lib.rs @@ -37,7 +37,7 @@ pub mod highlight; pub mod html; pub mod image; pub mod keys; -pub mod language; +pub use pi_ast::language; pub mod power; diff --git a/crates/pi-natives/src/ps.rs b/crates/pi-natives/src/ps.rs index aed582aff..8cecb16e1 100644 --- a/crates/pi-natives/src/ps.rs +++ b/crates/pi-natives/src/ps.rs @@ -1,22 +1,22 @@ -//! Cross-platform process tree management. +//! N-API bindings for cross-platform process tree management. //! -//! Provides process tree enumeration and termination without requiring -//! processes to be spawned with `detached: true`. -//! -//! # Platform Implementation -//! - **Linux**: Owns pidfds and signals through `pidfd_send_signal` -//! - **macOS**: Uses `libproc` (`proc_listchildpids`) and PID validation -//! - **Windows**: Owns process handles opened from Toolhelp snapshots +//! The platform-specific implementation lives in [`pi_shell::process`]; this +//! module is a thin shim that exposes that crate's `Process` surface to +//! JavaScript and re-exports the termination primitives used by other native +//! modules (e.g. [`crate::pty`]). -use std::{collections::HashSet, time::Duration}; +use std::time::Duration; use napi::{ Env, Result, bindgen_prelude::{PromiseRaw, Unknown}, }; use napi_derive::napi; +use pi_shell::process::{self as core_process, ProcessStatus as CoreProcessStatus}; +pub use pi_shell::process::{KILL_SIGNAL, TERM_SIGNAL, TerminationTargets, kill_process_group}; use crate::task; + #[derive(Default)] #[napi(object)] pub struct ProcessTerminateOptions<'env> { @@ -54,1183 +54,12 @@ pub enum ProcessStatus { Exited, } -#[cfg(target_os = "linux")] -mod platform { - use std::{ - collections::HashSet, - ffi::OsStr, - fs, - os::fd::{AsRawFd, FromRawFd, OwnedFd, RawFd}, - ptr, - sync::Arc, - }; - - use super::ProcessStatus; - - /// Stable Linux process reference backed by a pidfd. - #[derive(Clone)] - pub struct Process { - pid: i32, - pidfd: Arc<OwnedFd>, - start_time: u64, - } - - impl Process { - pub fn from_pid(pid: i32) -> Option<Self> { - if pid <= 0 { - return None; - } - let pidfd = open_pidfd(pid)?; - let start_time = read_start_time(pid)?; - Some(Self { pid, pidfd, start_time }) +impl From<CoreProcessStatus> for ProcessStatus { + fn from(value: CoreProcessStatus) -> Self { + match value { + CoreProcessStatus::Running => Self::Running, + CoreProcessStatus::Exited => Self::Exited, } - - pub const fn pid(&self) -> i32 { - self.pid - } - - pub fn children(&self) -> Vec<Self> { - if !self.live_identity() { - return Vec::new(); - } - - // `/proc/{pid}/task/{tid}/children` is per-task: a child fork()ed from a - // worker thread appears under that thread's `tid`, not the tgid. Walk - // every task subdir and union the lists, then re-validate parentage. - let task_dir = format!("/proc/{}/task", self.pid); - let Ok(entries) = fs::read_dir(&task_dir) else { - return Vec::new(); - }; - - let mut seen: HashSet<i32> = HashSet::new(); - let mut out = Vec::new(); - for entry in entries.flatten() { - let name = entry.file_name(); - let Some(tid_str) = name.to_str() else { - continue; - }; - if tid_str.parse::<i32>().is_err() { - continue; - } - let children_path = format!("/proc/{}/task/{}/children", self.pid, tid_str); - let Ok(content) = fs::read_to_string(&children_path) else { - continue; - }; - for part in content.split_whitespace() { - let Ok(child_pid) = part.parse::<i32>() else { - continue; - }; - if !seen.insert(child_pid) { - continue; - } - let Some(child) = Self::from_pid(child_pid) else { - continue; - }; - if child.status() == ProcessStatus::Running - && current_parent_pid(child.pid) == Some(self.pid) - { - out.push(child); - } - } - } - out - } - - pub fn parent_pid(&self) -> Option<i32> { - if self.status() == ProcessStatus::Running { - current_parent_pid(self.pid) - } else { - None - } - } - - pub fn args(&self) -> Vec<String> { - if !self.live_identity() { - return Vec::new(); - } - - let cmdline_path = format!("/proc/{}/cmdline", self.pid); - let Ok(content) = fs::read(cmdline_path) else { - return Vec::new(); - }; - // Re-validate after the read: PID reuse between identity check and read - // would otherwise leak an impostor's command line to callers. - if !self.live_identity() { - return Vec::new(); - } - split_nul_arguments(&content) - } - - pub fn kill(&self, signal: i32) -> bool { - // SAFETY: `self.pidfd` is an owned file descriptor returned by a successful - // `pidfd_open` call and remains open for the duration of this syscall. A null - // `siginfo_t` pointer is explicitly accepted by `pidfd_send_signal` and makes - // the kernel synthesize the same signal metadata as `kill(2)`. Flags are zero, - // which is the documented default behavior. - let ret = unsafe { - libc::syscall( - libc::SYS_pidfd_send_signal, - self.pidfd.as_raw_fd(), - signal, - ptr::null::<libc::siginfo_t>(), - 0, - ) - }; - ret == 0 - } - - pub fn group_id(&self) -> Option<i32> { - if self.status() != ProcessStatus::Running { - return None; - } - - // SAFETY: `self.pid` names the process currently referenced by `self.pidfd` - // unless it exits concurrently. If it exits, `getpgid` reports failure rather - // than dereferencing caller-owned memory. - let pgid = unsafe { libc::getpgid(self.pid) }; - if pgid > 0 { Some(pgid) } else { None } - } - - pub fn status(&self) -> ProcessStatus { - loop { - let mut pollfd = - libc::pollfd { fd: self.pidfd.as_raw_fd(), events: libc::POLLIN, revents: 0 }; - // SAFETY: `pollfd` points to one initialized `pollfd` element, and the pidfd - // remains open for the duration of the call. Timeout zero makes this a - // non-blocking readiness probe. - let ready = unsafe { libc::poll(&raw mut pollfd, 1, 0) }; - if ready < 0 { - // Retry on EINTR; for any other transient poll error treat the pidfd as - // still running. The pidfd is still owned and the kernel has not reported - // the process gone — a spurious `Exited` here makes every downstream - // signal/kill fall through silently. - if std::io::Error::last_os_error().raw_os_error() == Some(libc::EINTR) { - continue; - } - return ProcessStatus::Running; - } - if ready == 0 { - return ProcessStatus::Running; - } - if (pollfd.revents & (libc::POLLIN | libc::POLLHUP | libc::POLLERR | libc::POLLNVAL)) - != 0 - { - return ProcessStatus::Exited; - } - return ProcessStatus::Running; - } - } - - /// Walk the descendant tree in post-order (leaves first), de-duplicating - /// by PID so concurrent reparenting cannot trap us in a cycle. - pub fn descendants(&self) -> Vec<Self> { - let mut out = Vec::new(); - let mut visited = HashSet::new(); - visited.insert(self.pid); - self.descendants_into(&mut out, &mut visited); - out - } - - fn descendants_into(&self, out: &mut Vec<Self>, visited: &mut HashSet<i32>) { - for child in self.children() { - if visited.insert(child.pid) { - child.descendants_into(out, visited); - out.push(child); - } - } - } - - fn live_identity(&self) -> bool { - self.status() == ProcessStatus::Running - && read_start_time(self.pid) == Some(self.start_time) - } - } - - fn split_nul_arguments(content: &[u8]) -> Vec<String> { - content - .split(|byte| *byte == 0) - .filter(|part| !part.is_empty()) - .map(|part| String::from_utf8_lossy(part).into_owned()) - .collect() - } - - fn current_parent_pid(pid: i32) -> Option<i32> { - let status_path = format!("/proc/{pid}/status"); - let content = fs::read_to_string(status_path).ok()?; - content.lines().find_map(|line| { - line - .strip_prefix("PPid:") - .and_then(|ppid| ppid.trim().parse::<i32>().ok()) - }) - } - - fn read_start_time(pid: i32) -> Option<u64> { - // `/proc/[pid]/stat` field 22 is the process start time in clock ticks since - // boot. The comm field (between parens) may itself contain spaces and parens, - // so locate the *last* `)` and split the trailing whitespace-separated fields. - let stat_path = format!("/proc/{pid}/stat"); - let content = fs::read_to_string(stat_path).ok()?; - let last_paren = content.rfind(')')?; - let rest = &content[last_paren + 1..]; - rest.split_whitespace().nth(19)?.parse().ok() - } - - fn open_pidfd(pid: i32) -> Option<Arc<OwnedFd>> { - // SAFETY: `pidfd_open` takes the PID by value and does not read caller-owned - // memory. Flags are zero, which is valid. On success the returned descriptor is - // newly owned by this process and is immediately wrapped in `OwnedFd` below. - let fd = unsafe { libc::syscall(libc::SYS_pidfd_open, pid, 0) }; - if fd < 0 { - return None; - } - - // SAFETY: `fd` is non-negative and was just returned by `pidfd_open`, so it is - // an open descriptor owned by this process. `OwnedFd` takes sole ownership and - // will close it exactly once. - Some(Arc::new(unsafe { OwnedFd::from_raw_fd(fd as RawFd) })) - } - - /// Send `signal` to the process group `pgid`. - /// Returns true when the signal is delivered successfully. - pub fn kill_process_group(pgid: i32, signal: i32) -> bool { - // SAFETY: `kill` takes integer identifiers by value and does not access - // caller-owned memory. A negative PID is the POSIX process-group form. - unsafe { libc::kill(-pgid, signal) == 0 } - } - - /// Find processes whose `/proc/{pid}/exe` symlink resolves to exactly - /// `target`. - pub fn find_by_path(target: &str) -> Vec<Process> { - let mut matches = Vec::new(); - let Ok(entries) = fs::read_dir("/proc") else { - return matches; - }; - let target_os = OsStr::new(target); - for entry in entries.flatten() { - let name = entry.file_name(); - let Some(name_str) = name.to_str() else { - continue; - }; - let Ok(pid) = name_str.parse::<i32>() else { - continue; - }; - let exe_path = format!("/proc/{pid}/exe"); - let Ok(resolved) = fs::read_link(&exe_path) else { - continue; - }; - if resolved.as_os_str() == target_os - && let Some(process) = Process::from_pid(pid) - { - matches.push(process); - } - } - matches - } -} - -#[cfg(target_os = "macos")] -mod platform { - use std::{collections::HashSet, ptr}; - - use super::ProcessStatus; - - #[link(name = "proc", kind = "dylib")] - unsafe extern "C" { - fn proc_listchildpids(ppid: i32, buffer: *mut i32, buffersize: i32) -> i32; - fn proc_listallpids(buffer: *mut i32, buffersize: i32) -> i32; - fn proc_pidpath(pid: i32, buffer: *mut std::ffi::c_void, buffersize: u32) -> i32; - } - - /// macOS does not expose pidfds; identity is pinned via the kernel-reported - /// process start time so a recycled PID does not silently impersonate the - /// original target. - #[derive(Clone)] - pub struct Process { - pid: i32, - start_tvsec: u64, - start_tvusec: u64, - } - - impl Process { - pub fn from_pid(pid: i32) -> Option<Self> { - if pid <= 0 { - return None; - } - let info = read_bsdinfo(pid)?; - if i32::try_from(info.pbi_pid).ok()? != pid { - return None; - } - Some(Self { pid, start_tvsec: info.pbi_start_tvsec, start_tvusec: info.pbi_start_tvusec }) - } - - pub const fn pid(&self) -> i32 { - self.pid - } - - pub fn children(&self) -> Vec<Self> { - if self.live_bsdinfo().is_none() { - return Vec::new(); - } - - // SAFETY: Passing a null buffer with size 0 is the documented libproc query - // form for obtaining the byte count needed for child PIDs; libproc does not - // dereference the null pointer in this mode. - let bytes = unsafe { proc_listchildpids(self.pid, ptr::null_mut(), 0) }; - if bytes <= 0 { - return Vec::new(); - } - - let count = bytes as usize / size_of::<i32>(); - let mut buffer = vec![0i32; count]; - // SAFETY: `buffer` is valid for `buffer.len() * size_of::<i32>()` bytes and - // is properly aligned for `i32`; libproc writes at most the supplied size. - let actual = unsafe { - proc_listchildpids( - self.pid, - buffer.as_mut_ptr(), - (buffer.len() * size_of::<i32>()) as i32, - ) - }; - if actual <= 0 { - return Vec::new(); - } - - let child_count = ((actual as usize) / size_of::<i32>()).min(buffer.len()); - buffer[..child_count] - .iter() - .copied() - .filter_map(Self::from_pid) - .collect() - } - - pub fn parent_pid(&self) -> Option<i32> { - let info = self.live_bsdinfo()?; - i32::try_from(info.pbi_ppid).ok().filter(|ppid| *ppid > 0) - } - - pub fn args(&self) -> Vec<String> { - if self.live_bsdinfo().is_none() { - return Vec::new(); - } - process_args(self.pid) - } - - pub fn kill(&self, signal: i32) -> bool { - // Re-validate identity right before signaling. There is no atomic - // "kill iff start_time matches" primitive on macOS, so a vanishingly small - // window remains between this check and the syscall — but matching against - // the recorded `(pid, start_tvsec, start_tvusec)` triple eliminates the - // PID-reuse race in every practical case. - if self.live_bsdinfo().is_none() { - return false; - } - // SAFETY: `kill` takes integer identifiers by value and does not access - // caller-owned memory. - unsafe { libc::kill(self.pid, signal) == 0 } - } - - pub fn group_id(&self) -> Option<i32> { - let info = self.live_bsdinfo()?; - i32::try_from(info.pbi_pgid).ok().filter(|pgid| *pgid > 0) - } - - /// Walk the descendant tree in post-order (leaves first), de-duplicating - /// by PID so concurrent reparenting cannot trap us in a cycle. - pub fn descendants(&self) -> Vec<Self> { - let mut out = Vec::new(); - let mut visited = HashSet::new(); - visited.insert(self.pid); - self.descendants_into(&mut out, &mut visited); - out - } - - fn descendants_into(&self, out: &mut Vec<Self>, visited: &mut HashSet<i32>) { - for child in self.children() { - if visited.insert(child.pid) { - child.descendants_into(out, visited); - out.push(child); - } - } - } - - pub fn status(&self) -> ProcessStatus { - if self.live_bsdinfo().is_some() { - ProcessStatus::Running - } else { - ProcessStatus::Exited - } - } - - /// Returns the current `proc_bsdinfo` only if it still describes the same - /// process this reference was opened on — i.e. the start time has not - /// changed. - fn live_bsdinfo(&self) -> Option<libc::proc_bsdinfo> { - let info = read_bsdinfo(self.pid)?; - if info.pbi_start_tvsec == self.start_tvsec && info.pbi_start_tvusec == self.start_tvusec { - Some(info) - } else { - None - } - } - } - - /// Send `signal` to the process group `pgid`. - /// Returns true when the signal is delivered successfully. - pub fn kill_process_group(pgid: i32, signal: i32) -> bool { - // SAFETY: `kill` takes integer identifiers by value and does not access - // caller-owned memory. A negative PID is the POSIX process-group form. - unsafe { libc::kill(-pgid, signal) == 0 } - } - - const KERN_PROCARGS2: libc::c_int = 49; - - const PROC_PIDPATHINFO_MAXSIZE: usize = 4096; - - /// Find processes whose libproc-reported executable path equals `target`. - pub fn find_by_path(target: &str) -> Vec<Process> { - // SAFETY: Passing a null buffer with size 0 is the documented libproc query - // form for obtaining the byte count needed for all PIDs; libproc does not - // dereference the null pointer in this mode. - let bytes = unsafe { proc_listallpids(ptr::null_mut(), 0) }; - if bytes <= 0 { - return Vec::new(); - } - // macOS truncates the second `proc_listallpids` call's result tightly to - // the buffer size we report — even when the buffer is large enough on paper — - // so a near-fit buffer can silently lose ~half the pids. Pad generously. - let count = (bytes as usize) / size_of::<i32>(); - let cap = count.saturating_mul(4).max(2048); - let mut buffer = vec![0i32; cap]; - // SAFETY: `buffer` is valid for `buffer.len() * size_of::<i32>()` bytes and - // is properly aligned for `i32`; libproc writes at most the supplied size. - let actual = - unsafe { proc_listallpids(buffer.as_mut_ptr(), (buffer.len() * size_of::<i32>()) as i32) }; - if actual <= 0 { - return Vec::new(); - } - let pid_count = ((actual as usize) / size_of::<i32>()).min(buffer.len()); - - let mut path_buf = vec![0u8; PROC_PIDPATHINFO_MAXSIZE]; - let mut matches = Vec::new(); - for &pid in &buffer[..pid_count] { - if pid <= 0 { - continue; - } - // SAFETY: `path_buf` is valid for `path_buf.len()` bytes; libproc writes a - // NUL-terminated path no longer than the supplied capacity and returns the - // number of bytes written. - let len = unsafe { - proc_pidpath( - pid, - path_buf.as_mut_ptr().cast::<std::ffi::c_void>(), - path_buf.len() as u32, - ) - }; - if len <= 0 { - continue; - } - let path_bytes = &path_buf[..len as usize]; - let path_bytes = match path_bytes.iter().position(|byte| *byte == 0) { - Some(end) => &path_bytes[..end], - None => path_bytes, - }; - let Ok(path) = std::str::from_utf8(path_bytes) else { - continue; - }; - if path == target - && let Some(process) = Process::from_pid(pid) - { - matches.push(process); - } - } - matches - } - - fn read_bsdinfo(pid: i32) -> Option<libc::proc_bsdinfo> { - // SAFETY: `proc_bsdinfo` is a plain C data struct. Zero initialization is - // valid because every field is an integer or fixed-size integer array, and - // libproc fully overwrites the fields it reports on a successful call. - let mut info = unsafe { std::mem::zeroed::<libc::proc_bsdinfo>() }; - // SAFETY: `info` is a writable `proc_bsdinfo` buffer whose exact byte size is - // supplied to libproc. The PID, flavor, and arg are scalar values passed by - // value; libproc writes at most the supplied buffer size. - let actual = unsafe { - libc::proc_pidinfo( - pid, - libc::PROC_PIDTBSDINFO, - 0, - (&raw mut info).cast::<std::ffi::c_void>(), - size_of::<libc::proc_bsdinfo>() as i32, - ) - }; - if actual < size_of::<libc::proc_bsdinfo>() as i32 { - return None; - } - Some(info) - } - - fn process_args(pid: i32) -> Vec<String> { - let mut mib = [libc::CTL_KERN, KERN_PROCARGS2, pid]; - let mut size = 0usize; - // SAFETY: `mib` points to three initialized integers and the old-value buffer - // is null with a zero-length query, which is the documented `sysctl` sizing - // pattern. `size` is a valid out-parameter for the required byte count. - let sizing_ok = unsafe { - libc::sysctl( - mib.as_mut_ptr(), - mib.len() as u32, - ptr::null_mut(), - &raw mut size, - ptr::null_mut(), - 0, - ) - } == 0; - if !sizing_ok || size <= size_of::<libc::c_int>() { - return Vec::new(); - } - - let mut buffer = vec![0u8; size]; - // SAFETY: `mib` still points to three initialized integers. `buffer` is - // writable for `size` bytes, and `size` is provided as the in/out byte count. - let read_ok = unsafe { - libc::sysctl( - mib.as_mut_ptr(), - mib.len() as u32, - buffer.as_mut_ptr().cast::<std::ffi::c_void>(), - &raw mut size, - ptr::null_mut(), - 0, - ) - } == 0; - if !read_ok { - return Vec::new(); - } - buffer.truncate(size); - parse_macos_procargs(&buffer) - } - - fn parse_macos_procargs(buffer: &[u8]) -> Vec<String> { - // KERN_PROCARGS2 layout: `argc: i32 | exec_path: NUL-padded | argv[0..argc] | - // env[..]`. argc covers only argv, so we must skip the exec_path NUL padding - // and stop after exactly argc entries — otherwise environment variables leak - // into the arg list (each NUL-terminated env=value is indistinguishable from - // an arg). - let argc_size = size_of::<libc::c_int>(); - if buffer.len() <= argc_size { - return Vec::new(); - } - - let argc_bytes: [u8; 4] = match buffer[..argc_size].try_into() { - Ok(bytes) => bytes, - Err(_) => return Vec::new(), - }; - let argc = libc::c_int::from_ne_bytes(argc_bytes); - if argc <= 0 { - return Vec::new(); - } - - let mut offset = argc_size; - while offset < buffer.len() && buffer[offset] != 0 { - offset += 1; - } - while offset < buffer.len() && buffer[offset] == 0 { - offset += 1; - } - - let mut args = Vec::with_capacity(argc as usize); - while offset < buffer.len() && args.len() < argc as usize { - let end = buffer[offset..] - .iter() - .position(|byte| *byte == 0) - .map_or(buffer.len(), |position| offset + position); - if end == offset { - break; - } - args.push(String::from_utf8_lossy(&buffer[offset..end]).into_owned()); - offset = end + 1; - } - args - } -} -#[cfg(target_os = "windows")] -mod platform { - use std::{ - collections::{HashMap, HashSet}, - ffi::c_void, - mem, - sync::Arc, - }; - - use smallvec::SmallVec; - - use super::ProcessStatus; - - #[repr(C)] - #[allow(non_snake_case, reason = "Windows PROCESSENTRY32W field names must match Win32 ABI")] - struct PROCESSENTRY32W { - dwSize: u32, - cntUsage: u32, - th32ProcessID: u32, - th32DefaultHeapID: usize, - th32ModuleID: u32, - cntThreads: u32, - th32ParentProcessID: u32, - pcPriClassBase: i32, - dwFlags: u32, - szExeFile: [u16; 260], - } - - #[repr(C)] - struct ProcessBasicInformation { - exit_status: i32, - peb_base_address: usize, - affinity_mask: usize, - base_priority: i32, - unique_process_id: usize, - inherited_from_unique_process_id: usize, - } - - #[repr(C)] - #[derive(Clone, Copy)] - struct UnicodeString { - length: u16, - maximum_length: u16, - buffer: usize, - } - - #[repr(C)] - #[derive(Clone, Copy)] - struct PebPartial { - reserved1: [u8; 2], - being_debugged: u8, - reserved2: [u8; 1], - reserved3: [usize; 2], - loader: usize, - process_parameters: usize, - } - - #[repr(C)] - #[derive(Clone, Copy)] - struct UserProcessParametersPartial { - reserved1: [u8; 16], - reserved2: [usize; 10], - image_path_name: UnicodeString, - command_line: UnicodeString, - } - - #[repr(C)] - #[derive(Clone, Copy, Default)] - struct Filetime { - dw_low_date_time: u32, - dw_high_date_time: u32, - } - - type Handle = *mut c_void; - type NtStatus = i32; - const INVALID_HANDLE_VALUE: Handle = -1isize as Handle; - const PROCESS_QUERY_INFORMATION: u32 = 0x0400; - const PROCESS_VM_READ: u32 = 0x0010; - const PROCESS_BASIC_INFORMATION_CLASS: u32 = 0; - const STATUS_SUCCESS: NtStatus = 0; - const TH32CS_SNAPPROCESS: u32 = 0x00000002; - const PROCESS_TERMINATE: u32 = 0x0001; - const PROCESS_QUERY_LIMITED_INFORMATION: u32 = 0x1000; - const SYNCHRONIZE: u32 = 0x00100000; - const PROCESS_REFERENCE_ACCESS: u32 = - PROCESS_TERMINATE | PROCESS_QUERY_LIMITED_INFORMATION | SYNCHRONIZE; - const WAIT_OBJECT_0: u32 = 0; - - #[link(name = "kernel32")] - unsafe extern "system" { - fn CreateToolhelp32Snapshot(dwFlags: u32, th32ProcessID: u32) -> Handle; - fn Process32FirstW(hSnapshot: Handle, lppe: *mut PROCESSENTRY32W) -> i32; - fn Process32NextW(hSnapshot: Handle, lppe: *mut PROCESSENTRY32W) -> i32; - fn CloseHandle(hObject: Handle) -> i32; - fn OpenProcess(dwDesiredAccess: u32, bInheritHandle: i32, dwProcessId: u32) -> Handle; - fn TerminateProcess(hProcess: Handle, uExitCode: u32) -> i32; - fn QueryFullProcessImageNameW( - hProcess: Handle, - dwFlags: u32, - lpExeName: *mut u16, - lpdwSize: *mut u32, - ) -> i32; - fn WaitForSingleObject(hHandle: Handle, dwMilliseconds: u32) -> u32; - fn GetProcessTimes( - hProcess: Handle, - lpCreationTime: *mut Filetime, - lpExitTime: *mut Filetime, - lpKernelTime: *mut Filetime, - lpUserTime: *mut Filetime, - ) -> i32; - fn ReadProcessMemory( - hProcess: Handle, - lpBaseAddress: *const c_void, - lpBuffer: *mut c_void, - nSize: usize, - lpNumberOfBytesRead: *mut usize, - ) -> i32; - fn LocalFree(hMem: Handle) -> Handle; - } - - #[link(name = "shell32")] - unsafe extern "system" { - fn CommandLineToArgvW(lpCmdLine: *const u16, pNumArgs: *mut i32) -> *mut *mut u16; - } - - #[link(name = "ntdll")] - unsafe extern "system" { - fn NtQueryInformationProcess( - ProcessHandle: Handle, - ProcessInformationClass: u32, - ProcessInformation: *mut c_void, - ProcessInformationLength: u32, - ReturnLength: *mut u32, - ) -> NtStatus; - } - - struct OwnedHandle { - raw: isize, - } - - impl OwnedHandle { - fn from_raw(raw: Handle) -> Option<Self> { - if raw.is_null() || raw == INVALID_HANDLE_VALUE { - None - } else { - Some(Self { raw: raw as isize }) - } - } - - fn as_raw(&self) -> Handle { - self.raw as Handle - } - } - - impl Drop for OwnedHandle { - fn drop(&mut self) { - // SAFETY: `self.raw` was returned by a successful Win32 handle-producing - // function and stored only in this `OwnedHandle`. `Drop` runs once, so this - // closes the owned handle exactly once and no code uses it afterward. - let _ = unsafe { CloseHandle(self.as_raw()) }; - } - } - - #[derive(Clone)] - /// Stable Windows process reference backed by an owned process handle plus - /// the kernel-reported creation time, which pins identity even if the PID is - /// recycled while we hold the handle. - pub struct Process { - pid: i32, - handle: Arc<OwnedHandle>, - creation_time: u64, - } - - impl Process { - pub fn from_pid(pid: i32) -> Option<Self> { - if pid <= 0 { - return None; - } - let pid_u32 = u32::try_from(pid).ok()?; - let handle = open_process(pid_u32, PROCESS_REFERENCE_ACCESS)?; - let creation_time = process_creation_time(handle.as_raw())?; - Some(Self { pid, handle, creation_time }) - } - - pub const fn pid(&self) -> i32 { - self.pid - } - - pub fn parent_pid(&self) -> Option<i32> { - process_basic_information(self.handle.as_raw()) - .and_then(|info| i32::try_from(info.inherited_from_unique_process_id).ok()) - .filter(|pid| *pid > 0) - } - - pub fn args(&self) -> Vec<String> { - process_command_line(self) - .as_deref() - .map(split_windows_command_line) - .unwrap_or_default() - } - - pub fn children(&self) -> Vec<Self> { - let tree = build_process_tree(); - Self::children_from_tree(self.pid, &tree) - } - - /// Walk the entire descendant tree using a single Toolhelp snapshot. - /// - /// `children()` recursing per-node would re-snapshot the whole process - /// table for every visited descendant, making tree termination - /// `O(N · D)` snapshots. One snapshot per termination wave is enough. - pub fn descendants(&self) -> Vec<Self> { - let tree = build_process_tree(); - let Ok(root) = u32::try_from(self.pid) else { - return Vec::new(); - }; - let mut visited: HashSet<u32> = HashSet::new(); - visited.insert(root); - let mut out = Vec::new(); - Self::collect_descendants_from_tree(root, &tree, &mut visited, &mut out); - out - } - - fn children_from_tree(pid: i32, tree: &HashMap<u32, SmallVec<[u32; 4]>>) -> Vec<Self> { - let Ok(pid_u32) = u32::try_from(pid) else { - return Vec::new(); - }; - tree - .get(&pid_u32) - .into_iter() - .flatten() - .filter_map(|&child_pid| { - let child = Self::from_pid(i32::try_from(child_pid).ok()?)?; - (child.status() == ProcessStatus::Running).then_some(child) - }) - .collect() - } - - fn collect_descendants_from_tree( - parent: u32, - tree: &HashMap<u32, SmallVec<[u32; 4]>>, - visited: &mut HashSet<u32>, - out: &mut Vec<Self>, - ) { - let Some(children) = tree.get(&parent) else { - return; - }; - for &child_pid in children { - if !visited.insert(child_pid) { - continue; - } - let Ok(child_pid_i) = i32::try_from(child_pid) else { - continue; - }; - let Some(child) = Self::from_pid(child_pid_i) else { - continue; - }; - if child.status() != ProcessStatus::Running { - continue; - } - // Post-order: collect grandchildren first so leaves are signalled before - // their parents during tree termination. - Self::collect_descendants_from_tree(child_pid, tree, visited, out); - out.push(child); - } - } - - pub fn kill(&self, _signal: i32) -> bool { - // The handle pins the original kernel process object even after the PID is - // recycled, so `TerminateProcess` cannot accidentally hit a different - // process. SAFETY: `self.handle` is an owned process handle opened with - // `PROCESS_TERMINATE` access and remains valid for the duration of this - // call. The exit code is passed by value. - unsafe { TerminateProcess(self.handle.as_raw(), 1) != 0 } - } - - pub const fn group_id(&self) -> Option<i32> { - None - } - - pub fn status(&self) -> ProcessStatus { - // `WaitForSingleObject` on a process handle opened with `SYNCHRONIZE` is - // the definitive liveness probe: the handle becomes signalled iff the - // process has exited. This avoids the `STILL_ACTIVE == 259` pitfall in - // `GetExitCodeProcess`, where a process that legitimately exits with code - // 259 is indistinguishable from a still-running one. - // - // SAFETY: `self.handle` is an owned process handle opened with - // `SYNCHRONIZE` access. A zero timeout makes this a non-blocking probe. - let result = unsafe { WaitForSingleObject(self.handle.as_raw(), 0) }; - if result == WAIT_OBJECT_0 { - ProcessStatus::Exited - } else { - ProcessStatus::Running - } - } - } - - fn process_basic_information(handle: Handle) -> Option<ProcessBasicInformation> { - let mut info = ProcessBasicInformation { - exit_status: 0, - peb_base_address: 0, - affinity_mask: 0, - base_priority: 0, - unique_process_id: 0, - inherited_from_unique_process_id: 0, - }; - let mut returned = 0u32; - // SAFETY: `handle` is a valid process handle. `info` is writable for exactly - // `size_of::<ProcessBasicInformation>()` bytes, and `returned` is a valid - // optional out-parameter for the byte count. - let status = unsafe { - NtQueryInformationProcess( - handle, - PROCESS_BASIC_INFORMATION_CLASS, - (&raw mut info).cast::<c_void>(), - mem::size_of::<ProcessBasicInformation>() as u32, - &raw mut returned, - ) - }; - (status == STATUS_SUCCESS).then_some(info) - } - - fn process_command_line(process: &Process) -> Option<String> { - let pid_u32 = u32::try_from(process.pid).ok()?; - let read_handle = open_process(pid_u32, PROCESS_QUERY_INFORMATION | PROCESS_VM_READ)?; - // PID-reuse defense: `OpenProcess` resolves a PID to *whichever* process owns - // it right now, which need not be the one our original handle pinned. Compare - // the freshly opened handle's creation time against the recorded value to - // reject reads from an unrelated process that happens to share the PID. - if process_creation_time(read_handle.as_raw())? != process.creation_time { - return None; - } - let info = process_basic_information(read_handle.as_raw())?; - let peb: PebPartial = read_remote(read_handle.as_raw(), info.peb_base_address)?; - if peb.process_parameters == 0 { - return None; - } - let params: UserProcessParametersPartial = - read_remote(read_handle.as_raw(), peb.process_parameters)?; - read_remote_unicode_string(read_handle.as_raw(), params.command_line) - } - - fn process_creation_time(handle: Handle) -> Option<u64> { - let mut creation = Filetime::default(); - let mut exit = Filetime::default(); - let mut kernel = Filetime::default(); - let mut user = Filetime::default(); - // SAFETY: `handle` is a valid process handle opened with at least - // `PROCESS_QUERY_LIMITED_INFORMATION`. All four out-parameters point to - // initialized, writable `Filetime` values that live until the call returns. - let ok = unsafe { - GetProcessTimes(handle, &raw mut creation, &raw mut exit, &raw mut kernel, &raw mut user) - != 0 - }; - if !ok { - return None; - } - Some((u64::from(creation.dw_high_date_time) << 32) | u64::from(creation.dw_low_date_time)) - } - - fn read_remote<T: Copy>(handle: Handle, address: usize) -> Option<T> { - if address == 0 { - return None; - } - let mut value = mem::MaybeUninit::<T>::uninit(); - let mut bytes_read = 0usize; - // SAFETY: `handle` is opened with `PROCESS_VM_READ`. `address` comes from - // kernel-reported process structures for that same process. `value` points to - // uninitialized local storage large enough for `T`, and `bytes_read` is a valid - // out-parameter. The value is only assumed initialized after the OS reports a - // full-size successful read. - let ok = unsafe { - ReadProcessMemory( - handle, - address as *const c_void, - value.as_mut_ptr().cast::<c_void>(), - mem::size_of::<T>(), - &raw mut bytes_read, - ) != 0 - }; - if ok && bytes_read == mem::size_of::<T>() { - // SAFETY: The successful `ReadProcessMemory` call above initialized exactly - // `size_of::<T>()` bytes in `value`. - Some(unsafe { value.assume_init() }) - } else { - None - } - } - - fn read_remote_unicode_string(handle: Handle, value: UnicodeString) -> Option<String> { - if value.length == 0 || value.buffer == 0 || value.length % 2 != 0 { - return None; - } - let code_units = usize::from(value.length) / size_of::<u16>(); - let mut buffer = vec![0u16; code_units]; - let mut bytes_read = 0usize; - // SAFETY: `handle` is opened with `PROCESS_VM_READ`. `value.buffer` and - // `value.length` come from the remote process' own `UNICODE_STRING`. `buffer` - // is writable for exactly `value.length` bytes, and `bytes_read` is a valid - // out-parameter. The string is decoded only after a full successful read. - let ok = unsafe { - ReadProcessMemory( - handle, - value.buffer as *const c_void, - buffer.as_mut_ptr().cast::<c_void>(), - usize::from(value.length), - &raw mut bytes_read, - ) != 0 - }; - if ok && bytes_read == usize::from(value.length) { - Some(String::from_utf16_lossy(&buffer)) - } else { - None - } - } - - fn split_windows_command_line(command_line: &str) -> Vec<String> { - use std::os::windows::ffi::OsStringExt; - - let mut wide: Vec<u16> = command_line.encode_utf16().chain([0]).collect(); - let mut argc = 0i32; - // SAFETY: `wide` is a local, NUL-terminated UTF-16 buffer that remains alive - // for the duration of the call. `argc` is a valid out-parameter. The returned - // argv block is released with `LocalFree` below as required by - // `CommandLineToArgvW`. - let argv = unsafe { CommandLineToArgvW(wide.as_mut_ptr(), &raw mut argc) }; - if argv.is_null() || argc <= 0 { - return Vec::new(); - } - let argc = argc as usize; - // SAFETY: `CommandLineToArgvW` returned a non-null pointer to `argc` argument - // pointers, valid until freed with `LocalFree`. - let pointers = unsafe { std::slice::from_raw_parts(argv, argc) }; - let args = pointers - .iter() - .filter_map(|&arg| { - if arg.is_null() { - return None; - } - let mut len = 0usize; - // SAFETY: Each pointer in the argv block is a NUL-terminated UTF-16 - // string owned by the argv block and valid until `LocalFree` below. - while unsafe { *arg.add(len) } != 0 { - len += 1; - } - // SAFETY: The loop above found the terminating NUL, so the preceding - // `len` code units form a valid readable slice. - let slice = unsafe { std::slice::from_raw_parts(arg, len) }; - Some( - std::ffi::OsString::from_wide(slice) - .to_string_lossy() - .into_owned(), - ) - }) - .collect(); - // SAFETY: `argv` is the allocation returned by `CommandLineToArgvW` and has - // not been freed yet. No pointers into it are used after this call. - let _ = unsafe { LocalFree(argv.cast::<c_void>()) }; - args - } - - fn open_process(pid: u32, access: u32) -> Option<Arc<OwnedHandle>> { - // SAFETY: `OpenProcess` takes the PID and access mask by value and does not - // dereference caller-owned memory. Handle inheritance is disabled. Identity - // is established by the caller (typically `Process::from_pid`) capturing the - // creation time immediately after a successful open and re-checking it on - // every subsequent operation that re-resolves the PID. - let handle = unsafe { OpenProcess(access, 0, pid) }; - OwnedHandle::from_raw(handle).map(Arc::new) - } - - fn create_process_snapshot() -> Option<OwnedHandle> { - // SAFETY: The process snapshot API takes flags and a process ID by value and - // does not dereference caller-owned memory. PID zero requests all processes. - let snapshot = unsafe { CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0) }; - OwnedHandle::from_raw(snapshot) - } - - fn process_entry() -> PROCESSENTRY32W { - PROCESSENTRY32W { - dwSize: mem::size_of::<PROCESSENTRY32W>() as u32, - cntUsage: 0, - th32ProcessID: 0, - th32DefaultHeapID: 0, - th32ModuleID: 0, - cntThreads: 0, - th32ParentProcessID: 0, - pcPriClassBase: 0, - dwFlags: 0, - szExeFile: [0; 260], - } - } - - /// Build a map of `parent_pid` -> [`child_pids`] for all processes. - fn build_process_tree() -> HashMap<u32, SmallVec<[u32; 4]>> { - let mut tree: HashMap<u32, SmallVec<[u32; 4]>> = HashMap::new(); - let Some(snapshot) = create_process_snapshot() else { - return tree; - }; - - let mut entry = process_entry(); - // SAFETY: `snapshot` is a valid Toolhelp snapshot handle. `entry` points to a - // writable `PROCESSENTRY32W` whose `dwSize` field was initialized to the exact - // ABI size before the call. - if unsafe { Process32FirstW(snapshot.as_raw(), &raw mut entry) } == 0 { - return tree; - } - - loop { - tree - .entry(entry.th32ParentProcessID) - .or_default() - .push(entry.th32ProcessID); - - // SAFETY: `snapshot` remains a valid Toolhelp snapshot handle, and `entry` - // remains a writable `PROCESSENTRY32W` with its ABI size preserved. - if unsafe { Process32NextW(snapshot.as_raw(), &raw mut entry) } == 0 { - break; - } - } - - tree - } - - /// Process groups are not exposed on Windows. - /// Always returns `false`. - pub const fn kill_process_group(_pgid: i32, _signal: i32) -> bool { - false - } - - /// Find processes whose `QueryFullProcessImageNameW` result equals `target`. - pub fn find_by_path(target: &str) -> Vec<Process> { - use std::{ffi::OsString, os::windows::ffi::OsStringExt}; - - let mut matches = Vec::new(); - let Some(snapshot) = create_process_snapshot() else { - return matches; - }; - - let mut entry = process_entry(); - let mut buf = vec![0u16; 32_768]; - let target = OsString::from(target); - - // SAFETY: `snapshot` is a valid Toolhelp snapshot handle. `entry` points to a - // writable `PROCESSENTRY32W` whose `dwSize` field was initialized to the exact - // ABI size before the call. - if unsafe { Process32FirstW(snapshot.as_raw(), &raw mut entry) } == 0 { - return matches; - } - - loop { - let pid = entry.th32ProcessID; - if let Some(handle) = open_process(pid, PROCESS_QUERY_LIMITED_INFORMATION) { - let mut size = buf.len() as u32; - // SAFETY: `handle` was opened with query access and remains valid for the - // call. `buf` is writable for `size` UTF-16 code units, and `size` is a valid - // in/out parameter initialized to that capacity. - let ok = unsafe { - QueryFullProcessImageNameW(handle.as_raw(), 0, buf.as_mut_ptr(), &raw mut size) != 0 - }; - if ok { - let path = OsString::from_wide(&buf[..size as usize]); - if path == target - && let Some(process) = Process::from_pid(i32::try_from(pid).unwrap_or_default()) - { - matches.push(process); - } - } - } - - // SAFETY: `snapshot` remains a valid Toolhelp snapshot handle, and `entry` - // remains a writable `PROCESSENTRY32W` with its ABI size preserved. - if unsafe { Process32NextW(snapshot.as_raw(), &raw mut entry) } == 0 { - break; - } - } - - matches } } @@ -1238,7 +67,7 @@ mod platform { #[napi] #[derive(Clone)] pub struct Process { - inner: platform::Process, + inner: core_process::Process, } #[napi] @@ -1247,13 +76,13 @@ impl Process { /// Open a stable process reference from a PID. #[napi] pub fn from_pid(pid: i32) -> Option<Process> { - platform::Process::from_pid(pid).map(Self::from_inner) + core_process::Process::from_pid(pid).map(Self::from_inner) } /// Open stable process references whose executable path matches exactly. #[napi] pub fn from_path(path: String) -> Vec<Process> { - platform::find_by_path(&path) + core_process::Process::from_path(path) .into_iter() .map(Self::from_inner) .collect() @@ -1268,7 +97,7 @@ impl Process { /// Parent process id for this process, when available. #[napi(getter)] pub fn ppid(&self) -> Option<i32> { - self.inner.parent_pid() + self.inner.ppid() } /// Launch arguments for this process. @@ -1285,7 +114,7 @@ impl Process { /// hard-kill signal. #[napi] pub fn kill_tree(&self, signal: Option<i32>) -> u32 { - self.signal_tree(signal.unwrap_or(KILL_SIGNAL)) + self.inner.kill_tree(signal) } /// Gracefully terminate this process and its descendants. @@ -1303,11 +132,12 @@ impl Process { let graceful_ms = options.graceful_ms.unwrap_or(1000); let timeout_ms = options.timeout_ms.unwrap_or(5000); let ct = task::CancelToken::new(None, options.signal); - let process = self.clone(); + let process = self.inner.clone(); task::future(env, "process.terminate", async move { process - .terminate_tree(group, graceful_ms, timeout_ms, ct) + .terminate_tree(group, graceful_ms, timeout_ms, ct.into_core()) .await + .map_err(|err| napi::Error::from_reason(err.to_string())) }) } @@ -1325,9 +155,12 @@ impl Process { let timeout = options .timeout_ms .map(|ms| Duration::from_millis(u64::from(ms))); - let process = self.clone(); + let process = self.inner.clone(); task::future(env, "process.wait_for_exit", async move { - wait_for_exit(&process, &[], timeout, ct).await + process + .wait_for_exit(timeout, ct.into_core()) + .await + .map_err(|err| napi::Error::from_reason(err.to_string())) }) } @@ -1351,207 +184,12 @@ impl Process { /// Current status of this process reference. #[napi] pub fn status(&self) -> ProcessStatus { - self.inner.status() + self.inner.status().into() } } impl Process { - const fn from_inner(inner: platform::Process) -> Self { + const fn from_inner(inner: core_process::Process) -> Self { Self { inner } } - - /// Walk the live descendant tree from scratch. Cheap and idempotent — call - /// it again before each signal wave so grandchildren spawned during a grace - /// period are not missed. - fn live_descendants(&self) -> Vec<Self> { - self - .inner - .descendants() - .into_iter() - .map(Self::from_inner) - .collect() - } - - fn signal_tree(&self, signal: i32) -> u32 { - let descendants = self.live_descendants(); - let mut signaled = 0u32; - // If self leads its own process group, also signal the group — this catches - // grandchildren reparented to init when their immediate parent died inside - // the descendant walk. - if let Some(pgid) = self.inner.group_id() - && pgid == self.inner.pid() - { - let _ = kill_process_group(pgid, signal); - } - for child in &descendants { - if child.inner.kill(signal) { - signaled += 1; - } - } - if self.inner.kill(signal) { - signaled += 1; - } - signaled - } - - async fn terminate_tree( - &self, - group: bool, - graceful_ms: i32, - timeout_ms: u32, - ct: task::CancelToken, - ) -> Result<bool> { - if self.status() != ProcessStatus::Running { - return Ok(true); - } - - let process_group = if group { self.group_id() } else { None }; - - // Polite wave: SIGTERM the group, every live descendant, then the root. - if let Some(pgid) = process_group { - let _ = kill_process_group(pgid, TERM_SIGNAL); - } - let mut descendants = self.live_descendants(); - for child in &descendants { - let _ = child.inner.kill(TERM_SIGNAL); - } - let _ = self.inner.kill(TERM_SIGNAL); - - // Optional grace wait. A negative `graceful_ms` skips the wait entirely - // (we still emit the polite signal so cleanup handlers can run before KILL). - if graceful_ms >= 0 { - let exited = wait_for_exit( - self, - &descendants, - Some(Duration::from_millis(graceful_ms as u64)), - ct.clone(), - ) - .await?; - if exited { - return Ok(true); - } - } - - // Hard wave. Re-walk the tree so any grandchild spawned during the grace - // period — or any process re-parented to the root — is signalled too. - if let Some(pgid) = process_group { - let _ = kill_process_group(pgid, KILL_SIGNAL); - } - descendants = self.live_descendants(); - for child in &descendants { - let _ = child.inner.kill(KILL_SIGNAL); - } - let _ = self.inner.kill(KILL_SIGNAL); - - wait_for_exit(self, &descendants, Some(Duration::from_millis(u64::from(timeout_ms))), ct) - .await - } -} - -async fn wait_for_exit( - root: &Process, - descendants: &[Process], - timeout: Option<Duration>, - ct: task::CancelToken, -) -> Result<bool> { - ct.heartbeat()?; - if root.status() != ProcessStatus::Running - && descendants - .iter() - .all(|process| process.status() != ProcessStatus::Running) - { - return Ok(true); - } - - let poll_interval = Duration::from_millis(50); - let mut elapsed = Duration::ZERO; - while timeout.is_none_or(|limit| elapsed < limit) { - let sleep_for = - timeout.map_or(poll_interval, |limit| limit.saturating_sub(elapsed).min(poll_interval)); - if sleep_for.is_zero() { - break; - } - ct.heartbeat()?; - tokio::time::sleep(sleep_for).await; - elapsed += sleep_for; - - if root.status() != ProcessStatus::Running - && descendants - .iter() - .all(|process| process.status() != ProcessStatus::Running) - { - return Ok(true); - } - } - - Ok(false) -} - -/// Send `signal` to the process group `pgid`. -/// Returns false when process groups are unsupported on the platform. -#[allow(clippy::missing_const_for_fn, reason = "Dispatches to platform-specific implementation")] -pub fn kill_process_group(pgid: i32, signal: i32) -> bool { - platform::kill_process_group(pgid, signal) -} - -/// POSIX `SIGTERM` / Windows polite termination sentinel. -pub const TERM_SIGNAL: i32 = 15; - -/// POSIX `SIGKILL` / Windows hard-termination sentinel. -pub const KILL_SIGNAL: i32 = 9; - -/// A collection of process groups and process trees scheduled for -/// termination together. -/// -/// Built incrementally from job records or PTY metadata, then signalled -/// in escalating waves (typically `TERM_SIGNAL` followed by -/// `KILL_SIGNAL` after a grace period). Process-group calls are no-ops -/// on platforms that do not expose process groups. -#[derive(Default)] -pub struct TerminationTargets { - pgids: Vec<i32>, - processes: Vec<Process>, - seen_pids: HashSet<i32>, -} - -impl TerminationTargets { - /// Create an empty target set. - pub fn new() -> Self { - Self::default() - } - - /// Record a process group id. Duplicates are ignored. - pub fn add_pgid(&mut self, pgid: i32) { - if pgid > 0 && !self.pgids.contains(&pgid) { - self.pgids.push(pgid); - } - } - - /// Record a pid. Duplicates are ignored. If the pid is alive, opens - /// a stable [`Process`] reference so the descendant tree can be - /// killed even if the original pid is reused later. - pub fn add_pid(&mut self, pid: i32) { - if self.seen_pids.insert(pid) - && let Some(process) = Process::from_pid(pid) - { - self.processes.push(process); - } - } - - /// True when no targets have been recorded. - pub const fn is_empty(&self) -> bool { - self.pgids.is_empty() && self.processes.is_empty() - } - - /// Send `signal` to every recorded target. Failures are swallowed: - /// targets routinely exit between collection and signalling, and - /// the caller's policy is "best effort". - pub fn signal(&self, signal: i32) { - for &pgid in &self.pgids { - let _ = kill_process_group(pgid, signal); - } - for process in &self.processes { - let _ = process.signal_tree(signal); - } - } } diff --git a/crates/pi-natives/src/shell.rs b/crates/pi-natives/src/shell.rs index ba216013e..5489764f9 100644 --- a/crates/pi-natives/src/shell.rs +++ b/crates/pi-natives/src/shell.rs @@ -1,91 +1,59 @@ //! Brush-based shell execution exported via N-API. -//! -//! # Overview -//! Executes shell commands in a non-interactive brush-core shell, streaming -//! output back to JavaScript via a threadsafe callback. -//! -//! # Example -//! ```ignore -//! const shell = new natives.Shell(); -//! const result = await shell.run({ command: "ls" }, (chunk) => { -//! console.log(chunk); -//! }); -//! ``` -#[cfg(windows)] -use std::collections::HashSet; -use std::{ - collections::HashMap, - fs, - io::{self, Write}, - str, - sync::Arc, - time::Duration, -}; +use std::{collections::HashMap, sync::Arc}; -#[cfg(windows)] -mod windows; - -mod minimizer; - -use brush_builtins::{BuiltinSet, default_builtins}; -use brush_core::{ - ExecutionContext, ExecutionControlFlow, ExecutionExitCode, ExecutionResult, ProcessGroupPolicy, - ProfileLoadBehavior, RcLoadBehavior, Shell as BrushShell, ShellValue, ShellVariable, SourceInfo, - builtins, - env::EnvironmentScope, - openfiles::{self, OpenFile, OpenFiles}, -}; -use clap::Parser; use napi::{ + Env, Result, bindgen_prelude::*, threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode}, - tokio::{ - self, - sync::{Mutex as TokioMutex, mpsc}, - time, - }, + tokio::sync::mpsc, }; use napi_derive::napi; -#[cfg(not(unix))] -use tokio::io::AsyncReadExt as _; -use tokio_util::sync::CancellationToken; -#[cfg(windows)] -use windows::configure_windows_path; +use pi_shell::{ + MinimizerResult as CoreMinimizerResult, Shell as CoreShell, + ShellExecuteOptions as CoreShellExecuteOptions, ShellOptions as CoreShellOptions, + ShellRunOptions as CoreShellRunOptions, ShellRunResult as CoreShellRunResult, + execute_shell as core_execute_shell, minimizer, +}; -use crate::{ps, task}; +use crate::task; -struct ShellSessionCore { - shell: BrushShell, +/// N-API opt-in handle for the minimizer. +#[napi(object)] +#[derive(Debug, Clone, Default)] +pub struct MinimizerOptions { + /// Master switch. Absent / false = disabled. + pub enabled: Option<bool>, + /// Optional path to a TOML settings file whose values override + /// field-level defaults. `~` is expanded. + pub settings_path: Option<String>, + /// Optional xxHash64 digest (hex) of the settings file contents. When + /// supplied, the engine refuses to honor a settings file whose hash does + /// not match — a lightweight trust gate for agent-controllable paths. + pub settings_hash: Option<String>, + /// Opt-in allowlist of program names (e.g. `"git"`). When empty or + /// absent, all built-in filters are active. + pub only: Option<Vec<String>>, + /// Program names explicitly excluded from minimization. + pub except: Option<Vec<String>>, + /// Maximum captured bytes per command before the engine falls back to + /// the raw, un-minimized output. Default 4 MiB. + pub max_capture_bytes: Option<u32>, } -#[derive(Clone, Default)] -struct ShellAbortState(Arc<TokioMutex<Option<task::AbortToken>>>); - -impl ShellAbortState { - async fn set(&self, abort_token: task::AbortToken) { - *self.0.lock().await = Some(abort_token); - } - - async fn clear(&self) { - *self.0.lock().await = None; - } - - async fn abort(&self) { - let abort_token = self.0.lock().await.clone(); - if let Some(abort_token) = abort_token { - abort_token.abort(task::AbortReason::Signal); +impl From<MinimizerOptions> for minimizer::MinimizerOptions { + fn from(value: MinimizerOptions) -> Self { + Self { + enabled: value.enabled, + settings_path: value.settings_path, + settings_hash: value.settings_hash, + only: value.only, + except: value.except, + max_capture_bytes: value.max_capture_bytes, } } } -#[derive(Clone)] -struct ShellConfig { - session_env: Option<HashMap<String, String>>, - snapshot_path: Option<String>, - minimizer: Option<minimizer::MinimizerConfig>, -} - /// Options for configuring a persistent shell session. #[napi(object)] pub struct ShellOptions { @@ -94,19 +62,17 @@ pub struct ShellOptions { /// Optional snapshot file to source on session creation. pub snapshot_path: Option<String>, /// Optional per-command output minimizer configuration. - pub minimizer: Option<minimizer::MinimizerOptions>, + pub minimizer: Option<MinimizerOptions>, } -/// Options for running a shell command (internal, lifetime-free). -struct ShellRunConfig { - /// Command string to execute in the shell. - command: String, - /// Working directory for the command. - cwd: Option<String>, - /// Environment variables to apply for this command only. - env: Option<HashMap<String, String>>, - /// Resolved output minimizer config for this command. - minimizer: Option<minimizer::MinimizerConfig>, +impl From<ShellOptions> for CoreShellOptions { + fn from(value: ShellOptions) -> Self { + Self { + session_env: value.session_env, + snapshot_path: value.snapshot_path, + minimizer: value.minimizer.map(Into::into), + } + } } /// Options for running a shell command. @@ -124,6 +90,27 @@ pub struct ShellRunOptions<'env> { pub signal: Option<Unknown<'env>>, } +/// Options for executing a shell command via brush-core. +#[napi(object)] +pub struct ShellExecuteOptions<'env> { + /// Command string to execute in the shell. + pub command: String, + /// Working directory for the command. + pub cwd: Option<String>, + /// Environment variables to apply for this command only. + pub env: Option<HashMap<String, String>>, + /// Environment variables to apply once per session. + pub session_env: Option<HashMap<String, String>>, + /// Timeout in milliseconds before cancelling the command. + pub timeout_ms: Option<u32>, + /// Optional snapshot file to source on session creation. + pub snapshot_path: Option<String>, + /// Optional per-command output minimizer configuration. + pub minimizer: Option<MinimizerOptions>, + /// Abort signal for cancelling the operation. + pub signal: Option<Unknown<'env>>, +} + /// Telemetry for a single minimization. /// /// Surfaced when the minimizer actually rewrote the command's output. The @@ -148,6 +135,18 @@ pub struct MinimizerResult { pub output_bytes: u32, } +impl From<CoreMinimizerResult> for MinimizerResult { + fn from(value: CoreMinimizerResult) -> Self { + Self { + filter: value.filter, + text: value.text, + original_text: value.original_text, + input_bytes: value.input_bytes, + output_bytes: value.output_bytes, + } + } +} + /// Result of running a shell command. #[napi(object)] pub struct ShellRunResult { @@ -164,40 +163,31 @@ pub struct ShellRunResult { pub minimized: Option<MinimizerResult>, } +impl From<CoreShellRunResult> for ShellRunResult { + fn from(value: CoreShellRunResult) -> Self { + Self { + exit_code: value.exit_code, + cancelled: value.cancelled, + timed_out: value.timed_out, + minimized: value.minimized.map(Into::into), + } + } +} + /// Persistent brush-core shell session. #[napi] pub struct Shell { - session: Arc<TokioMutex<Option<ShellSessionCore>>>, - abort_state: ShellAbortState, - config: ShellConfig, + inner: Arc<CoreShell>, } #[napi] impl Shell { - #[napi(constructor)] /// Create a new shell session from optional configuration. /// /// The options set session-scoped environment variables and a snapshot path. + #[napi(constructor)] pub fn new(options: Option<ShellOptions>) -> Self { - let config = match options { - None => ShellConfig { session_env: None, snapshot_path: None, minimizer: None }, - Some(opt) => { - let minimizer = opt - .minimizer - .as_ref() - .map(minimizer::MinimizerConfig::from_options); - ShellConfig { - session_env: opt.session_env, - snapshot_path: opt.snapshot_path, - minimizer, - } - }, - }; - Self { - session: Arc::new(TokioMutex::new(None)), - abort_state: ShellAbortState::default(), - config, - } + Self { inner: Arc::new(CoreShell::new(options.map(Into::into))) } } /// Run a shell command using the provided options. @@ -206,27 +196,32 @@ impl Shell { /// the exit code when the command completes, or flags when cancelled or /// timed out. #[napi] - pub fn run<'e>( + pub fn run<'env>( &self, - env: &'e Env, - options: ShellRunOptions<'e>, + env: &'env Env, + options: ShellRunOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] on_chunk: Option<ThreadsafeFunction<String>>, - ) -> Result<PromiseRaw<'e, ShellRunResult>> { - let ct = task::CancelToken::new(options.timeout_ms, options.signal); - let session = self.session.clone(); - let abort_state = self.abort_state.clone(); - let config = self.config.clone(); - - let run_config = ShellRunConfig { - command: options.command, - cwd: options.cwd, - env: options.env, - minimizer: config.minimizer.clone(), + ) -> Result<PromiseRaw<'env, ShellRunResult>> { + let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal); + let inner = Arc::clone(&self.inner); + let run_options = CoreShellRunOptions { + command: options.command, + cwd: options.cwd, + env: options.env, + timeout_ms: options.timeout_ms, }; - task::future(env, "shell.run", async move { - run_shell_session(session, abort_state, config, run_config, on_chunk, ct).await + let (chunk_tx, drain_handle) = bridge_chunks(on_chunk); + let result = inner + .run(run_options, chunk_tx, cancel_token.into_core()) + .await + .map(Into::into) + .map_err(|err| Error::from_reason(err.to_string())); + if let Some(handle) = drain_handle { + let _ = handle.await; + } + result }) } @@ -235,114 +230,11 @@ impl Shell { /// Returns `Ok(())` even when no commands are running. #[napi] pub async fn abort(&self) -> Result<()> { - self.abort_state.abort().await; + self.inner.abort().await; Ok(()) } } -/// Run a shell command within a persistent session. -async fn run_shell_session( - session: Arc<TokioMutex<Option<ShellSessionCore>>>, - abort_state: ShellAbortState, - config: ShellConfig, - run_config: ShellRunConfig, - on_chunk: Option<ThreadsafeFunction<String>>, - mut ct: task::CancelToken, -) -> Result<ShellRunResult> { - let tokio_cancel = CancellationToken::new(); - - let mut run_task = tokio::spawn({ - let session = session.clone(); - let abort_state = abort_state.clone(); - let tokio_cancel = tokio_cancel.clone(); - let at = ct.emplace_abort_token(); - async move { - let mut session_guard = session.lock().await; - - let session = match &mut *session_guard { - Some(session) => session, - None => session_guard.insert(create_session(&config).await?), - }; - abort_state.set(at).await; - run_shell_command(session, &run_config, on_chunk, tokio_cancel).await - } - }); - - let res = tokio::select! { - res = &mut run_task => res, - reason = ct.wait() => { - tokio_cancel.cancel(); - let graceful = time::timeout(Duration::from_secs(2), &mut run_task).await; - if graceful.is_err() { - run_task.abort(); - let _ = run_task.await; - } - abort_state.clear().await; - // Use try_lock to avoid deadlocking if another task holds the session. - // If we can't acquire the lock, the session will be cleaned up when the - // holding task finishes. - if let Ok(mut guard) = session.try_lock() { - *guard = None; - } - return Ok(ShellRunResult { - exit_code: None, - cancelled: matches!(reason, task::AbortReason::Signal), - timed_out: matches!(reason, task::AbortReason::Timeout), - minimized: None, - }); - } - }; - let res = - res.unwrap_or_else(|e| Err(Error::from_reason(format!("Shell execution task failed: {e}")))); - abort_state.clear().await; - - let keepalive = res.as_ref().is_ok_and(|pair| session_keepalive(&pair.0)); - if !keepalive { - *session.lock().await = None; - } - let (exec, minimized) = res?; - Ok(ShellRunResult { - exit_code: Some(exit_code(&exec)), - cancelled: false, - timed_out: false, - minimized, - }) -} - -/// Options for executing a shell command via brush-core. -#[napi(object)] -pub struct ShellExecuteOptions<'env> { - /// Command string to execute in the shell. - pub command: String, - /// Working directory for the command. - pub cwd: Option<String>, - /// Environment variables to apply for this command only. - pub env: Option<HashMap<String, String>>, - /// Environment variables to apply once per session. - pub session_env: Option<HashMap<String, String>>, - /// Timeout in milliseconds before cancelling the command. - pub timeout_ms: Option<u32>, - /// Optional snapshot file to source on session creation. - pub snapshot_path: Option<String>, - /// Optional per-command output minimizer configuration. - pub minimizer: Option<minimizer::MinimizerOptions>, - /// Abort signal for cancelling the operation. - pub signal: Option<Unknown<'env>>, -} - -/// Result of executing a shell command via brush-core. -#[napi(object)] -pub struct ShellExecuteResult { - /// Exit code when the command completes normally. - pub exit_code: Option<i32>, - /// Whether the command was cancelled via abort. - pub cancelled: bool, - /// Whether the command timed out before completion. - pub timed_out: bool, - /// See [`ShellRunResult::minimized`]. - pub minimized: Option<MinimizerResult>, -} - /// Execute a brush shell command. /// /// Creates a fresh session for each call. The `on_chunk` callback receives @@ -354,1219 +246,163 @@ pub fn execute_shell<'env>( options: ShellExecuteOptions<'env>, #[napi(ts_arg_type = "((error: Error | null, chunk: string) => void) | undefined | null")] on_chunk: Option<ThreadsafeFunction<String>>, -) -> Result<PromiseRaw<'env, ShellExecuteResult>> { - let minimizer = options - .minimizer - .as_ref() - .map(minimizer::MinimizerConfig::from_options); - let config = ShellConfig { +) -> Result<PromiseRaw<'env, ShellRunResult>> { + let cancel_token = task::CancelToken::new(options.timeout_ms, options.signal); + let exec_options = CoreShellExecuteOptions { + command: options.command, + cwd: options.cwd, + env: options.env, session_env: options.session_env, + timeout_ms: options.timeout_ms, snapshot_path: options.snapshot_path, - minimizer: minimizer.clone(), + minimizer: options.minimizer.map(Into::into), }; - let run_config = - ShellRunConfig { command: options.command, cwd: options.cwd, env: options.env, minimizer }; - - let ct = task::CancelToken::new(options.timeout_ms, options.signal); task::future(env, "shell.execute", async move { - run_shell_oneshot(config, run_config, on_chunk, ct).await + let (chunk_tx, drain_handle) = bridge_chunks(on_chunk); + let result = core_execute_shell(exec_options, chunk_tx, cancel_token.into_core()) + .await + .map(Into::into) + .map_err(|err| Error::from_reason(err.to_string())); + if let Some(handle) = drain_handle { + let _ = handle.await; + } + result }) } -/// Run a shell command in a fresh session (one-shot execution). -async fn run_shell_oneshot( - config: ShellConfig, - run_config: ShellRunConfig, +fn bridge_chunks( on_chunk: Option<ThreadsafeFunction<String>>, - ct: task::CancelToken, -) -> Result<ShellExecuteResult> { - let tokio_cancel = CancellationToken::new(); - - let mut task = tokio::spawn({ - let tokio_cancel = tokio_cancel.clone(); - async move { - let mut session = create_session(&config).await?; - run_shell_command(&mut session, &run_config, on_chunk, tokio_cancel).await +) -> (Option<mpsc::UnboundedSender<String>>, Option<napi::tokio::task::JoinHandle<()>>) { + let Some(on_chunk) = on_chunk else { + return (None, None); + }; + let (tx, mut rx) = mpsc::unbounded_channel::<String>(); + let handle = napi::tokio::spawn(async move { + while let Some(chunk) = rx.recv().await { + on_chunk.call(Ok(chunk), ThreadsafeFunctionCallMode::NonBlocking); } }); - - let run_result = tokio::select! { - result = &mut task => result, - reason = ct.wait() => { - tokio_cancel.cancel(); - let graceful = time::timeout(Duration::from_secs(2), &mut task).await; - if graceful.is_err() { - task.abort(); - let _ = task.await; - } - return Ok(ShellExecuteResult { - exit_code: None, - cancelled: matches!(reason, task::AbortReason::Signal), - timed_out: matches!(reason, task::AbortReason::Timeout), - minimized: None, - }) - }, - }; - - let res = run_result - .unwrap_or_else(|e| Err(Error::from_reason(format!("Shell execution task failed: {e}")))); - - let (exec, minimized) = res?; - Ok(ShellExecuteResult { - exit_code: Some(exit_code(&exec)), - cancelled: false, - timed_out: false, - minimized, - }) -} - -fn null_file() -> Result<OpenFile> { - openfiles::null().map_err(|err| Error::from_reason(format!("Failed to create null file: {err}"))) -} - -const fn exit_code(result: &ExecutionResult) -> i32 { - match result.exit_code { - ExecutionExitCode::Success => 0, - ExecutionExitCode::GeneralError => 1, - ExecutionExitCode::InvalidUsage => 2, - ExecutionExitCode::Unimplemented => 99, - ExecutionExitCode::CannotExecute => 126, - ExecutionExitCode::NotFound => 127, - ExecutionExitCode::Interrupted => 130, - ExecutionExitCode::BrokenPipe => 141, - ExecutionExitCode::Custom(code) => code as i32, - } -} - -#[cfg(windows)] -const fn normalize_env_key(key: &str) -> &str { - if key.eq_ignore_ascii_case("PATH") { - "PATH" - } else { - key - } -} - -#[cfg(not(windows))] -const fn normalize_env_key(key: &str) -> &str { - key -} - -#[cfg(windows)] -fn merge_path_values(existing: &str, incoming: &str) -> String { - let mut merged = Vec::new(); - let mut seen = HashSet::new(); - push_unique_paths(&mut merged, &mut seen, existing); - push_unique_paths(&mut merged, &mut seen, incoming); - - std::env::join_paths(merged.iter()) - .map_or_else(|_| merged.join(";"), |paths| paths.to_string_lossy().into_owned()) -} - -#[cfg(windows)] -fn push_unique_paths(merged: &mut Vec<String>, seen: &mut HashSet<String>, value: &str) { - for segment in std::env::split_paths(value) { - let segment_str = segment.to_string_lossy().into_owned(); - let normalized = normalize_path_segment(&segment_str); - if normalized.is_empty() { - continue; - } - if seen.insert(normalized) { - merged.push(segment_str); - } - } -} - -#[cfg(windows)] -fn normalize_path_segment(segment: &str) -> String { - let trimmed = segment.trim().trim_matches('"'); - if trimmed.is_empty() { - return String::new(); - } - - let mut normalized = std::path::PathBuf::new(); - for component in std::path::Path::new(trimmed).components() { - normalized.push(component.as_os_str()); - } - - normalized.to_string_lossy().to_ascii_lowercase() -} - -#[cfg(not(windows))] -fn merge_path_values(_existing: &str, incoming: &str) -> String { - incoming.to_string() -} - -async fn create_session(config: &ShellConfig) -> Result<ShellSessionCore> { - let mut shell = BrushShell::builder() - .do_not_inherit_env(true) - .profile(ProfileLoadBehavior::Skip) - .rc(RcLoadBehavior::Skip) - .builtins(default_builtins(BuiltinSet::BashMode)) - .build() - .await - .map_err(|err| Error::from_reason(format!("Failed to initialize shell: {err}")))?; - - if let Some(exec_builtin) = shell.builtin_mut("exec") { - exec_builtin.disabled = true; - } - if let Some(suspend_builtin) = shell.builtin_mut("suspend") { - suspend_builtin.disabled = true; - } - shell.register_builtin("sleep", builtins::builtin::<SleepCommand, _>()); - shell.register_builtin("timeout", builtins::builtin::<TimeoutCommand, _>()); - - let mut merged_path: Option<String> = None; - for (key, value) in std::env::vars() { - let normalized_key = normalize_env_key(&key); - if should_skip_env_var(normalized_key) { - continue; - } - if normalized_key == "PATH" { - merged_path = Some(match merged_path { - Some(existing) => merge_path_values(&existing, &value), - None => value, - }); - continue; - } - let mut var = ShellVariable::new(ShellValue::String(value)); - var.export(); - shell - .env_mut() - .set_global(normalized_key, var) - .map_err(|err| Error::from_reason(format!("Failed to set env: {err}")))?; - } - - #[cfg(windows)] - if merged_path.is_none() - && let Some(value) = std::env::var_os("Path").or_else(|| std::env::var_os("PATH")) - { - merged_path = Some(value.to_string_lossy().into_owned()); - } - - if let Some(path_value) = merged_path { - let mut var = ShellVariable::new(ShellValue::String(path_value)); - var.export(); - shell - .env_mut() - .set_global("PATH", var) - .map_err(|err| Error::from_reason(format!("Failed to set env: {err}")))?; - } - - if let Some(env) = config.session_env.as_ref() { - for (key, value) in env { - let normalized_key = normalize_env_key(key); - if should_skip_env_var(normalized_key) { - continue; - } - let mut var = ShellVariable::new(ShellValue::String(value.clone())); - var.export(); - shell - .env_mut() - .set_global(normalized_key, var) - .map_err(|err| Error::from_reason(format!("Failed to set env: {err}")))?; - } - } - - #[cfg(windows)] - configure_windows_path(&mut shell)?; - - if let Some(snapshot_path) = config.snapshot_path.as_ref() { - source_snapshot(&mut shell, snapshot_path).await?; - } - - Ok(ShellSessionCore { shell }) -} - -async fn source_snapshot(shell: &mut BrushShell, snapshot_path: &str) -> Result<()> { - let mut params = shell.default_exec_params(); - let source_info = SourceInfo::from("pi-natives:snapshot"); - params.set_fd(OpenFiles::STDIN_FD, null_file()?); - params.set_fd(OpenFiles::STDOUT_FD, null_file()?); - params.set_fd(OpenFiles::STDERR_FD, null_file()?); - - let escaped = snapshot_path.replace('\'', "'\\''"); - let command = format!("source '{escaped}'"); - shell - .run_string(command, &source_info, ¶ms) - .await - .map_err(|err| Error::from_reason(format!("Failed to source snapshot: {err}")))?; - Ok(()) -} - -async fn run_shell_command( - session: &mut ShellSessionCore, - options: &ShellRunConfig, - on_chunk: Option<ThreadsafeFunction<String>>, - cancel_token: CancellationToken, -) -> Result<(ExecutionResult, Option<MinimizerResult>)> { - if let Some(cwd) = options.cwd.as_deref() { - session - .shell - .set_working_dir(cwd) - .map_err(|err| Error::from_reason(format!("Failed to set cwd: {err}")))?; - } - - let (reader_file, writer_file) = pipe_to_files("output")?; - - let stdout_file = OpenFile::from( - writer_file - .try_clone() - .map_err(|err| Error::from_reason(format!("Failed to clone pipe: {err}")))?, - ); - let stderr_file = OpenFile::from(writer_file); - - let mut params = session.shell.default_exec_params(); - params.set_fd(OpenFiles::STDIN_FD, null_file()?); - params.set_fd(OpenFiles::STDOUT_FD, stdout_file); - params.set_fd(OpenFiles::STDERR_FD, stderr_file); - params.process_group_policy = ProcessGroupPolicy::NewProcessGroup; - params.set_cancel_token(cancel_token.clone()); - - let mut env_scope_pushed = false; - if let Some(env) = options.env.as_ref() { - session - .shell - .env_mut() - .push_scope(EnvironmentScope::Command); - env_scope_pushed = true; - for (key, value) in env { - let normalized_key = normalize_env_key(key); - if should_skip_env_var(normalized_key) { - continue; - } - let mut var = ShellVariable::new(ShellValue::String(value.clone())); - var.export(); - if let Err(err) = - session - .shell - .env_mut() - .add(normalized_key, var, EnvironmentScope::Command) - { - let _ = session.shell.env_mut().pop_scope(EnvironmentScope::Command); - return Err(Error::from_reason(format!("Failed to set env: {err}"))); - } - } - } - - let minimizer_mode = if let Some(config) = options.minimizer.as_ref() { - minimizer::engine::mode_for(&options.command, config) - } else { - minimizer::engine::MinimizerMode::None - }; - let should_minimize = !matches!(minimizer_mode, minimizer::engine::MinimizerMode::None); - let max_capture_bytes = if let Some(config) = options.minimizer.as_ref() { - config.max_capture_bytes as usize - } else { - 0 - }; - - let reader_cancel = CancellationToken::new(); - let (activity_tx, mut activity_rx) = mpsc::channel::<()>(1); - // Stream every raw chunk to the caller live, regardless of whether - // minimization is enabled. When minimization actually transforms the - // output, we propagate the replacement text via `MinimizerResult.text` - // so the caller can swap their accumulated buffer for the minimized - // version without losing intermediate progress updates. - let reader_callback = on_chunk; - let mut reader_handle = tokio::spawn({ - let reader_cancel = reader_cancel.clone(); - async move { - if should_minimize { - let output = read_output_buffered( - reader_file, - reader_callback, - reader_cancel, - activity_tx, - max_capture_bytes, - ) - .await; - Result::<OutputRead>::Ok(OutputRead::Buffered(output)) - } else { - Box::pin(read_output(reader_file, reader_callback, reader_cancel, activity_tx)).await; - Result::<OutputRead>::Ok(OutputRead::Streaming) - } - } - }); - let cancel_bridge = tokio::spawn({ - let cancel_token = cancel_token.clone(); - let reader_cancel = reader_cancel.clone(); - async move { - cancel_token.cancelled().await; - reader_cancel.cancel(); - } - }); - let source_info = SourceInfo::from("pi-natives:command"); - let result = session - .shell - .run_string(options.command.clone(), &source_info, ¶ms) - .await; - - if cancel_token.is_cancelled() { - terminate_background_jobs(&session.shell); - } - - if env_scope_pushed { - session - .shell - .env_mut() - .pop_scope(EnvironmentScope::Command) - .map_err(|err| Error::from_reason(format!("Failed to pop env scope: {err}")))?; - } - - drop(params); - - // The foreground command can complete while background jobs keep the - // stdout/stderr pipe open. Don't hang forever waiting for EOF; drain output - // for a short period, then cancel. - const POST_EXIT_IDLE: Duration = Duration::from_millis(250); - const POST_EXIT_MAX: Duration = Duration::from_secs(2); - const READER_SHUTDOWN_TIMEOUT: Duration = Duration::from_millis(250); - - let mut reader_finished = false; - let mut reader_output = None; - let mut idle_timer = Box::pin(time::sleep(POST_EXIT_IDLE)); - let mut max_timer = Box::pin(time::sleep(POST_EXIT_MAX)); - - loop { - tokio::select! { - res = &mut reader_handle => { - if let Ok(Ok(output)) = res { - reader_output = Some(output); - } - reader_finished = true; - break; - } - msg = activity_rx.recv() => { - if msg.is_none() { - break; - } - idle_timer.as_mut().reset(time::Instant::now() + POST_EXIT_IDLE); - } - () = &mut idle_timer => break, - () = &mut max_timer => break, - } - } - - if !reader_finished { - reader_cancel.cancel(); - if let Ok(res) = time::timeout(READER_SHUTDOWN_TIMEOUT, &mut reader_handle).await { - if let Ok(output) = res - && let Ok(output) = output - { - reader_output = Some(output); - } - } else { - reader_handle.abort(); - let _ = reader_handle.await; - } - } - cancel_bridge.abort(); - let _ = cancel_bridge.await; - - let result = - result.map_err(|err| Error::from_reason(format!("Shell execution failed: {err}")))?; - let mut minimized_out: Option<MinimizerResult> = None; - if let Some(OutputRead::Buffered(output)) = reader_output - && let Some(config) = options.minimizer.as_ref() - && !output.exceeded - { - let minimized = match minimizer_mode { - minimizer::engine::MinimizerMode::WholeCommand => { - minimizer::apply(&options.command, &output.text, exit_code(&result), config) - }, - minimizer::engine::MinimizerMode::None => { - minimizer::MinimizerOutput::passthrough(&output.text) - }, - }; - if minimized.changed - && let Some(original) = minimized.original_text - { - let output_bytes = u32::try_from(minimized.text.len()).unwrap_or(u32::MAX); - minimized_out = Some(MinimizerResult { - filter: minimized.filter.to_string(), - text: minimized.text, - original_text: original, - input_bytes: u32::try_from(minimized.input_bytes).unwrap_or(u32::MAX), - output_bytes, - }); - } - } - Ok((result, minimized_out)) -} - -fn terminate_background_jobs(shell: &BrushShell) { - if shell.jobs().jobs.is_empty() { - return; - } - let mut targets = ps::TerminationTargets::new(); - for job in &shell.jobs().jobs { - if let Some(pgid) = job.process_group_id() { - targets.add_pgid(pgid); - } - if let Some(pid) = job.representative_pid() { - targets.add_pid(pid); - } - } - if targets.is_empty() { - return; - } - - targets.signal(ps::TERM_SIGNAL); - tokio::spawn(async move { - time::sleep(Duration::from_millis(500)).await; - targets.signal(ps::KILL_SIGNAL); - }); -} -fn should_skip_env_var(key: &str) -> bool { - if key.starts_with("BASH_FUNC_") && key.ends_with("%%") { - return true; - } - - matches!( - key, - "BASH_ENV" - | "ENV" - | "HISTFILE" - | "HISTTIMEFORMAT" - | "HISTCMD" - | "PS0" - | "PS1" - | "PS2" - | "PS4" - | "BRUSH_PS_ALT" - | "READLINE_LINE" - | "READLINE_POINT" - | "BRUSH_VERSION" - | "BASH" - | "BASHOPTS" - | "BASH_ALIASES" - | "BASH_ARGV0" - | "BASH_CMDS" - | "BASH_SOURCE" - | "BASH_SUBSHELL" - | "BASH_VERSINFO" - | "BASH_VERSION" - | "SHELLOPTS" - | "SHLVL" - | "SHELL" - | "COMP_WORDBREAKS" - | "DIRSTACK" - | "EPOCHREALTIME" - | "EPOCHSECONDS" - | "FUNCNAME" - | "GROUPS" - | "IFS" - | "LINENO" - | "MACHTYPE" - | "OSTYPE" - | "OPTERR" - | "OPTIND" - | "PIPESTATUS" - | "PPID" - | "PWD" - | "OLDPWD" - | "RANDOM" - | "SRANDOM" - | "SECONDS" - | "UID" - | "EUID" - | "HOSTNAME" - | "HOSTTYPE" - ) -} - -const fn session_keepalive(result: &ExecutionResult) -> bool { - match result.next_control_flow { - ExecutionControlFlow::Normal => true, - ExecutionControlFlow::BreakLoop { .. } => false, - ExecutionControlFlow::ContinueLoop { .. } => false, - ExecutionControlFlow::ReturnFromFunctionOrScript => false, - ExecutionControlFlow::ExitShell => false, - } -} - -enum OutputRead { - Streaming, - Buffered(BufferedOutput), -} - -struct BufferedOutput { - text: String, - exceeded: bool, -} - -async fn read_output( - reader: fs::File, - on_chunk: Option<ThreadsafeFunction<String>>, - cancel_token: CancellationToken, - activity: mpsc::Sender<()>, -) { - const REPLACEMENT: &str = "\u{FFFD}"; - const BUF: usize = 65536; - let mut buf = vec![0u8; BUF + 4]; // +4 for max UTF-8 char - let mut it = 0; - - #[cfg(unix)] - let Ok(reader) = register_nonblocking_pipe(reader) else { - return; - }; - #[cfg(not(unix))] - let reader = tokio::fs::File::from_std(reader); - #[cfg(not(unix))] - tokio::pin!(reader); - - loop { - #[cfg(unix)] - let n = { - let Ok(mut readiness) = (tokio::select! { - ready = reader.readable() => ready, - () = cancel_token.cancelled() => break, - }) else { - break; - }; - match readiness.try_io(|inner| read_nonblocking(inner.get_ref(), &mut buf[it..BUF])) { - Ok(Ok(0)) => break, - Ok(Ok(n)) => n, - Ok(Err(e)) if e.kind() == io::ErrorKind::Interrupted => continue, - Ok(Err(_)) => break, - Err(_would_block) => continue, - } - }; - #[cfg(not(unix))] - let n = { - let read_future = reader.read(&mut buf[it..BUF]); - tokio::pin!(read_future); - match tokio::select! { - res = &mut read_future => res, - () = cancel_token.cancelled() => break, - } { - Ok(0) => break, // EOF - Ok(n) => n, - Err(e) if e.kind() == io::ErrorKind::Interrupted => continue, - Err(_) => break, - } - }; - if n > 0 { - let _ = activity.try_send(()); - } - it += n; - - // Consume as much of `pending` as is decodable *right now*. - while it > 0 { - let pending = &buf[..it]; - match str::from_utf8(pending) { - Ok(text) => { - emit_chunk(text, on_chunk.as_ref()); - it = 0; - break; - }, - Err(err) => { - let p = err.valid_up_to(); - if p > 0 { - // SAFETY: [..p] is guaranteed valid UTF-8 by valid_up_to(). - let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; - emit_chunk(text, on_chunk.as_ref()); - // copy p..it to the beginning of the buffer - buf.copy_within(p..it, 0); - it -= p; - } - - match err.error_len() { - Some(p) => { - // Invalid byte sequence: emit replacement and drop those bytes. - emit_chunk(REPLACEMENT, on_chunk.as_ref()); - // copy p..it to the beginning of the buffer - buf.copy_within(p..it, 0); - it -= p; - // continue loop in case more bytes remain after the - // invalid sequence - }, - None => { - // Incomplete UTF-8 sequence at end: keep bytes for next read. - break; - }, - } - }, - } - } - } - - // Flush whatever is left at EOF (including an incomplete final sequence). - for chunk in buf[..it].utf8_chunks() { - let valid = chunk.valid(); - if !valid.is_empty() { - emit_chunk(valid, on_chunk.as_ref()); - } - if !chunk.invalid().is_empty() { - emit_chunk(REPLACEMENT, on_chunk.as_ref()); - } - } -} - -async fn read_output_buffered( - reader: fs::File, - on_chunk: Option<ThreadsafeFunction<String>>, - cancel_token: CancellationToken, - activity: mpsc::Sender<()>, - max_capture_bytes: usize, -) -> BufferedOutput { - const REPLACEMENT: &str = "\u{FFFD}"; - const BUF: usize = 65536; - let mut buf = vec![0u8; BUF]; - let mut captured = Vec::new(); - let mut exceeded = false; - // Pending bytes from a prior read that ended mid-UTF-8 sequence. We hold - // them back so we emit only valid UTF-8 to the streaming callback while - // still capturing every byte into `captured` for post-processing. - let mut pending = Vec::<u8>::new(); - - #[cfg(unix)] - let Ok(reader) = register_nonblocking_pipe(reader) else { - return BufferedOutput { text: String::new(), exceeded: true }; - }; - #[cfg(not(unix))] - let reader = tokio::fs::File::from_std(reader); - #[cfg(not(unix))] - tokio::pin!(reader); - - loop { - #[cfg(unix)] - let n = { - let Ok(mut readiness) = (tokio::select! { - ready = reader.readable() => ready, - () = cancel_token.cancelled() => break, - }) else { - break; - }; - match readiness.try_io(|inner| read_nonblocking(inner.get_ref(), &mut buf)) { - Ok(Ok(0)) => break, - Ok(Ok(n)) => n, - Ok(Err(e)) if e.kind() == io::ErrorKind::Interrupted => continue, - Ok(Err(_)) => break, - Err(_would_block) => continue, - } - }; - #[cfg(not(unix))] - let n = { - let read_future = reader.read(&mut buf); - tokio::pin!(read_future); - match tokio::select! { - res = &mut read_future => res, - () = cancel_token.cancelled() => break, - } { - Ok(0) => break, - Ok(n) => n, - Err(e) if e.kind() == io::ErrorKind::Interrupted => continue, - Err(_) => break, - } - }; - if n > 0 { - let _ = activity.try_send(()); - } - // Once `exceeded`, the post-process minimizer is bypassed (see the - // `!output.exceeded` gate at the call site), so further appends just - // grow `captured` without serving any purpose. Stop accumulating to - // bound peak memory on commands that produce very large output. - if !exceeded { - if captured.len().saturating_add(n) > max_capture_bytes { - exceeded = true; - } else { - captured.extend_from_slice(&buf[..n]); - } - } - - // Stream whatever is validly decodable *right now* to the callback, - // carrying incomplete trailing UTF-8 bytes over to the next iteration. - if let Some(cb) = on_chunk.as_ref() { - pending.extend_from_slice(&buf[..n]); - while !pending.is_empty() { - match str::from_utf8(&pending) { - Ok(text) => { - emit_chunk(text, Some(cb)); - pending.clear(); - break; - }, - Err(err) => { - let p = err.valid_up_to(); - if p > 0 { - // SAFETY: [..p] is valid UTF-8 per valid_up_to(). - let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; - emit_chunk(text, Some(cb)); - pending.drain(..p); - } - match err.error_len() { - Some(skip) => { - emit_chunk(REPLACEMENT, Some(cb)); - pending.drain(..skip); - }, - None => break, - } - }, - } - } - } - } - - // Flush any trailing bytes the streaming decoder held back at EOF. - if let Some(cb) = on_chunk.as_ref() { - for chunk in pending.utf8_chunks() { - let valid = chunk.valid(); - if !valid.is_empty() { - emit_chunk(valid, Some(cb)); - } - if !chunk.invalid().is_empty() { - emit_chunk(REPLACEMENT, Some(cb)); - } - } - } - - BufferedOutput { text: String::from_utf8_lossy(&captured).into_owned(), exceeded } -} - -#[cfg(unix)] -fn register_nonblocking_pipe(reader: fs::File) -> io::Result<tokio::io::unix::AsyncFd<fs::File>> { - set_nonblocking(&reader)?; - tokio::io::unix::AsyncFd::new(reader) -} - -#[cfg(unix)] -fn set_nonblocking<T: std::os::fd::AsRawFd>(file: &T) -> io::Result<()> { - let fd = file.as_raw_fd(); - // SAFETY: `fd` is owned by `file` and remains valid for the duration of - // these `fcntl` calls. - let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) }; - if flags < 0 { - return Err(io::Error::last_os_error()); - } - if flags & libc::O_NONBLOCK != 0 { - return Ok(()); - } - - // SAFETY: `fd` remains valid here and we are only toggling `O_NONBLOCK`. - let result = unsafe { libc::fcntl(fd, libc::F_SETFL, flags | libc::O_NONBLOCK) }; - if result < 0 { - Err(io::Error::last_os_error()) - } else { - Ok(()) - } -} - -#[cfg(unix)] -fn read_nonblocking<T: std::os::fd::AsRawFd>(file: &T, buf: &mut [u8]) -> io::Result<usize> { - // SAFETY: `buf` is writable for `buf.len()` bytes, and the raw fd obtained - // from `file` stays valid for the duration of the syscall. - let read = unsafe { libc::read(file.as_raw_fd(), buf.as_mut_ptr().cast(), buf.len()) }; - if read < 0 { - Err(io::Error::last_os_error()) - } else { - Ok(read as usize) - } -} - -fn emit_chunk(text: &str, callback: Option<&ThreadsafeFunction<String>>) { - if let Some(callback) = callback { - callback.call(Ok(text.to_string()), ThreadsafeFunctionCallMode::NonBlocking); - } -} - -fn pipe_to_files(label: &str) -> Result<(fs::File, fs::File)> { - let (r, w) = os_pipe::pipe() - .map_err(|err| Error::from_reason(format!("Failed to create {label} pipe: {err}")))?; - - #[cfg(unix)] - let (r, w): (fs::File, fs::File) = { - use std::os::unix::io::{FromRawFd, IntoRawFd}; - let r = r.into_raw_fd(); - let w = w.into_raw_fd(); - // SAFETY: We just obtained these fds from os_pipe and own them exclusively. - unsafe { (FromRawFd::from_raw_fd(r), FromRawFd::from_raw_fd(w)) } - }; - - #[cfg(windows)] - let (r, w): (fs::File, fs::File) = { - use std::os::windows::io::{FromRawHandle, IntoRawHandle}; - let r = r.into_raw_handle(); - let w = w.into_raw_handle(); - // SAFETY: We just obtained these handles from os_pipe and own them exclusively. - unsafe { (FromRawHandle::from_raw_handle(r), FromRawHandle::from_raw_handle(w)) } - }; - - Ok((r, w)) -} - -#[derive(Parser)] -#[command(disable_help_flag = true)] -struct SleepCommand { - #[arg(required = true)] - durations: Vec<String>, -} - -impl builtins::Command for SleepCommand { - type Error = brush_core::Error; - - fn execute<SE: brush_core::ShellExtensions>( - &self, - context: ExecutionContext<'_, SE>, - ) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send { - let durations = self.durations.clone(); - async move { - if context.is_cancelled() { - return Ok(ExecutionExitCode::Interrupted.into()); - } - let mut total = Duration::from_millis(0); - for duration in &durations { - let Some(parsed) = parse_duration(duration) else { - let _ = writeln!(context.stderr(), "sleep: invalid time interval '{duration}'"); - return Ok(ExecutionResult::new(1)); - }; - total += parsed; - } - let sleep = time::sleep(total); - tokio::pin!(sleep); - if let Some(cancel_token) = context.cancel_token() { - tokio::select! { - () = &mut sleep => Ok(ExecutionResult::success()), - () = cancel_token.cancelled() => Ok(ExecutionExitCode::Interrupted.into()), - } - } else { - sleep.await; - Ok(ExecutionResult::success()) - } - } - } -} - -#[derive(Parser)] -#[command(disable_help_flag = true)] -struct TimeoutCommand { - #[arg(required = true)] - duration: String, - #[arg(required = true, num_args = 1.., trailing_var_arg = true)] - command: Vec<String>, -} - -impl builtins::Command for TimeoutCommand { - type Error = brush_core::Error; - - fn execute<SE: brush_core::ShellExtensions>( - &self, - context: ExecutionContext<'_, SE>, - ) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send { - let duration = self.duration.clone(); - let command = self.command.clone(); - async move { - if context.is_cancelled() { - return Ok(ExecutionExitCode::Interrupted.into()); - } - let Some(timeout) = parse_duration(&duration) else { - let _ = writeln!(context.stderr(), "timeout: invalid time interval '{duration}'"); - return Ok(ExecutionResult::new(125)); - }; - if command.is_empty() { - let _ = writeln!(context.stderr(), "timeout: missing command"); - return Ok(ExecutionResult::new(125)); - } - - let child_cancel = CancellationToken::new(); - let mut params = context.params.clone(); - params.process_group_policy = ProcessGroupPolicy::NewProcessGroup; - params.set_cancel_token(child_cancel.clone()); - - let mut command_line = String::new(); - for (idx, arg) in command.iter().enumerate() { - if idx > 0 { - command_line.push(' '); - } - command_line.push_str("e_arg(arg)); - } - - let cancel_token = context.cancel_token(); - let source_info = SourceInfo::from("pi-natives:timeout"); - let run_future = context - .shell - .run_string(command_line, &source_info, ¶ms); - tokio::pin!(run_future); - - if let Some(cancel_token) = cancel_token { - tokio::select! { - result = &mut run_future => result, - () = time::sleep(timeout) => { - child_cancel.cancel(); - // Wait briefly for the child to exit after cancellation. - let _ = time::timeout(Duration::from_secs(2), &mut run_future).await; - Ok(ExecutionResult::new(124)) - }, - () = cancel_token.cancelled() => { - child_cancel.cancel(); - Ok(ExecutionExitCode::Interrupted.into()) - }, - } - } else { - tokio::select! { - result = &mut run_future => result, - () = time::sleep(timeout) => { - child_cancel.cancel(); - // Wait briefly for the child to exit after cancellation. - let _ = time::timeout(Duration::from_secs(2), &mut run_future).await; - Ok(ExecutionResult::new(124)) - }, - } - } - } - } -} -fn parse_duration(input: &str) -> Option<Duration> { - let trimmed = input.trim(); - if trimmed.is_empty() { - return None; - } - let (number, multiplier) = match trimmed.chars().last()? { - 's' => (&trimmed[..trimmed.len() - 1], 1.0), - 'm' => (&trimmed[..trimmed.len() - 1], 60.0), - 'h' => (&trimmed[..trimmed.len() - 1], 3600.0), - 'd' => (&trimmed[..trimmed.len() - 1], 86400.0), - ch if ch.is_ascii_alphabetic() => return None, - _ => (trimmed, 1.0), - }; - let value = number.parse::<f64>().ok()?; - if value.is_sign_negative() { - return None; - } - let millis = value * multiplier * 1000.0; - if !millis.is_finite() || millis < 0.0 { - return None; - } - Some(Duration::from_millis(millis.round() as u64)) -} - -fn quote_arg(arg: &str) -> String { - if arg.is_empty() { - return "''".to_string(); - } - let safe = arg - .chars() - .all(|ch| ch.is_ascii_alphanumeric() || matches!(ch, '-' | '_' | '.' | '/' | ':' | '+')); - if safe { - return arg.to_string(); - } - let escaped = arg.replace('\'', "'\"'\"'"); - format!("'{escaped}'") + (Some(tx), Some(handle)) } #[cfg(test)] mod tests { - use super::*; + use std::time::Duration; - /// Truth-table coverage for `brush_core::commands::child_session_action`. - /// - /// Lives in `pi-natives` because the brush-core crate is excluded from the - /// workspace (vendored upstream) and cannot be tested standalone — its tokio - /// dependency only resolves the `net` feature via feature-unification with - /// other workspace members. - mod child_session_action { - use brush_core::commands::{ChildSessionAction, child_session_action}; + use pi_shell::{ + ShellRunOptions as CoreShellRunOptions, + cancel::{AbortReason, CancelToken}, + }; + use tokio::{sync::mpsc, time}; + + use super::CoreShell; + + mod child_session_action_tests { + use pi_shell::{ChildSessionAction, child_session_action}; - /// Interactive brush, leading its own pgroup, terminal stdin: foreground. #[test] fn interactive_with_terminal_stdin_takes_foreground() { - assert_eq!(child_session_action(true, true, false), ChildSessionAction::TakeForeground,); - // Terminal foregrounding wins even when this is the first stage of a - // pipeline; no detach is attempted. - assert_eq!(child_session_action(true, true, true), ChildSessionAction::TakeForeground,); + assert_eq!(child_session_action(true, true, false), ChildSessionAction::TakeForeground); + assert_eq!(child_session_action(true, true, true), ChildSessionAction::TakeForeground); } - /// Brush leading a new pgroup with non-terminal stdin detaches only when - /// it is not part of a multi-command pipeline. Pipeline leaders must stay - /// in the parent session so later stages can join their process group. #[test] fn non_terminal_stdin_leading_new_pgroup_detaches_unless_pipeline() { - assert_eq!(child_session_action(true, false, false), ChildSessionAction::DetachSession,); - assert_eq!(child_session_action(true, false, true), ChildSessionAction::None,); + assert_eq!(child_session_action(true, false, false), ChildSessionAction::DetachSession); + assert_eq!(child_session_action(true, false, true), ChildSessionAction::None); } - /// Non-interactive brush, terminal stdin, no pipeline: nothing to do. #[test] fn non_interactive_with_terminal_stdin_does_nothing() { - assert_eq!(child_session_action(false, true, false), ChildSessionAction::None,); + assert_eq!(child_session_action(false, true, false), ChildSessionAction::None); } - /// Non-interactive brush, terminal stdin, joining a pipeline pgroup: - /// nothing to do (parent already wired pgroup membership). #[test] fn non_interactive_terminal_stdin_in_pipeline_does_nothing() { - assert_eq!(child_session_action(false, true, true), ChildSessionAction::None,); + assert_eq!(child_session_action(false, true, true), ChildSessionAction::None); } - /// **Embedded host bug fix.** Non-interactive brush, non-terminal stdin, - /// no pipeline pgroup: detach so the child cannot SIGTTIN/SIGTTOU the - /// host. This is the case that regressed before this fix and is the - /// motivating bug for PR #895. #[test] fn embedded_host_with_non_terminal_stdin_detaches() { - assert_eq!(child_session_action(false, false, false), ChildSessionAction::DetachSession,); + assert_eq!(child_session_action(false, false, false), ChildSessionAction::DetachSession); } - /// **Pipeline carve-out.** Non-interactive brush, non-terminal stdin - /// (pipe), and a multi-command pipeline: MUST NOT detach. For the first - /// external stage, `setsid()` puts the process-group leader into a - /// different session, so later stages fail to join its group with - /// EPERM. For later stages, `setsid()` would either fail with EPERM or - /// move the child into a new session, breaking the pipeline's shared - /// process group and job-control signal propagation. #[test] fn pipeline_stage_does_not_detach() { - assert_eq!(child_session_action(false, false, true), ChildSessionAction::None,); + assert_eq!(child_session_action(false, false, true), ChildSessionAction::None); } } - /// End-to-end verification that brush, when embedded as a non-interactive - /// library (`interactive: false`, exactly what `create_session` produces), - /// spawns external commands in a **separate session** from the host. - /// - /// The truth-table tests in `child_session_action` cover the decision in - /// isolation. This test covers the wiring: it boots a real `BrushShell`, - /// runs a child that prints its PID then sleeps, and asks the kernel for - /// that PID's session via `getsid(2)` while the child is still alive. - /// Pre-fix (`new_pg=false` skipped `detach_session`), the child inherited - /// the host's session, so `getsid(child_pid) == getsid(0)`. Post-fix, - /// `setsid` ran and the child is its own session leader - /// (`getsid(child_pid) == child_pid`). #[cfg(unix)] #[tokio::test(flavor = "multi_thread")] async fn embedded_external_command_runs_in_its_own_session() { - use std::io::Read as _; - - // SAFETY: `getsid(0)` only queries the current process session; the return - // value is checked. + let shell = CoreShell::new(None); + let (tx, mut rx) = mpsc::unbounded_channel::<String>(); + let handle = tokio::spawn(async move { + shell + .run( + CoreShellRunOptions { + command: "/bin/sh -c 'printf \"%d\\n\" \"$$\"; sleep 0.5'".to_string(), + cwd: None, + env: None, + timeout_ms: None, + }, + Some(tx), + CancelToken::default(), + ) + .await + }); + let child_pid = time::timeout(Duration::from_secs(5), rx.recv()) + .await + .expect("timed out waiting for child pid") + .expect("missing child pid chunk") + .trim() + .parse::<i32>() + .expect("child pid parses"); + // SAFETY: `getsid(0)` only queries the current process session; the + // return value is checked below. let host_sid = unsafe { libc::getsid(0) }; assert!(host_sid > 0, "getsid(0) failed: {}", std::io::Error::last_os_error()); - - // Build the same kind of session pi-natives uses in production. - let config = ShellConfig { session_env: None, snapshot_path: None, minimizer: None }; - let mut session = create_session(&config).await.expect("create_session"); - - // Output pipe shared between the brush child and a concurrent reader. The - // reader runs on a blocking thread because `os_pipe` reads are blocking. - let (mut reader, writer) = pipe_to_files("e2e").expect("pipe"); - let stdout_file = OpenFile::from(writer.try_clone().expect("clone")); - let stderr_file = OpenFile::from(writer); - - let mut params = session.shell.default_exec_params(); - params.set_fd(OpenFiles::STDIN_FD, null_file().expect("null stdin")); - params.set_fd(OpenFiles::STDOUT_FD, stdout_file); - params.set_fd(OpenFiles::STDERR_FD, stderr_file); - - // (pid_tx, pid_rx) — reader task signals the test as soon as it has the PID. - let (pid_tx, pid_rx) = tokio::sync::oneshot::channel::<i32>(); - let reader_handle = tokio::task::spawn_blocking(move || { - let mut buf = Vec::new(); - // Read just enough to capture the PID line. The child sleeps after - // printing so the pipe will not back-pressure. - let mut chunk = [0u8; 64]; - let mut pid_tx = Some(pid_tx); - while let Ok(n) = reader.read(&mut chunk) - && n > 0 - { - buf.extend_from_slice(&chunk[..n]); - if pid_tx.is_some() - && let Some(line_end) = buf.iter().position(|&byte| byte == b'\n') - && let Ok(line) = std::str::from_utf8(&buf[..line_end]) - && let Ok(pid) = line.trim().parse::<i32>() - { - let _ = pid_tx - .take() - .expect("pid sender should be present") - .send(pid); - } - } - buf - }); - - // Run brush in the background so we can call `getsid(child_pid)` while - // the child is still alive. - let shell_handle = tokio::spawn(async move { - let source_info = SourceInfo::from("pi-natives:test"); - // `printf '%d\n' "$$"` then `sleep 0.5`. Long enough for our `getsid`. - let exec = session - .shell - .run_string("/bin/sh -c 'printf \"%d\\n\" \"$$\"; sleep 0.5'", &source_info, ¶ms) - .await - .expect("run_string"); - drop(params); - (session, exec) - }); - - let child_pid = time::timeout(Duration::from_secs(5), pid_rx) - .await - .expect("timed out waiting for child PID") - .expect("reader closed pid channel without sending"); - assert!(child_pid > 0, "got non-positive child pid: {child_pid}"); - - // Snapshot the child's session ID immediately, while the child is still - // in `sleep`. POSIX guarantees `getsid` against a live PID returns the - // session of that process. - // SAFETY: `child_pid` is a positive PID from the child; errors are reported via - // the checked return value. + // SAFETY: `child_pid` is a live positive PID reported by the child; the + // return value is checked below. let child_sid = unsafe { libc::getsid(child_pid) }; - assert!( - child_sid > 0, - "getsid({child_pid}) failed: {} (child may have already exited)", - std::io::Error::last_os_error(), - ); - - // Drain the brush task and the pipe reader. - let (_session, exec) = time::timeout(Duration::from_secs(5), shell_handle) + assert!(child_sid > 0, "getsid({child_pid}) failed: {}", std::io::Error::last_os_error()); + let result = handle .await - .expect("shell timed out") - .expect("shell task panicked"); - assert!( - matches!(exec.exit_code, ExecutionExitCode::Success), - "unexpected exit: {}", - exit_code(&exec), - ); - let _ = time::timeout(Duration::from_secs(2), reader_handle).await; - - assert_ne!( - child_sid, host_sid, - "child PID {child_pid} inherited host session {host_sid}; setsid() did not run — the \ - embedded-host bug is back", - ); - assert_eq!( - child_sid, child_pid, - "child PID {child_pid} should be its own session leader after setsid", - ); + .expect("shell task panicked") + .expect("shell run"); + assert_eq!(result.exit_code, Some(0)); + assert_ne!(child_sid, host_sid); + assert_eq!(child_sid, child_pid); } - #[tokio::test] - async fn abort_state_signals_cancel_token() { - let abort_state = ShellAbortState::default(); - let mut cancel_token = task::CancelToken::default(); - let abort_token = cancel_token.emplace_abort_token(); - - abort_state.set(abort_token).await; - abort_state.abort().await; - - let reason = time::timeout(Duration::from_millis(100), cancel_token.wait()) - .await - .expect("cancel token should be signalled"); - assert!(matches!(reason, task::AbortReason::Signal)); - } - - #[cfg(unix)] #[tokio::test] async fn read_output_stops_when_cancelled_before_pipe_eof() { - let (reader, _writer) = pipe_to_files("test").expect("test pipe should be created"); - let cancel = CancellationToken::new(); - let (activity_tx, _activity_rx) = mpsc::channel(1); - let handle = tokio::spawn(read_output(reader, None, cancel.clone(), activity_tx)); + let shell = CoreShell::new(None); + let mut cancel = CancelToken::default(); + let abort = cancel.emplace_abort_token(); + let handle = tokio::spawn(async move { + shell + .run( + CoreShellRunOptions { + command: "sh -c 'sleep 30 & wait'".to_string(), + cwd: None, + env: None, + timeout_ms: None, + }, + None, + cancel, + ) + .await + }); time::sleep(Duration::from_millis(10)).await; - cancel.cancel(); - - time::timeout(Duration::from_millis(100), handle) + abort.abort(AbortReason::Signal); + let result = time::timeout(Duration::from_secs(3), handle) .await - .expect("reader task should stop after cancellation") - .expect("reader task should not panic"); + .expect("shell run should stop after cancellation") + .expect("shell task should not panic") + .expect("shell run should return"); + assert!(result.cancelled); } } diff --git a/crates/pi-natives/src/summary.rs b/crates/pi-natives/src/summary.rs index db1e4199f..1298fbcb6 100644 --- a/crates/pi-natives/src/summary.rs +++ b/crates/pi-natives/src/summary.rs @@ -1,16 +1,7 @@ //! Structural source summaries powered by tree-sitter. -use std::{collections::BTreeSet, path::Path}; - -use ast_grep_core::{Language, tree_sitter::LanguageExt}; use napi::bindgen_prelude::*; use napi_derive::napi; -use tree_sitter::{Node, Parser}; - -use crate::language::SupportLang; - -const DEFAULT_MIN_BODY_LINES: u32 = 4; -const DEFAULT_MIN_COMMENT_LINES: u32 = 6; #[napi(object)] pub struct SummaryOptions { @@ -52,994 +43,38 @@ pub struct SummaryResult { pub segments: Vec<SummarySegment>, } -#[derive(Clone, Copy, Debug, Eq, PartialEq)] -struct LineSpan { - start: u32, - end: u32, +impl From<pi_ast::summary::SummarySegment> for SummarySegment { + fn from(value: pi_ast::summary::SummarySegment) -> Self { + Self { + kind: value.kind, + start_line: value.start_line, + end_line: value.end_line, + text: value.text, + } + } +} + +impl From<pi_ast::summary::SummaryResult> for SummaryResult { + fn from(value: pi_ast::summary::SummaryResult) -> Self { + Self { + language: value.language, + parsed: value.parsed, + elided: value.elided, + total_lines: value.total_lines, + segments: value.segments.into_iter().map(Into::into).collect(), + } + } } #[napi] pub fn summarize_code(options: SummaryOptions) -> Result<SummaryResult> { - let source = options.code; - let total_lines = count_lines(&source); - if source.is_empty() { - return Ok(unparsed_result(source, total_lines)); - } - - let Some(language) = resolve_language(options.lang.as_deref(), options.path.as_deref()) else { - return Ok(unparsed_result(source, total_lines)); - }; - - let mut parser = Parser::new(); - parser - .set_language(&language.get_ts_language()) - .map_err(|err| Error::from_reason(format!("Failed to load tree-sitter language: {err}")))?; - let Some(tree) = parser.parse(&source, None) else { - return Ok(unparsed_result(source, total_lines)); - }; - let root = tree.root_node(); - if root.has_error() { - return Ok(unparsed_result(source, total_lines)); - } - - let min_body_lines = options - .min_body_lines - .unwrap_or(DEFAULT_MIN_BODY_LINES) - .max(2); - let min_comment_lines = options - .min_comment_lines - .unwrap_or(DEFAULT_MIN_COMMENT_LINES) - .max(4); - let mut spans = Vec::new(); - collect_elisions(root, language, min_body_lines, min_comment_lines, &mut spans); - let spans = normalize_spans(spans, total_lines); - let segments = build_segments(&source, total_lines, &spans); - - Ok(SummaryResult { - language: Some(language.canonical_name().to_string()), - parsed: true, - elided: !spans.is_empty(), - total_lines, - segments, + pi_ast::summary::summarize_code(pi_ast::summary::SummaryOptions { + code: options.code, + lang: options.lang, + path: options.path, + min_body_lines: options.min_body_lines, + min_comment_lines: options.min_comment_lines, }) -} - -fn resolve_language(lang: Option<&str>, path: Option<&str>) -> Option<SupportLang> { - if let Some(lang) = lang.map(str::trim).filter(|lang| !lang.is_empty()) { - return SupportLang::from_alias(lang); - } - let path = path?.trim(); - if path.is_empty() { - return None; - } - SupportLang::from_path(Path::new(path)) -} - -fn unparsed_result(source: String, total_lines: u32) -> SummaryResult { - let segments = if source.is_empty() { - Vec::new() - } else { - vec![SummarySegment { - kind: "kept".to_string(), - start_line: 1, - end_line: total_lines, - text: Some(source), - }] - }; - SummaryResult { language: None, parsed: false, elided: false, total_lines, segments } -} - -fn count_lines(source: &str) -> u32 { - if source.is_empty() { - 0 - } else { - source.lines().count().max(1).min(u32::MAX as usize) as u32 - } -} - -fn collect_elisions( - node: Node<'_>, - language: SupportLang, - min_body_lines: u32, - min_comment_lines: u32, - spans: &mut Vec<LineSpan>, -) { - let total_lines = node_line_count(node); - if is_comment_kind(language, node.kind()) { - if total_lines >= min_comment_lines { - let start_line = node_start_line(node) + 2; - let end_line = node_end_line(node).saturating_sub(1); - if start_line <= end_line { - spans.push(LineSpan { start: start_line, end: end_line }); - } - } - return; - } - - if is_elidable_kind(language, node.kind()) && total_lines >= min_body_lines { - let start_line = node_start_line(node) + 1; - let end_line = node_end_line(node).saturating_sub(1); - if start_line <= end_line { - spans.push(LineSpan { start: start_line, end: end_line }); - return; - } - } - - // Detect consecutive runs of groupable siblings (e.g. import statements). - // When the run's total line span meets `min_body_lines`, elide the lines - // strictly between the first and last sibling's content, leaving the - // boundary statements visible. - let child_count = node.child_count(); - let mut run_first: Option<Node<'_>> = None; - let mut run_last: Option<Node<'_>> = None; - let mut run_count: u32 = 0; - for index in 0..child_count { - let Some(child) = node.child(index) else { - continue; - }; - if is_groupable_kind(language, child.kind()) { - if run_first.is_none() { - run_first = Some(child); - } - run_last = Some(child); - run_count += 1; - } else { - flush_groupable_run(run_first, run_last, run_count, min_body_lines, spans); - run_first = None; - run_last = None; - run_count = 0; - } - } - flush_groupable_run(run_first, run_last, run_count, min_body_lines, spans); - - for index in 0..child_count { - if let Some(child) = node.child(index) { - collect_elisions(child, language, min_body_lines, min_comment_lines, spans); - } - } -} - -fn flush_groupable_run( - first: Option<Node<'_>>, - last: Option<Node<'_>>, - count: u32, - min_body_lines: u32, - spans: &mut Vec<LineSpan>, -) { - if count < 2 { - return; - } - let (Some(first), Some(last)) = (first, last) else { - return; - }; - let first_start = node_start_line(first); - let last_start = node_start_line(last); - let last_end = node_end_line(last); - let span_lines = last_end.saturating_sub(first_start).saturating_add(1); - if span_lines < min_body_lines { - return; - } - // Use the line of the first node's last visible content as the lower bound - // (some grammars include trailing newlines in the node range, which would - // otherwise place `end_line` on the next sibling's first line). - let first_content_end = node_content_end_line(first).min(last_start.saturating_sub(1)); - let start = first_content_end.saturating_add(1); - let end = last_start.saturating_sub(1); - if start <= end { - spans.push(LineSpan { start, end }); - } -} - -fn node_start_line(node: Node<'_>) -> u32 { - node - .start_position() - .row - .saturating_add(1) - .min(u32::MAX as usize) as u32 -} - -fn node_end_line(node: Node<'_>) -> u32 { - node - .end_position() - .row - .saturating_add(1) - .min(u32::MAX as usize) as u32 -} - -/// Last source line containing a content byte from `node`. -/// -/// Tree-sitter reports `end_position` as the position one past the last byte. -/// When that byte is a newline, the resulting position lands at column 0 of -/// the next row, which makes the naive `row + 1` answer one greater than the -/// row of the last visible content. This helper subtracts that off. -fn node_content_end_line(node: Node<'_>) -> u32 { - let pos = node.end_position(); - let row = if pos.column == 0 && pos.row > 0 { - pos.row - 1 - } else { - pos.row - }; - row.saturating_add(1).min(u32::MAX as usize) as u32 -} - -fn node_line_count(node: Node<'_>) -> u32 { - node_end_line(node) - .saturating_sub(node_start_line(node)) - .saturating_add(1) -} - -fn is_comment_kind(language: SupportLang, kind: &str) -> bool { - match language { - SupportLang::TypeScript | SupportLang::Tsx | SupportLang::JavaScript => kind == "comment", - SupportLang::Rust => kind == "block_comment", - SupportLang::Python => kind == "comment", - SupportLang::Go => kind == "comment", - SupportLang::Java => kind == "block_comment", - SupportLang::C | SupportLang::Cpp | SupportLang::ObjC => kind == "comment", - SupportLang::CSharp => kind == "comment", - SupportLang::Ruby => kind == "comment", - SupportLang::Php => kind == "comment", - SupportLang::Swift => kind == "comment", - SupportLang::Kotlin => kind == "block_comment", - SupportLang::Scala => kind == "block_comment", - SupportLang::Lua => kind == "comment", - _ => false, - } -} - -fn is_elidable_kind(language: SupportLang, kind: &str) -> bool { - match language { - SupportLang::TypeScript | SupportLang::Tsx | SupportLang::JavaScript => matches!( - kind, - "statement_block" - | "function_body" - | "object" - | "array" - | "template_string" - | "class_body" - | "interface_body" - | "enum_body" - | "object_type" - | "switch_body" - | "jsx_element" - | "jsx_self_closing_element" - ), - SupportLang::Rust => matches!( - kind, - "block" - | "array_expression" - | "tuple_expression" - | "struct_expression" - | "match_block" - | "raw_string_literal" - | "declaration_list" - | "field_declaration_list" - | "ordered_field_declaration_list" - | "enum_variant_list" - | "where_clause" - | "use_list" - | "macro_definition" - | "token_tree" - ), - SupportLang::Python => matches!( - kind, - "block" - | "dictionary" - | "list" | "set" - | "string" - | "tuple" - | "argument_list" - | "parameters" - | "parenthesized_expression" - | "list_comprehension" - | "set_comprehension" - | "dictionary_comprehension" - | "generator_expression" - | "import_from_statement" - | "subscript" - ), - SupportLang::Go => matches!( - kind, - "block" - | "composite_literal" - | "interpreted_string_literal" - | "raw_string_literal" - | "import_spec_list" - | "const_declaration" - | "var_declaration" - | "field_declaration_list" - | "interface_type" - | "expression_switch_statement" - | "type_switch_statement" - | "select_statement" - ), - SupportLang::Java => matches!( - kind, - "block" - | "array_initializer" - | "class_body" - | "interface_body" - | "enum_body" - | "annotation_type_body" - | "constructor_body" - | "switch_block" - | "string_literal" - ), - SupportLang::C => matches!( - kind, - "compound_statement" - | "initializer_list" - | "string_literal" - | "field_declaration_list" - | "enumerator_list" - | "concatenated_string" - ), - SupportLang::Cpp => matches!( - kind, - "compound_statement" - | "initializer_list" - | "string_literal" - | "field_declaration_list" - | "enumerator_list" - | "concatenated_string" - | "declaration_list" - | "raw_string_literal" - | "requires_clause" - ), - SupportLang::ObjC => matches!( - kind, - "compound_statement" - | "initializer_list" - | "string_literal" - | "protocol_declaration" - | "class_interface" - | "class_implementation" - | "instance_variables" - | "array_literal" - | "dictionary_literal" - ), - SupportLang::CSharp => matches!( - kind, - "block" - | "initializer_expression" - | "array_initializer_expression" - | "declaration_list" - | "enum_member_declaration_list" - | "switch_expression" - | "raw_string_literal" - | "interpolated_string_expression" - ), - SupportLang::Ruby => matches!( - kind, - "body_statement" - | "method" - | "do_block" - | "array" - | "hash" | "block" - | "case" | "heredoc_body" - ), - SupportLang::Php => matches!( - kind, - "compound_statement" - | "array_creation_expression" - | "declaration_list" - | "enum_declaration_list" - | "match_block" - | "heredoc" - | "nowdoc" - ), - SupportLang::Swift => matches!( - kind, - "function_body" - | "array_literal" - | "dictionary_literal" - | "multi_line_string_literal" - | "class_body" - | "protocol_body" - | "enum_class_body" - | "computed_property" - | "lambda_literal" - ), - SupportLang::Kotlin => matches!( - kind, - "function_body" - | "collection_literal" - | "multi_line_string_literal" - | "class_body" - | "enum_class_body" - | "when_expression" - | "import_list" - ), - SupportLang::Scala => matches!( - kind, - "block" - | "collection_literal" - | "template_body" - | "enum_body" - | "match_expression" - | "for_expression" - | "string" - ), - SupportLang::Lua => matches!(kind, "block" | "table_constructor" | "string"), - SupportLang::Perl => { - matches!(kind, "block" | "list_expression" | "heredoc_content" | "regexp_content") - }, - SupportLang::Dart => matches!( - kind, - "block" - | "function_expression_body" - | "class_body" - | "enum_body" - | "extension_body" - | "mixin_body" - | "list_literal" - | "set_or_map_literal" - | "string_literal" - ), - SupportLang::Bash => matches!( - kind, - "compound_statement" - | "if_statement" - | "case_statement" - | "do_group" - | "subshell" - | "array" - | "heredoc_body" - ), - SupportLang::Powershell => matches!( - kind, - "script_block" - | "statement_block" - | "class_statement" - | "param_block" - | "hash_literal_expression" - | "array_expression" - | "expandable_here_string_literal" - | "verbatim_here_string_characters" - ), - SupportLang::Haskell => matches!( - kind, - "imports" - | "data_type" - | "class" - | "instance" - | "function" - | "do" | "case" - | "let" | "local_binds" - | "list" | "tuple" - ), - SupportLang::Ocaml => matches!( - kind, - "structure" - | "signature" - | "variant_declaration" - | "record_declaration" - | "match_expression" - | "match_case" - | "let_expression" - | "value_definition" - | "list_expression" - ), - SupportLang::Elixir => matches!(kind, "do_block" | "list" | "map" | "string" | "sigil"), - SupportLang::Erlang => matches!( - kind, - "fun_decl" - | "case_expr" - | "if_expr" - | "receive_expr" - | "record_decl" - | "list" | "map_expr" - | "tuple" - ), - SupportLang::Clojure => { - matches!(kind, "list_lit" | "map_lit" | "vec_lit" | "set_lit" | "str_lit") - }, - SupportLang::Solidity => { - matches!(kind, "contract_body" | "function_body" | "struct_body" | "enum_body") - }, - SupportLang::Sql => matches!(kind, "column_definitions" | "case"), - SupportLang::Zig => matches!(kind, "Block" | "ContainerDecl" | "InitList"), - SupportLang::Odin => matches!( - kind, - "block" | "struct_declaration" | "enum_declaration" | "union_declaration" | "struct" - ), - SupportLang::Verilog => matches!( - kind, - "module_declaration" - | "seq_block" - | "case_statement" - | "function_declaration" - | "task_declaration" - | "list_of_port_declarations" - ), - SupportLang::Tlaplus => matches!(kind, "module" | "theorem" | "let_in"), - SupportLang::Nix => matches!( - kind, - "attrset_expression" | "list_expression" | "let_expression" | "indented_string_expression" - ), - SupportLang::Proto => matches!(kind, "message_body" | "enum_body" | "oneof" | "service"), - SupportLang::Julia => matches!( - kind, - "function_definition" - | "struct_definition" - | "module_definition" - | "do_clause" - | "vector_expression" - | "string_literal" - ), - SupportLang::R => matches!(kind, "braced_expression" | "call" | "string"), - SupportLang::Starlark => matches!(kind, "block" | "list" | "dictionary" | "string"), - SupportLang::Astro => { - matches!(kind, "frontmatter_js_block" | "script_element" | "style_element" | "element") - }, - SupportLang::Vue => { - matches!(kind, "template_element" | "script_element" | "style_element" | "element") - }, - SupportLang::Svelte => matches!(kind, "script_element" | "style_element" | "element"), - SupportLang::Html => matches!(kind, "element" | "script_element" | "style_element"), - SupportLang::Css => matches!(kind, "block" | "keyframe_block_list"), - SupportLang::Json => matches!(kind, "object" | "array"), - SupportLang::Xml => kind == "element", - SupportLang::Markdown => matches!(kind, "fenced_code_block" | "pipe_table" | "list"), - SupportLang::Graphql => matches!( - kind, - "fields_definition" - | "enum_values_definition" - | "input_fields_definition" - | "schema_definition" - ), - SupportLang::Hcl => matches!(kind, "body" | "object"), - SupportLang::Dockerfile => kind == "shell_command", - SupportLang::Cmake => matches!(kind, "argument_list" | "body"), - SupportLang::Make => kind == "recipe", - SupportLang::Just => kind == "recipe_body", - // Skip: data formats with no closing-token anchor (Yaml mappings, - // Toml tables, Ini sections), the diff format whose informational - // content IS the lines inside hunks, and the leaf-token-only Regex - // grammar. Eliding any of these deletes the only content worth - // reading. - SupportLang::Yaml - | SupportLang::Toml - | SupportLang::Ini - | SupportLang::Diff - | SupportLang::Regex => false, - } -} - -fn is_groupable_kind(language: SupportLang, kind: &str) -> bool { - match language { - SupportLang::TypeScript | SupportLang::Tsx | SupportLang::JavaScript => { - kind == "import_statement" - }, - SupportLang::Rust => matches!(kind, "use_declaration" | "extern_crate_declaration"), - SupportLang::Python => { - matches!(kind, "import_statement" | "import_from_statement" | "future_import_statement") - }, - SupportLang::Go => kind == "import_declaration", - SupportLang::Java => kind == "import_declaration", - SupportLang::C | SupportLang::Cpp => kind == "preproc_include", - SupportLang::ObjC => matches!(kind, "preproc_include" | "import_declaration"), - SupportLang::CSharp => kind == "using_directive", - SupportLang::Php => kind == "namespace_use_declaration", - SupportLang::Swift => kind == "import_declaration", - SupportLang::Scala => matches!(kind, "import_declaration" | "import"), - SupportLang::Dart => kind == "import_or_export", - SupportLang::Ocaml => kind == "open_module", - SupportLang::Solidity => kind == "import_directive", - SupportLang::Julia => matches!(kind, "import_statement" | "using_statement"), - SupportLang::Proto => kind == "import", - SupportLang::Perl => kind == "use_statement", - // Languages where imports either have no run pattern, are wrapped in a - // single AST node already covered by `is_elidable_kind` (Kotlin's - // `import_list`, Haskell's `imports`), or live inside a too-generic - // container (Powershell `statement_list`). - SupportLang::Kotlin - | SupportLang::Haskell - | SupportLang::Powershell - | SupportLang::Ruby - | SupportLang::Lua - | SupportLang::Elixir - | SupportLang::Erlang - | SupportLang::Clojure - | SupportLang::Sql - | SupportLang::Zig - | SupportLang::Odin - | SupportLang::Verilog - | SupportLang::Tlaplus - | SupportLang::Nix - | SupportLang::R - | SupportLang::Starlark - | SupportLang::Bash - | SupportLang::Astro - | SupportLang::Vue - | SupportLang::Svelte - | SupportLang::Html - | SupportLang::Css - | SupportLang::Json - | SupportLang::Xml - | SupportLang::Markdown - | SupportLang::Graphql - | SupportLang::Hcl - | SupportLang::Dockerfile - | SupportLang::Cmake - | SupportLang::Make - | SupportLang::Just - | SupportLang::Yaml - | SupportLang::Toml - | SupportLang::Ini - | SupportLang::Diff - | SupportLang::Regex => false, - } -} - -fn normalize_spans(mut spans: Vec<LineSpan>, total_lines: u32) -> Vec<LineSpan> { - if total_lines == 0 { - return Vec::new(); - } - spans.retain(|span| span.start <= span.end && span.start <= total_lines); - for span in &mut spans { - span.end = span.end.min(total_lines); - } - spans.sort_by_key(|span| (span.start, span.end)); - let mut merged: Vec<LineSpan> = Vec::new(); - for span in spans { - if let Some(last) = merged.last_mut() - && span.start <= last.end.saturating_add(1) - { - last.end = last.end.max(span.end); - continue; - } - merged.push(span); - } - merged -} - -fn build_segments(source: &str, total_lines: u32, spans: &[LineSpan]) -> Vec<SummarySegment> { - if total_lines == 0 { - return Vec::new(); - } - let source_lines: Vec<&str> = source.lines().collect(); - let elided_lines = spans - .iter() - .flat_map(|span| span.start..=span.end) - .collect::<BTreeSet<_>>(); - let mut segments = Vec::new(); - let mut current_kind: Option<&str> = None; - let mut current_start = 1; - let mut current_lines: Vec<&str> = Vec::new(); - - for line_number in 1..=total_lines { - let is_elided = elided_lines.contains(&line_number); - let kind = if is_elided { "elided" } else { "kept" }; - if current_kind.is_some_and(|existing| existing != kind) { - push_segment( - &mut segments, - current_kind.expect("kind set"), - current_start, - line_number - 1, - ¤t_lines, - ); - current_start = line_number; - current_lines.clear(); - } - current_kind = Some(kind); - if !is_elided { - let index = line_number.saturating_sub(1) as usize; - current_lines.push(source_lines.get(index).copied().unwrap_or_default()); - } - } - - if let Some(kind) = current_kind { - push_segment(&mut segments, kind, current_start, total_lines, ¤t_lines); - } - segments -} - -fn push_segment( - segments: &mut Vec<SummarySegment>, - kind: &str, - start_line: u32, - end_line: u32, - lines: &[&str], -) { - segments.push(SummarySegment { - kind: kind.to_string(), - start_line, - end_line, - text: (kind == "kept").then(|| lines.join("\n")), - }); -} - -#[cfg(test)] -mod tests { - use super::*; - - fn summarize(code: &str, path: &str) -> SummaryResult { - summarize_code(SummaryOptions { - code: code.to_string(), - lang: None, - path: Some(path.to_string()), - min_body_lines: None, - min_comment_lines: None, - }) - .expect("summary succeeds") - } - - fn segment_kinds(result: &SummaryResult) -> Vec<&str> { - result - .segments - .iter() - .map(|segment| segment.kind.as_str()) - .collect() - } - - #[test] - fn summarizes_typescript_function_body() { - let result = summarize( - "export function greet(name: string): string {\n\tconst clean = name.trim();\n\tconst \ - label = clean || 'world';\n\treturn `hello ${label}`;\n}\n", - "fixture.ts", - ); - - assert!(result.parsed); - assert!(result.elided); - assert_eq!(result.language.as_deref(), Some("typescript")); - assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); - assert_eq!( - result.segments[0].text.as_deref(), - Some("export function greet(name: string): string {") - ); - assert_eq!(result.segments[1].start_line, 2); - assert_eq!(result.segments[1].end_line, 4); - assert_eq!(result.segments[2].text.as_deref(), Some("}")); - } - - #[test] - fn summarizes_rust_method_body_but_keeps_impl_boundaries() { - let result = summarize( - "struct Greeter;\n\nimpl Greeter {\n\tfn greet(&self) -> String {\n\t\tlet name = \ - \"world\";\n\t\tlet label = name.to_uppercase();\n\t\tformat!(\"hello \ - {label}\")\n\t}\n}\n", - "fixture.rs", - ); - - assert!(result.parsed); - assert!(result.elided); - let rendered = result - .segments - .iter() - .map(|segment| segment.text.clone().unwrap_or_else(|| "...".to_string())) - .collect::<Vec<_>>() - .join("\n"); - assert!(rendered.contains("impl Greeter {\n...\n}")); - } - - #[test] - fn summarizes_python_function_body() { - let result = - summarize( - "class Greeter:\n def greet(self, name: str) -> str:\n clean = \ - name.strip()\n label = clean or 'world'\n return f'hello {label}'\n", - "fixture.py", - ); - - assert!(result.parsed); - assert!(result.elided); - assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); - assert!( - result.segments[0] - .text - .as_deref() - .unwrap_or_default() - .contains("def greet") - ); - assert!( - result.segments[2] - .text - .as_deref() - .unwrap_or_default() - .contains("return") - ); - } - - #[test] - fn min_body_lines_controls_short_body_elision() { - let code = "function small() {\n\treturn 1;\n}\n"; - let default_result = summarize(code, "fixture.ts"); - assert!(default_result.parsed); - assert!(!default_result.elided); - - let override_result = summarize_code(SummaryOptions { - code: code.to_string(), - lang: Some("typescript".to_string()), - path: None, - min_body_lines: Some(3), - min_comment_lines: None, - }) - .expect("summary succeeds"); - assert!(override_result.elided); - } - - #[test] - fn parse_failure_falls_back_to_unparsed() { - let result = summarize("export function broken( {\n", "fixture.ts"); - assert!(!result.parsed); - assert!(!result.elided); - assert_eq!(result.segments.len(), 1); - } - - #[test] - fn unsupported_language_is_unparsed() { - let result = summarize("plain text\nwith lines\n", "fixture.txt"); - assert!(!result.parsed); - assert_eq!(result.segments[0].text.as_deref(), Some("plain text\nwith lines\n")); - } - - #[test] - fn summarizes_typescript_interface_body() { - let result = summarize( - "export interface Args {\n\tcwd?: string;\n\tprovider?: string;\n\tmodel?: \ - string;\n\tapiKey?: string;\n}\n", - "fixture.ts", - ); - - assert!(result.parsed); - assert!(result.elided); - assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); - assert_eq!(result.segments[0].text.as_deref(), Some("export interface Args {")); - assert_eq!(result.segments[2].text.as_deref(), Some("}")); - } - - #[test] - fn summarizes_typescript_class_body() { - let result = summarize( - "export class Greeter {\n\tname: string = \"world\";\n\tlength(): number { return \ - this.name.length; }\n\tgreet(): string { return this.name; }\n\tshout(): string { \ - return this.name.toUpperCase(); }\n}\n", - "fixture.ts", - ); - - assert!(result.parsed); - assert!(result.elided); - assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); - assert!( - result.segments[0] - .text - .as_deref() - .unwrap_or_default() - .contains("class Greeter") - ); - assert_eq!(result.segments[2].text.as_deref(), Some("}")); - } - - #[test] - fn summarizes_rust_trait_declaration_list() { - let result = summarize( - "pub trait Greeter {\n\tfn greet(&self) -> String;\n\tfn length(&self) -> usize;\n\tfn \ - shout(&self) -> String;\n\tfn whisper(&self) -> String;\n}\n", - "fixture.rs", - ); - - assert!(result.parsed); - assert!(result.elided); - assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); - assert_eq!(result.segments[0].text.as_deref(), Some("pub trait Greeter {")); - assert_eq!(result.segments[2].text.as_deref(), Some("}")); - } - - #[test] - fn summarizes_java_class_body() { - let result = summarize( - "public class Greeter {\n\tprivate String name;\n\tpublic Greeter(String n) { this.name \ - = n; }\n\tpublic String greet() { return name; }\n\tpublic int length() { return \ - name.length(); }\n}\n", - "fixture.java", - ); - - assert!(result.parsed); - assert!(result.elided); - assert_eq!(segment_kinds(&result), vec!["kept", "elided", "kept"]); - assert!( - result.segments[0] - .text - .as_deref() - .unwrap_or_default() - .contains("class Greeter") - ); - assert_eq!(result.segments[2].text.as_deref(), Some("}")); - } - - #[test] - fn summarizes_typescript_import_run() { - let code = "import a from \"a\";\nimport b from \"b\";\nimport c from \"c\";\nimport d from \ - \"d\";\nimport e from \"e\";\nimport f from \"f\";\n\nexport function main() \ - {}\n"; - let result = summarize(code, "fixture.ts"); - - assert!(result.parsed); - assert!(result.elided); - // Lines 2-5 are between the first and last imports and must be elided. - let elided = result - .segments - .iter() - .find(|seg| seg.kind == "elided") - .expect("elided segment"); - assert_eq!(elided.start_line, 2); - assert_eq!(elided.end_line, 5); - // First import line is kept. - assert!( - result.segments[0] - .text - .as_deref() - .unwrap_or_default() - .starts_with("import a from") - ); - } - - #[test] - fn does_not_elide_short_typescript_import_run() { - // 3 imports → total span 3 lines, below default min_body_lines (4). - let result = summarize( - "import a from \"a\";\nimport b from \"b\";\nimport c from \"c\";\n", - "fixture.ts", - ); - assert!(result.parsed); - assert!(!result.elided); - } - - #[test] - fn summarizes_python_import_run() { - let code = "import os\nimport sys\nfrom typing import List\nfrom pathlib import \ - Path\nimport json\nimport re\n\nprint('go')\n"; - let result = summarize(code, "fixture.py"); - - assert!(result.parsed); - assert!(result.elided); - let elided = result - .segments - .iter() - .find(|seg| seg.kind == "elided") - .expect("elided segment"); - assert_eq!(elided.start_line, 2); - assert_eq!(elided.end_line, 5); - } - - #[test] - fn summarizes_c_preproc_include_run() { - // C grammar puts each #include's `end_position` at column 0 of the next - // row (the trailing `\n`). Without `node_content_end_line`, the run - // elision would emit a span that starts past the second include and - // only collapse the third — verify the boundary statements stay - // visible and the middle is collapsed. - let code = "#include <stdio.h>\n#include \"a.h\"\n#include \"b.h\"\n#include \ - \"c.h\"\n#include <string.h>\nint main(void) { return 0; }\n"; - let result = summarize(code, "fixture.c"); - - assert!(result.parsed); - assert!(result.elided); - let elided = result - .segments - .iter() - .find(|seg| seg.kind == "elided") - .expect("elided segment"); - assert_eq!(elided.start_line, 2); - assert_eq!(elided.end_line, 4); - } - - #[test] - fn summarizes_rust_use_run() { - let code = "use std::fs;\nuse std::path::Path;\nuse std::collections::HashMap;\nuse \ - std::sync::Arc;\nuse std::io;\n\nfn main() {}\n"; - let result = summarize(code, "fixture.rs"); - - assert!(result.parsed); - assert!(result.elided); - let elided = result - .segments - .iter() - .find(|seg| seg.kind == "elided") - .expect("elided segment"); - assert_eq!(elided.start_line, 2); - assert_eq!(elided.end_line, 4); - } + .map(Into::into) + .map_err(|error| Error::from_reason(error.to_string())) } diff --git a/crates/pi-natives/src/task.rs b/crates/pi-natives/src/task.rs index 62619f25f..3f75edd6e 100644 --- a/crates/pi-natives/src/task.rs +++ b/crates/pi-natives/src/task.rs @@ -27,17 +27,10 @@ //! } //! ``` -use std::{ - future::Future, - sync::{ - Arc, Weak, - atomic::{AtomicU8, Ordering}, - }, - time::{Duration, Instant}, -}; +use std::future::Future; use napi::{Env, Error, Result, Task, bindgen_prelude::*}; -use tokio::sync::Notify; +use pi_shell::cancel as core_cancel; use crate::prof::profile_region; @@ -47,55 +40,31 @@ use crate::prof::profile_region; /// Reason for task abortion. #[derive(Debug, Clone, Copy)] -#[repr(u8)] pub enum AbortReason { - Unknown = 1, - Timeout = 2, - Signal = 3, - User = 4, + Unknown, + Timeout, + Signal, + User, } -impl TryFrom<u8> for AbortReason { - type Error = (); - - fn try_from(value: u8) -> std::result::Result<Self, ()> { +impl From<core_cancel::AbortReason> for AbortReason { + fn from(value: core_cancel::AbortReason) -> Self { match value { - 0 => Err(()), - 2 => Ok(Self::Timeout), - 3 => Ok(Self::Signal), - 4 => Ok(Self::User), - _ => Ok(Self::Unknown), + core_cancel::AbortReason::Unknown => Self::Unknown, + core_cancel::AbortReason::Timeout => Self::Timeout, + core_cancel::AbortReason::Signal => Self::Signal, + core_cancel::AbortReason::User => Self::User, } } } -#[derive(Default)] -struct Flag { - reason: AtomicU8, - notifier: Notify, -} - -impl Flag { - fn cause(&self) -> Option<AbortReason> { - self.reason.load(Ordering::Relaxed).try_into().ok() - } - - async fn wait(&self) -> AbortReason { - if let Some(reason) = self.cause() { - return reason; - } - let notifier = self.notifier.notified(); - if let Some(reason) = self.cause() { - return reason; - } - notifier.await; - self.cause().unwrap_or(AbortReason::Unknown) - } - - fn abort(&self, reason: AbortReason) { - let old = self.reason.swap(reason as u8, Ordering::SeqCst); - if old == 0 { - self.notifier.notify_waiters(); +impl From<AbortReason> for core_cancel::AbortReason { + fn from(value: AbortReason) -> Self { + match value { + AbortReason::Unknown => Self::Unknown, + AbortReason::Timeout => Self::Timeout, + AbortReason::Signal => Self::Signal, + AbortReason::User => Self::User, } } } @@ -106,8 +75,7 @@ impl Flag { /// cancellation requests from timeouts or abort signals. #[derive(Clone, Default)] pub struct CancelToken { - deadline: Option<Instant>, - flag: Option<Arc<Flag>>, + core: core_cancel::CancelToken, } impl From<()> for CancelToken { @@ -119,21 +87,10 @@ impl From<()> for CancelToken { impl CancelToken { /// Create a new cancel token from optional timeout and abort signal. pub fn new(timeout_ms: Option<u32>, signal: Option<Unknown>) -> Self { - let mut result = Self::default(); - if let Some(signal) = signal.and_then(|s| AbortSignal::from_unknown(s).ok()) { - let flag = Arc::new(Flag::default()); - signal.on_abort({ - let weak = Arc::downgrade(&flag); - move || { - if let Some(flag) = weak.upgrade() { - flag.abort(AbortReason::Signal); - } - } - }); - result.flag = Some(flag); - } - if let Some(timeout_ms) = timeout_ms { - result.deadline = Some(Instant::now() + Duration::from_millis(timeout_ms as u64)); + let mut result = Self { core: core_cancel::CancelToken::new(timeout_ms) }; + if let Some(signal) = signal.and_then(|value| AbortSignal::from_unknown(value).ok()) { + let abort_token = result.emplace_abort_token(); + signal.on_abort(move || abort_token.abort(AbortReason::Signal)); } result } @@ -143,92 +100,45 @@ impl CancelToken { /// Returns `Ok(())` if work should continue, or an error if cancelled. /// Call this periodically in long-running loops. pub fn heartbeat(&self) -> Result<()> { - if let Some(flag) = &self.flag - && let Some(reason) = flag.cause() - { - return Err(Error::from_reason(format!("Aborted: {reason:?}"))); - } - if let Some(deadline) = self.deadline - && deadline < Instant::now() - { - return Err(Error::from_reason("Aborted: Timeout")); - } - Ok(()) + self + .core + .heartbeat() + .map_err(|err| Error::from_reason(err.to_string())) } /// Wait for the cancel token to be aborted. pub async fn wait(&self) -> AbortReason { - let flag = self.flag.as_ref(); - if let Some(flag) = flag.and_then(|f| f.cause()) { - return flag; - } - let fflag = async { - let Some(flag) = self.flag.as_ref() else { - return std::future::pending().await; - }; - flag.wait().await - }; - - let fttl = async { - let Some(ttl) = self.deadline else { - return std::future::pending().await; - }; - tokio::time::sleep_until(ttl.into()).await; - AbortReason::Timeout - }; - - let fuser = async { - if tokio::signal::ctrl_c().await.is_err() { - return std::future::pending().await; - } - AbortReason::User - }; - - tokio::select! { - reason = fflag => reason, - reason = fttl => reason, - reason = fuser => reason, - } + self.core.wait().await.into() } /// Get an abort token for external cancellation. pub fn abort_token(&self) -> AbortToken { - AbortToken(self.flag.as_ref().map(Arc::downgrade)) + AbortToken(self.core.abort_token()) } /// Emplaces a cancel token if there is none, returns the abort token. pub fn emplace_abort_token(&mut self) -> AbortToken { - AbortToken(Some(Arc::downgrade(self.flag.get_or_insert_default()))) + AbortToken(self.core.emplace_abort_token()) } /// Check if already aborted (non-blocking). pub fn aborted(&self) -> bool { - if let Some(flag) = &self.flag - && flag.cause().is_some() - { - return true; - } - if let Some(deadline) = self.deadline - && deadline < Instant::now() - { - return true; - } - false + self.core.aborted() + } + + pub fn into_core(self) -> core_cancel::CancelToken { + self.core } } /// Token for requesting cancellation from outside the task. #[derive(Clone, Default)] -pub struct AbortToken(Option<Weak<Flag>>); +pub struct AbortToken(core_cancel::AbortToken); impl AbortToken { /// Request cancellation of the associated task. pub fn abort(&self, reason: AbortReason) { - if let Some(flag) = &self.0 - && let Some(flag) = flag.upgrade() - { - flag.abort(reason); - } + self.0.abort(reason.into()); } } diff --git a/crates/pi-shell/Cargo.toml b/crates/pi-shell/Cargo.toml new file mode 100644 index 000000000..053033f17 --- /dev/null +++ b/crates/pi-shell/Cargo.toml @@ -0,0 +1,37 @@ +[package] +name = "pi-shell" +version.workspace = true +edition.workspace = true +license.workspace = true +authors.workspace = true +repository.workspace = true + +[lints] +workspace = true + +[dependencies] +anyhow = "1.0" +tokio = { version = "1", features = ["full"] } +tokio-util = { version = "0.7", features = ["full"] } +brush-core = { version = "0.5.0", path = "../brush-core-vendored" } +brush-builtins = { version = "0.2.0", path = "../brush-builtins-vendored" } +brush-parser = "0.3" +clap = { version = "4", features = ["derive"] } +os_pipe = "1" +serde = { version = "1.0", features = ["derive"] } +serde_json = { version = "1.0", features = ["preserve_order"] } +toml = "1.1" +regex = "1" +xxhash-rust = { version = "0.8", features = ["xxh64"] } + +[target.'cfg(unix)'.dependencies] +libc = "0.2" + +[target.'cfg(windows)'.dependencies] +winreg = "0.56" +windows-sys = { version = "0.61", features = [ + "Win32_Foundation", + "Win32_Storage_ProjectedFileSystem", + "Win32_System_Com", + "Win32_System_LibraryLoader", +] } diff --git a/crates/pi-shell/build.rs b/crates/pi-shell/build.rs new file mode 100644 index 000000000..397e1a54c --- /dev/null +++ b/crates/pi-shell/build.rs @@ -0,0 +1,62 @@ +use std::{ + env, + fmt::Write as _, + fs, + path::{Path, PathBuf}, +}; + +fn main() { + generate_minimizer_builtin_filters(); +} + +fn generate_minimizer_builtin_filters() { + let manifest_dir = env::var("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR should be set"); + let defs_dir = Path::new(&manifest_dir) + .join("src") + .join("minimizer") + .join("defs"); + let out_dir = PathBuf::from(env::var("OUT_DIR").expect("OUT_DIR should be set")); + let output_path = out_dir.join("builtin_filters.toml"); + + println!("cargo:rerun-if-changed={}", defs_dir.display()); + + let mut concatenated = + String::from("# Auto-generated by build.rs -- do not edit.\nschema_version = 1\n\n"); + + let mut entries: Vec<PathBuf> = Vec::new(); + if let Ok(read_dir) = fs::read_dir(&defs_dir) { + for entry in read_dir.flatten() { + let path = entry.path(); + if path.extension().and_then(|extension| extension.to_str()) == Some("toml") { + entries.push(path); + } + } + } + entries.sort(); + + for path in entries { + println!("cargo:rerun-if-changed={}", path.display()); + match fs::read_to_string(&path) { + Ok(body) => { + let filename = path + .file_name() + .and_then(|name| name.to_str()) + .unwrap_or("unknown"); + writeln!(concatenated, "# --- {filename} ---").expect("write to String"); + for line in body.lines() { + let trimmed = line.trim_start(); + if trimmed.starts_with("schema_version") { + continue; + } + concatenated.push_str(line); + concatenated.push('\n'); + } + concatenated.push('\n'); + }, + Err(error) => panic!("failed to read filter definition {}: {error}", path.display()), + } + } + + fs::write(&output_path, concatenated) + .unwrap_or_else(|error| panic!("failed to write {}: {error}", output_path.display())); +} diff --git a/crates/pi-shell/src/cancel.rs b/crates/pi-shell/src/cancel.rs new file mode 100644 index 000000000..b27c05212 --- /dev/null +++ b/crates/pi-shell/src/cancel.rs @@ -0,0 +1,161 @@ +use std::{ + sync::{ + Arc, Weak, + atomic::{AtomicU8, Ordering}, + }, + time::{Duration, Instant}, +}; + +use anyhow::{Error, Result}; +use tokio::sync::Notify; + +#[derive(Debug, Clone, Copy)] +#[repr(u8)] +pub enum AbortReason { + Unknown = 1, + Timeout = 2, + Signal = 3, + User = 4, +} + +impl TryFrom<u8> for AbortReason { + type Error = (); + + fn try_from(value: u8) -> std::result::Result<Self, ()> { + match value { + 0 => Err(()), + 2 => Ok(Self::Timeout), + 3 => Ok(Self::Signal), + 4 => Ok(Self::User), + _ => Ok(Self::Unknown), + } + } +} + +#[derive(Default)] +struct Flag { + reason: AtomicU8, + notifier: Notify, +} + +impl Flag { + fn cause(&self) -> Option<AbortReason> { + self.reason.load(Ordering::Relaxed).try_into().ok() + } + + async fn wait(&self) -> AbortReason { + if let Some(reason) = self.cause() { + return reason; + } + let notifier = self.notifier.notified(); + if let Some(reason) = self.cause() { + return reason; + } + notifier.await; + self.cause().unwrap_or(AbortReason::Unknown) + } + + fn abort(&self, reason: AbortReason) { + let old = self.reason.swap(reason as u8, Ordering::SeqCst); + if old == 0 { + self.notifier.notify_waiters(); + } + } +} + +#[derive(Clone, Default)] +pub struct CancelToken { + deadline: Option<Instant>, + flag: Option<Arc<Flag>>, +} + +impl From<()> for CancelToken { + fn from((): ()) -> Self { + Self::default() + } +} + +impl CancelToken { + pub fn new(timeout_ms: Option<u32>) -> Self { + Self::with_timeout(timeout_ms.map(|ms| Duration::from_millis(u64::from(ms)))) + } + + pub fn with_timeout(timeout: Option<Duration>) -> Self { + Self { deadline: timeout.map(|duration| Instant::now() + duration), flag: None } + } + + pub fn heartbeat(&self) -> Result<()> { + if let Some(flag) = &self.flag + && let Some(reason) = flag.cause() + { + return Err(Error::msg(format!("Aborted: {reason:?}"))); + } + if let Some(deadline) = self.deadline + && deadline < Instant::now() + { + return Err(Error::msg("Aborted: Timeout")); + } + Ok(()) + } + + pub async fn wait(&self) -> AbortReason { + if let Some(flag) = self.flag.as_ref().and_then(|flag| flag.cause()) { + return flag; + } + + let by_flag = async { + let Some(flag) = self.flag.as_ref() else { + return std::future::pending().await; + }; + flag.wait().await + }; + + let by_timeout = async { + let Some(deadline) = self.deadline else { + return std::future::pending().await; + }; + tokio::time::sleep_until(deadline.into()).await; + AbortReason::Timeout + }; + + tokio::select! { + reason = by_flag => reason, + reason = by_timeout => reason, + } + } + + pub fn abort_token(&self) -> AbortToken { + AbortToken(self.flag.as_ref().map(Arc::downgrade)) + } + + pub fn emplace_abort_token(&mut self) -> AbortToken { + AbortToken(Some(Arc::downgrade(self.flag.get_or_insert_default()))) + } + + pub fn aborted(&self) -> bool { + if let Some(flag) = &self.flag + && flag.cause().is_some() + { + return true; + } + if let Some(deadline) = self.deadline + && deadline < Instant::now() + { + return true; + } + false + } +} + +#[derive(Clone, Default)] +pub struct AbortToken(Option<Weak<Flag>>); + +impl AbortToken { + pub fn abort(&self, reason: AbortReason) { + if let Some(flag) = &self.0 + && let Some(flag) = flag.upgrade() + { + flag.abort(reason); + } + } +} diff --git a/crates/pi-shell/src/lib.rs b/crates/pi-shell/src/lib.rs new file mode 100644 index 000000000..e1571845d --- /dev/null +++ b/crates/pi-shell/src/lib.rs @@ -0,0 +1,12 @@ +pub mod cancel; +pub mod minimizer; +pub mod process; +pub mod shell; +#[cfg(windows)] +pub mod windows; + +pub use brush_core::commands::{ChildSessionAction, child_session_action}; +pub use shell::{ + MinimizerResult, Shell, ShellExecuteOptions, ShellExecuteResult, ShellOptions, ShellRunOptions, + ShellRunResult, execute_shell, +}; diff --git a/crates/pi-natives/src/shell/minimizer.rs b/crates/pi-shell/src/minimizer.rs similarity index 100% rename from crates/pi-natives/src/shell/minimizer.rs rename to crates/pi-shell/src/minimizer.rs diff --git a/crates/pi-natives/src/shell/minimizer/config.rs b/crates/pi-shell/src/minimizer/config.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/config.rs rename to crates/pi-shell/src/minimizer/config.rs index a52b3d4de..e19bc7736 100644 --- a/crates/pi-natives/src/shell/minimizer/config.rs +++ b/crates/pi-shell/src/minimizer/config.rs @@ -12,15 +12,13 @@ use std::{ sync::Arc, }; -use napi_derive::napi; use serde::Deserialize; -use crate::shell::minimizer::pipeline::{self, PipelineRegistry, SUPPORTED_SCHEMA_VERSION}; +use crate::minimizer::pipeline::{self, PipelineRegistry, SUPPORTED_SCHEMA_VERSION}; const DEFAULT_MAX_CAPTURE_BYTES: u32 = 4 * 1024 * 1024; /// N-API opt-in handle for the minimizer. -#[napi(object)] #[derive(Debug, Clone, Default)] pub struct MinimizerOptions { /// Master switch. Absent / false = disabled. diff --git a/crates/pi-natives/src/shell/minimizer/defs/ansible-playbook.toml b/crates/pi-shell/src/minimizer/defs/ansible-playbook.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/ansible-playbook.toml rename to crates/pi-shell/src/minimizer/defs/ansible-playbook.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/ansible.toml b/crates/pi-shell/src/minimizer/defs/ansible.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/ansible.toml rename to crates/pi-shell/src/minimizer/defs/ansible.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/basedpyright.toml b/crates/pi-shell/src/minimizer/defs/basedpyright.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/basedpyright.toml rename to crates/pi-shell/src/minimizer/defs/basedpyright.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/biome.toml b/crates/pi-shell/src/minimizer/defs/biome.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/biome.toml rename to crates/pi-shell/src/minimizer/defs/biome.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/brew-install.toml b/crates/pi-shell/src/minimizer/defs/brew-install.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/brew-install.toml rename to crates/pi-shell/src/minimizer/defs/brew-install.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/bundle-install.toml b/crates/pi-shell/src/minimizer/defs/bundle-install.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/bundle-install.toml rename to crates/pi-shell/src/minimizer/defs/bundle-install.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/composer-install.toml b/crates/pi-shell/src/minimizer/defs/composer-install.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/composer-install.toml rename to crates/pi-shell/src/minimizer/defs/composer-install.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/df.toml b/crates/pi-shell/src/minimizer/defs/df.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/df.toml rename to crates/pi-shell/src/minimizer/defs/df.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/dotnet-build.toml b/crates/pi-shell/src/minimizer/defs/dotnet-build.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/dotnet-build.toml rename to crates/pi-shell/src/minimizer/defs/dotnet-build.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/du.toml b/crates/pi-shell/src/minimizer/defs/du.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/du.toml rename to crates/pi-shell/src/minimizer/defs/du.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/fail2ban-client.toml b/crates/pi-shell/src/minimizer/defs/fail2ban-client.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/fail2ban-client.toml rename to crates/pi-shell/src/minimizer/defs/fail2ban-client.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/fail2ban.toml b/crates/pi-shell/src/minimizer/defs/fail2ban.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/fail2ban.toml rename to crates/pi-shell/src/minimizer/defs/fail2ban.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/gcc.toml b/crates/pi-shell/src/minimizer/defs/gcc.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/gcc.toml rename to crates/pi-shell/src/minimizer/defs/gcc.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/gcloud.toml b/crates/pi-shell/src/minimizer/defs/gcloud.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/gcloud.toml rename to crates/pi-shell/src/minimizer/defs/gcloud.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/gradle.toml b/crates/pi-shell/src/minimizer/defs/gradle.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/gradle.toml rename to crates/pi-shell/src/minimizer/defs/gradle.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/hadolint.toml b/crates/pi-shell/src/minimizer/defs/hadolint.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/hadolint.toml rename to crates/pi-shell/src/minimizer/defs/hadolint.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/helm.toml b/crates/pi-shell/src/minimizer/defs/helm.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/helm.toml rename to crates/pi-shell/src/minimizer/defs/helm.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/iptables.toml b/crates/pi-shell/src/minimizer/defs/iptables.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/iptables.toml rename to crates/pi-shell/src/minimizer/defs/iptables.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/jira.toml b/crates/pi-shell/src/minimizer/defs/jira.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/jira.toml rename to crates/pi-shell/src/minimizer/defs/jira.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/jj.toml b/crates/pi-shell/src/minimizer/defs/jj.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/jj.toml rename to crates/pi-shell/src/minimizer/defs/jj.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/jq.toml b/crates/pi-shell/src/minimizer/defs/jq.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/jq.toml rename to crates/pi-shell/src/minimizer/defs/jq.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/just.toml b/crates/pi-shell/src/minimizer/defs/just.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/just.toml rename to crates/pi-shell/src/minimizer/defs/just.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/liquibase.toml b/crates/pi-shell/src/minimizer/defs/liquibase.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/liquibase.toml rename to crates/pi-shell/src/minimizer/defs/liquibase.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/make.toml b/crates/pi-shell/src/minimizer/defs/make.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/make.toml rename to crates/pi-shell/src/minimizer/defs/make.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/markdownlint.toml b/crates/pi-shell/src/minimizer/defs/markdownlint.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/markdownlint.toml rename to crates/pi-shell/src/minimizer/defs/markdownlint.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/maven.toml b/crates/pi-shell/src/minimizer/defs/maven.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/maven.toml rename to crates/pi-shell/src/minimizer/defs/maven.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/mise.toml b/crates/pi-shell/src/minimizer/defs/mise.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/mise.toml rename to crates/pi-shell/src/minimizer/defs/mise.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/mix-compile.toml b/crates/pi-shell/src/minimizer/defs/mix-compile.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/mix-compile.toml rename to crates/pi-shell/src/minimizer/defs/mix-compile.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/mix-format.toml b/crates/pi-shell/src/minimizer/defs/mix-format.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/mix-format.toml rename to crates/pi-shell/src/minimizer/defs/mix-format.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/mix.toml b/crates/pi-shell/src/minimizer/defs/mix.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/mix.toml rename to crates/pi-shell/src/minimizer/defs/mix.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/mvn-build.toml b/crates/pi-shell/src/minimizer/defs/mvn-build.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/mvn-build.toml rename to crates/pi-shell/src/minimizer/defs/mvn-build.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/nx.toml b/crates/pi-shell/src/minimizer/defs/nx.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/nx.toml rename to crates/pi-shell/src/minimizer/defs/nx.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/ollama.toml b/crates/pi-shell/src/minimizer/defs/ollama.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/ollama.toml rename to crates/pi-shell/src/minimizer/defs/ollama.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/oxlint.toml b/crates/pi-shell/src/minimizer/defs/oxlint.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/oxlint.toml rename to crates/pi-shell/src/minimizer/defs/oxlint.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/ping.toml b/crates/pi-shell/src/minimizer/defs/ping.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/ping.toml rename to crates/pi-shell/src/minimizer/defs/ping.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/pio-run.toml b/crates/pi-shell/src/minimizer/defs/pio-run.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/pio-run.toml rename to crates/pi-shell/src/minimizer/defs/pio-run.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/pio.toml b/crates/pi-shell/src/minimizer/defs/pio.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/pio.toml rename to crates/pi-shell/src/minimizer/defs/pio.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/poetry-install.toml b/crates/pi-shell/src/minimizer/defs/poetry-install.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/poetry-install.toml rename to crates/pi-shell/src/minimizer/defs/poetry-install.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/pre-commit.toml b/crates/pi-shell/src/minimizer/defs/pre-commit.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/pre-commit.toml rename to crates/pi-shell/src/minimizer/defs/pre-commit.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/ps.toml b/crates/pi-shell/src/minimizer/defs/ps.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/ps.toml rename to crates/pi-shell/src/minimizer/defs/ps.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/quarto-render.toml b/crates/pi-shell/src/minimizer/defs/quarto-render.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/quarto-render.toml rename to crates/pi-shell/src/minimizer/defs/quarto-render.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/quarto.toml b/crates/pi-shell/src/minimizer/defs/quarto.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/quarto.toml rename to crates/pi-shell/src/minimizer/defs/quarto.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/rsync.toml b/crates/pi-shell/src/minimizer/defs/rsync.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/rsync.toml rename to crates/pi-shell/src/minimizer/defs/rsync.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/shellcheck.toml b/crates/pi-shell/src/minimizer/defs/shellcheck.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/shellcheck.toml rename to crates/pi-shell/src/minimizer/defs/shellcheck.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/shopify-theme.toml b/crates/pi-shell/src/minimizer/defs/shopify-theme.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/shopify-theme.toml rename to crates/pi-shell/src/minimizer/defs/shopify-theme.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/skopeo.toml b/crates/pi-shell/src/minimizer/defs/skopeo.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/skopeo.toml rename to crates/pi-shell/src/minimizer/defs/skopeo.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/sops.toml b/crates/pi-shell/src/minimizer/defs/sops.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/sops.toml rename to crates/pi-shell/src/minimizer/defs/sops.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/spring-boot.toml b/crates/pi-shell/src/minimizer/defs/spring-boot.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/spring-boot.toml rename to crates/pi-shell/src/minimizer/defs/spring-boot.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/ssh.toml b/crates/pi-shell/src/minimizer/defs/ssh.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/ssh.toml rename to crates/pi-shell/src/minimizer/defs/ssh.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/stat.toml b/crates/pi-shell/src/minimizer/defs/stat.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/stat.toml rename to crates/pi-shell/src/minimizer/defs/stat.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/swift-build.toml b/crates/pi-shell/src/minimizer/defs/swift-build.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/swift-build.toml rename to crates/pi-shell/src/minimizer/defs/swift-build.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/systemctl-status.toml b/crates/pi-shell/src/minimizer/defs/systemctl-status.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/systemctl-status.toml rename to crates/pi-shell/src/minimizer/defs/systemctl-status.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/systemctl.toml b/crates/pi-shell/src/minimizer/defs/systemctl.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/systemctl.toml rename to crates/pi-shell/src/minimizer/defs/systemctl.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/task.toml b/crates/pi-shell/src/minimizer/defs/task.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/task.toml rename to crates/pi-shell/src/minimizer/defs/task.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/terraform-plan.toml b/crates/pi-shell/src/minimizer/defs/terraform-plan.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/terraform-plan.toml rename to crates/pi-shell/src/minimizer/defs/terraform-plan.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/terraform.toml b/crates/pi-shell/src/minimizer/defs/terraform.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/terraform.toml rename to crates/pi-shell/src/minimizer/defs/terraform.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/tofu-fmt.toml b/crates/pi-shell/src/minimizer/defs/tofu-fmt.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/tofu-fmt.toml rename to crates/pi-shell/src/minimizer/defs/tofu-fmt.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/tofu-init.toml b/crates/pi-shell/src/minimizer/defs/tofu-init.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/tofu-init.toml rename to crates/pi-shell/src/minimizer/defs/tofu-init.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/tofu-plan.toml b/crates/pi-shell/src/minimizer/defs/tofu-plan.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/tofu-plan.toml rename to crates/pi-shell/src/minimizer/defs/tofu-plan.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/tofu-validate.toml b/crates/pi-shell/src/minimizer/defs/tofu-validate.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/tofu-validate.toml rename to crates/pi-shell/src/minimizer/defs/tofu-validate.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/trunk-build.toml b/crates/pi-shell/src/minimizer/defs/trunk-build.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/trunk-build.toml rename to crates/pi-shell/src/minimizer/defs/trunk-build.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/trunk.toml b/crates/pi-shell/src/minimizer/defs/trunk.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/trunk.toml rename to crates/pi-shell/src/minimizer/defs/trunk.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/turbo.toml b/crates/pi-shell/src/minimizer/defs/turbo.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/turbo.toml rename to crates/pi-shell/src/minimizer/defs/turbo.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/ty.toml b/crates/pi-shell/src/minimizer/defs/ty.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/ty.toml rename to crates/pi-shell/src/minimizer/defs/ty.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/uv-sync.toml b/crates/pi-shell/src/minimizer/defs/uv-sync.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/uv-sync.toml rename to crates/pi-shell/src/minimizer/defs/uv-sync.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/xcodebuild.toml b/crates/pi-shell/src/minimizer/defs/xcodebuild.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/xcodebuild.toml rename to crates/pi-shell/src/minimizer/defs/xcodebuild.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/yadm.toml b/crates/pi-shell/src/minimizer/defs/yadm.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/yadm.toml rename to crates/pi-shell/src/minimizer/defs/yadm.toml diff --git a/crates/pi-natives/src/shell/minimizer/defs/yamllint.toml b/crates/pi-shell/src/minimizer/defs/yamllint.toml similarity index 100% rename from crates/pi-natives/src/shell/minimizer/defs/yamllint.toml rename to crates/pi-shell/src/minimizer/defs/yamllint.toml diff --git a/crates/pi-natives/src/shell/minimizer/detect.rs b/crates/pi-shell/src/minimizer/detect.rs similarity index 100% rename from crates/pi-natives/src/shell/minimizer/detect.rs rename to crates/pi-shell/src/minimizer/detect.rs diff --git a/crates/pi-natives/src/shell/minimizer/engine.rs b/crates/pi-shell/src/minimizer/engine.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/engine.rs rename to crates/pi-shell/src/minimizer/engine.rs index 49ae98195..0f5bd7f4d 100644 --- a/crates/pi-natives/src/shell/minimizer/engine.rs +++ b/crates/pi-shell/src/minimizer/engine.rs @@ -8,7 +8,7 @@ use std::{ }, }; -use crate::shell::minimizer::{ +use crate::minimizer::{ MinimizerConfig, MinimizerCtx, MinimizerOutput, detect, filters, pipeline::{self, CompiledPipeline, PipelineRegistry}, plan, @@ -420,7 +420,7 @@ mod tests { #[cfg(test)] mod pipeline_integration_tests { use super::*; - use crate::shell::minimizer::MinimizerOptions; + use crate::minimizer::MinimizerOptions; #[test] fn builtin_filters_parse_and_pass_inline_tests() { diff --git a/crates/pi-natives/src/shell/minimizer/filters/bun.rs b/crates/pi-shell/src/minimizer/filters/bun.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/bun.rs rename to crates/pi-shell/src/minimizer/filters/bun.rs index 342627011..e13bc98c8 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/bun.rs +++ b/crates/pi-shell/src/minimizer/filters/bun.rs @@ -1,7 +1,7 @@ //! Bun package-manager, test-runner, and tool output filters. use super::{cpp, generic, js_tools, lint, node_tests, pkg}; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; const BUN_PACKAGE_SUBCOMMANDS: &[&str] = &[ "install", "i", "add", "update", "up", "upgrade", "remove", "rm", "outdated", "pm", "audit", @@ -142,7 +142,7 @@ fn is_important(line: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn ctx<'a>( program: &'a str, diff --git a/crates/pi-natives/src/shell/minimizer/filters/cargo.rs b/crates/pi-shell/src/minimizer/filters/cargo.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/cargo.rs rename to crates/pi-shell/src/minimizer/filters/cargo.rs index 812690bff..b8df0bb8f 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/cargo.rs +++ b/crates/pi-shell/src/minimizer/filters/cargo.rs @@ -1,6 +1,6 @@ //! Cargo build/test output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(subcommand: Option<&str>) -> bool { matches!( @@ -293,7 +293,7 @@ fn is_general_cargo_noise(line: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; #[test] fn strips_compiling_noise() { diff --git a/crates/pi-natives/src/shell/minimizer/filters/cloud.rs b/crates/pi-shell/src/minimizer/filters/cloud.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/filters/cloud.rs rename to crates/pi-shell/src/minimizer/filters/cloud.rs index f1c2413c8..69b4289d5 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/cloud.rs +++ b/crates/pi-shell/src/minimizer/filters/cloud.rs @@ -1,6 +1,6 @@ //! Cloud and data command output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; const MAX_PSQL_ROWS: usize = 30; const MAX_LINE_CHARS: usize = 500; @@ -401,7 +401,7 @@ fn join_lines(lines: Vec<String>) -> String { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn ctx<'a>(program: &'a str, cfg: &'a MinimizerConfig) -> MinimizerCtx<'a> { MinimizerCtx { program, subcommand: None, command: program, config: cfg } diff --git a/crates/pi-natives/src/shell/minimizer/filters/cpp.rs b/crates/pi-shell/src/minimizer/filters/cpp.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/cpp.rs rename to crates/pi-shell/src/minimizer/filters/cpp.rs index e7a9cdef4..a9c0131af 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/cpp.rs +++ b/crates/pi-shell/src/minimizer/filters/cpp.rs @@ -2,7 +2,7 @@ use std::path::Path; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; #[derive(Clone, Copy, Debug, PartialEq, Eq)] enum CppTool { @@ -249,7 +249,7 @@ fn push_line(out: &mut String, line: &str) { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn ctx<'a>( program: &'a str, diff --git a/crates/pi-natives/src/shell/minimizer/filters/docker.rs b/crates/pi-shell/src/minimizer/filters/docker.rs similarity index 97% rename from crates/pi-natives/src/shell/minimizer/filters/docker.rs rename to crates/pi-shell/src/minimizer/filters/docker.rs index d781746e3..7b6c50c4a 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/docker.rs +++ b/crates/pi-shell/src/minimizer/filters/docker.rs @@ -1,6 +1,6 @@ //! Container and cloud command output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(subcommand: Option<&str>) -> bool { matches!( @@ -168,7 +168,7 @@ fn head_tail_dedup(input: &str) -> String { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; #[test] fn dedups_repeated_log_lines_before_truncation() { diff --git a/crates/pi-natives/src/shell/minimizer/filters/dotnet.rs b/crates/pi-shell/src/minimizer/filters/dotnet.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/dotnet.rs rename to crates/pi-shell/src/minimizer/filters/dotnet.rs index 4ef9fecc7..b49c3db60 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/dotnet.rs +++ b/crates/pi-shell/src/minimizer/filters/dotnet.rs @@ -1,6 +1,6 @@ //! .NET CLI output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(program: &str, subcommand: Option<&str>) -> bool { program == "dotnet" && matches!(subcommand, Some("build" | "test" | "restore" | "format")) @@ -326,7 +326,7 @@ fn contains_diagnostic_signal(lower: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; #[test] fn keeps_dotnet_build_diagnostic_and_strips_restore_noise() { diff --git a/crates/pi-natives/src/shell/minimizer/filters/generic.rs b/crates/pi-shell/src/minimizer/filters/generic.rs similarity index 86% rename from crates/pi-natives/src/shell/minimizer/filters/generic.rs rename to crates/pi-shell/src/minimizer/filters/generic.rs index 1ad46959e..19ecd7c1e 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/generic.rs +++ b/crates/pi-shell/src/minimizer/filters/generic.rs @@ -1,6 +1,6 @@ //! Generic fallback transforms. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn filter(_ctx: &MinimizerCtx<'_>, input: &str, _exit_code: i32) -> MinimizerOutput { let stripped = primitives::strip_ansi(input); diff --git a/crates/pi-natives/src/shell/minimizer/filters/gh.rs b/crates/pi-shell/src/minimizer/filters/gh.rs similarity index 97% rename from crates/pi-natives/src/shell/minimizer/filters/gh.rs rename to crates/pi-shell/src/minimizer/filters/gh.rs index 5444b3f53..9a7fa19c2 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/gh.rs +++ b/crates/pi-shell/src/minimizer/filters/gh.rs @@ -1,6 +1,6 @@ //! GitHub CLI output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(subcommand: Option<&str>) -> bool { matches!( @@ -157,7 +157,7 @@ fn head_tail_dedup(input: &str) -> String { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn test_ctx<'a>( subcommand: Option<&'a str>, diff --git a/crates/pi-natives/src/shell/minimizer/filters/git.rs b/crates/pi-shell/src/minimizer/filters/git.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/filters/git.rs rename to crates/pi-shell/src/minimizer/filters/git.rs index 145d3bfb7..e09a1693c 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/git.rs +++ b/crates/pi-shell/src/minimizer/filters/git.rs @@ -1,6 +1,6 @@ //! Git output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(subcommand: Option<&str>) -> bool { matches!( @@ -370,7 +370,7 @@ fn condense_noisy_output(input: &str) -> String { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn test_ctx<'a>( subcommand: Option<&'a str>, diff --git a/crates/pi-natives/src/shell/minimizer/filters/go.rs b/crates/pi-shell/src/minimizer/filters/go.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/filters/go.rs rename to crates/pi-shell/src/minimizer/filters/go.rs index 57a3bd35b..2c4913f29 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/go.rs +++ b/crates/pi-shell/src/minimizer/filters/go.rs @@ -1,6 +1,6 @@ //! Go toolchain output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(program: &str, subcommand: Option<&str>) -> bool { match program { @@ -326,7 +326,7 @@ fn is_golangci_noise(line: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; #[test] fn keeps_go_test_failure_from_json_lines() { diff --git a/crates/pi-natives/src/shell/minimizer/filters/gt.rs b/crates/pi-shell/src/minimizer/filters/gt.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/gt.rs rename to crates/pi-shell/src/minimizer/filters/gt.rs index b8a85a74c..76b1f4da3 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/gt.rs +++ b/crates/pi-shell/src/minimizer/filters/gt.rs @@ -1,7 +1,7 @@ //! Graphite (`gt`) output filters. use super::git; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; const GT_SUBCOMMANDS: &[&str] = &[ "log", "submit", "sync", "restack", "create", "branch", "diff", "show", "add", "push", "pull", @@ -177,7 +177,7 @@ fn is_low_value_status(line: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn test_ctx<'a>(subcommand: Option<&'a str>, config: &'a MinimizerConfig) -> MinimizerCtx<'a> { test_ctx_with_command(subcommand, "gt", config) diff --git a/crates/pi-natives/src/shell/minimizer/filters/js_tools.rs b/crates/pi-shell/src/minimizer/filters/js_tools.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/filters/js_tools.rs rename to crates/pi-shell/src/minimizer/filters/js_tools.rs index 97c1b4c1a..9c7d56be5 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/js_tools.rs +++ b/crates/pi-shell/src/minimizer/filters/js_tools.rs @@ -3,7 +3,7 @@ //! Covers command output that is not already handled by the package-manager, //! test-runner, or lint filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; const SUPPORTED_TOOLS: &[&str] = &["next", "prettier", "prisma"]; const NPX_ROUTABLE_TOOLS: &[&str] = &["tsc", "eslint", "prisma", "prettier", "next"]; @@ -402,7 +402,7 @@ fn push_line(out: &mut String, line: &str) { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn ctx<'a>( program: &'a str, diff --git a/crates/pi-natives/src/shell/minimizer/filters/lint.rs b/crates/pi-shell/src/minimizer/filters/lint.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/lint.rs rename to crates/pi-shell/src/minimizer/filters/lint.rs index d9111f306..10014ee18 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/lint.rs +++ b/crates/pi-shell/src/minimizer/filters/lint.rs @@ -2,7 +2,7 @@ use std::collections::BTreeMap; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(subcommand: Option<&str>) -> bool { supports_program("", subcommand) diff --git a/crates/pi-natives/src/shell/minimizer/filters/listing.rs b/crates/pi-shell/src/minimizer/filters/listing.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/filters/listing.rs rename to crates/pi-shell/src/minimizer/filters/listing.rs index 226db4283..c98277c96 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/listing.rs +++ b/crates/pi-shell/src/minimizer/filters/listing.rs @@ -2,7 +2,7 @@ use std::{collections::BTreeMap, path::Path}; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn filter(ctx: &MinimizerCtx<'_>, input: &str, exit_code: i32) -> MinimizerOutput { let cleaned = primitives::strip_ansi(input); @@ -724,7 +724,7 @@ fn has_content(text: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn ctx<'a>(program: &'a str, cfg: &'a MinimizerConfig) -> MinimizerCtx<'a> { MinimizerCtx { program, subcommand: None, command: program, config: cfg } diff --git a/crates/pi-natives/src/shell/minimizer/filters/mod.rs b/crates/pi-shell/src/minimizer/filters/mod.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/mod.rs rename to crates/pi-shell/src/minimizer/filters/mod.rs index 0e0d5eb6d..715ae5e1c 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/mod.rs +++ b/crates/pi-shell/src/minimizer/filters/mod.rs @@ -1,6 +1,6 @@ //! Filter dispatch table for built-in minimizer strategies. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput}; pub mod cloud; pub mod cpp; @@ -133,7 +133,7 @@ fn wrapper_invokes(ctx: &MinimizerCtx<'_>, tools: &[&str]) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn ctx<'a>( program: &'a str, diff --git a/crates/pi-natives/src/shell/minimizer/filters/node_tests.rs b/crates/pi-shell/src/minimizer/filters/node_tests.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/node_tests.rs rename to crates/pi-shell/src/minimizer/filters/node_tests.rs index 3bdaf6212..72c294f38 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/node_tests.rs +++ b/crates/pi-shell/src/minimizer/filters/node_tests.rs @@ -1,6 +1,6 @@ //! Jest, Vitest, and Playwright output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn filter(_ctx: &MinimizerCtx<'_>, input: &str, exit_code: i32) -> MinimizerOutput { let cleaned = primitives::strip_ansi(input); diff --git a/crates/pi-natives/src/shell/minimizer/filters/pkg.rs b/crates/pi-shell/src/minimizer/filters/pkg.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/pkg.rs rename to crates/pi-shell/src/minimizer/filters/pkg.rs index 173355b76..505589f5a 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/pkg.rs +++ b/crates/pi-shell/src/minimizer/filters/pkg.rs @@ -1,6 +1,6 @@ //! Package manager output filters. -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(subcommand: Option<&str>) -> bool { matches!( diff --git a/crates/pi-natives/src/shell/minimizer/filters/python.rs b/crates/pi-shell/src/minimizer/filters/python.rs similarity index 98% rename from crates/pi-natives/src/shell/minimizer/filters/python.rs rename to crates/pi-shell/src/minimizer/filters/python.rs index b8ab0486a..2d16fef6a 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/python.rs +++ b/crates/pi-shell/src/minimizer/filters/python.rs @@ -1,7 +1,7 @@ //! Python test, type-check, and lint output filters. use super::lint; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(program: &str, subcommand: Option<&str>) -> bool { matches!(program, "pytest" | "ruff" | "mypy") @@ -243,7 +243,7 @@ fn has_content(text: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; #[test] fn supports_direct_and_python_module_tools() { diff --git a/crates/pi-natives/src/shell/minimizer/filters/ruby.rs b/crates/pi-shell/src/minimizer/filters/ruby.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/filters/ruby.rs rename to crates/pi-shell/src/minimizer/filters/ruby.rs index 5a0682a20..eb921e448 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/ruby.rs +++ b/crates/pi-shell/src/minimizer/filters/ruby.rs @@ -1,7 +1,7 @@ //! Ruby test and lint output filters. use super::lint; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(program: &str, subcommand: Option<&str>) -> bool { matches!(program, "rspec" | "rubocop") @@ -334,7 +334,7 @@ fn has_content(text: &str) -> bool { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; #[test] fn supports_rspec_minitest_and_rubocop() { diff --git a/crates/pi-natives/src/shell/minimizer/filters/system.rs b/crates/pi-shell/src/minimizer/filters/system.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/filters/system.rs rename to crates/pi-shell/src/minimizer/filters/system.rs index f43038e17..76a0bdfc8 100644 --- a/crates/pi-natives/src/shell/minimizer/filters/system.rs +++ b/crates/pi-shell/src/minimizer/filters/system.rs @@ -2,7 +2,7 @@ use std::collections::HashMap; -use crate::shell::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; +use crate::minimizer::{MinimizerCtx, MinimizerOutput, primitives}; pub fn supports(program: &str) -> bool { matches!( @@ -596,7 +596,7 @@ fn compact_sops_output(input: &str) -> String { #[cfg(test)] mod tests { use super::*; - use crate::shell::minimizer::MinimizerConfig; + use crate::minimizer::MinimizerConfig; fn ctx<'a>(program: &'a str, cfg: &'a MinimizerConfig) -> MinimizerCtx<'a> { MinimizerCtx { program, subcommand: None, command: program, config: cfg } diff --git a/crates/pi-natives/src/shell/minimizer/pipeline.rs b/crates/pi-shell/src/minimizer/pipeline.rs similarity index 99% rename from crates/pi-natives/src/shell/minimizer/pipeline.rs rename to crates/pi-shell/src/minimizer/pipeline.rs index 87d8f236d..7e1cf7fbd 100644 --- a/crates/pi-natives/src/shell/minimizer/pipeline.rs +++ b/crates/pi-shell/src/minimizer/pipeline.rs @@ -25,7 +25,7 @@ use std::borrow::Cow; use regex::{Regex, RegexSet}; use serde::Deserialize; -use crate::shell::minimizer::primitives; +use crate::minimizer::primitives; /// Raw TOML shape for a single filter definition. #[derive(Debug, Deserialize, Default)] diff --git a/crates/pi-natives/src/shell/minimizer/plan.rs b/crates/pi-shell/src/minimizer/plan.rs similarity index 97% rename from crates/pi-natives/src/shell/minimizer/plan.rs rename to crates/pi-shell/src/minimizer/plan.rs index 50e73fe64..edd2f0d6f 100644 --- a/crates/pi-natives/src/shell/minimizer/plan.rs +++ b/crates/pi-shell/src/minimizer/plan.rs @@ -32,7 +32,7 @@ pub enum CommandPlan { /// arguments), verbatim from the parsed AST. Single { program: String }, /// The command contains at least one `|` pipeline. We intentionally do - /// NOT identify upstream / downstream programs here \u2014 any pipe defeats + /// NOT identify upstream / downstream programs here — any pipe defeats /// safe minimization for this engine. Piped, /// The command has multiple segments joined by `&&`, `||`, `;`, or `&`. @@ -84,7 +84,7 @@ fn classify(program: &Program) -> CommandPlan { let CompoundListItem(and_or, separator) = items[0]; // Async separator (`&`) backgrounds the command; treat as compound since - // the parent shell's stdout is the foreground command's \u2014 we don't know + // the parent shell's stdout is the foreground command's — we don't know // which one we're capturing. Conservative bail. if matches!(separator, SeparatorOperator::Async) { return CommandPlan::Compound; diff --git a/crates/pi-natives/src/shell/minimizer/primitives.rs b/crates/pi-shell/src/minimizer/primitives.rs similarity index 100% rename from crates/pi-natives/src/shell/minimizer/primitives.rs rename to crates/pi-shell/src/minimizer/primitives.rs diff --git a/crates/pi-shell/src/process.rs b/crates/pi-shell/src/process.rs new file mode 100644 index 000000000..8899d8134 --- /dev/null +++ b/crates/pi-shell/src/process.rs @@ -0,0 +1,1801 @@ +//! Cross-platform process tree management. + +use std::{collections::HashSet, time::Duration}; + +use anyhow::Result; + +use crate::cancel::CancelToken; + +/// Current state of a process reference. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ProcessStatus { + /// The referenced process is still running. + Running, + /// The referenced process has exited or is no longer observable. + Exited, +} + +#[cfg(target_os = "linux")] +mod platform { + use std::{ + collections::HashSet, + ffi::OsStr, + fs, + os::fd::{AsRawFd, FromRawFd, OwnedFd, RawFd}, + ptr, + sync::Arc, + }; + + use super::ProcessStatus; + + /// Stable Linux process reference backed by a pidfd. + #[derive(Clone)] + pub struct Process { + pid: i32, + pidfd: Arc<OwnedFd>, + start_time: u64, + } + + impl Process { + pub fn from_pid(pid: i32) -> Option<Self> { + if pid <= 0 { + return None; + } + let pidfd = open_pidfd(pid)?; + let start_time = read_start_time(pid)?; + Some(Self { pid, pidfd, start_time }) + } + + pub const fn pid(&self) -> i32 { + self.pid + } + + pub fn children(&self) -> Vec<Self> { + if !self.live_identity() { + return Vec::new(); + } + + // `/proc/{pid}/task/{tid}/children` is per-task: a child fork()ed from a + // worker thread appears under that thread's `tid`, not the tgid. Walk + // every task subdir and union the lists, then re-validate parentage. + let task_dir = format!("/proc/{}/task", self.pid); + let Ok(entries) = fs::read_dir(&task_dir) else { + return Vec::new(); + }; + + let mut seen: HashSet<i32> = HashSet::new(); + let mut out = Vec::new(); + for entry in entries.flatten() { + let name = entry.file_name(); + let Some(tid_str) = name.to_str() else { + continue; + }; + if tid_str.parse::<i32>().is_err() { + continue; + } + let children_path = format!("/proc/{}/task/{}/children", self.pid, tid_str); + let Ok(content) = fs::read_to_string(&children_path) else { + continue; + }; + for part in content.split_whitespace() { + let Ok(child_pid) = part.parse::<i32>() else { + continue; + }; + if !seen.insert(child_pid) { + continue; + } + let Some(child) = Self::from_pid(child_pid) else { + continue; + }; + if child.status() == ProcessStatus::Running + && current_parent_pid(child.pid) == Some(self.pid) + { + out.push(child); + } + } + } + out + } + + pub fn parent_pid(&self) -> Option<i32> { + if self.status() == ProcessStatus::Running { + current_parent_pid(self.pid) + } else { + None + } + } + + pub fn args(&self) -> Vec<String> { + if !self.live_identity() { + return Vec::new(); + } + + let cmdline_path = format!("/proc/{}/cmdline", self.pid); + let Ok(content) = fs::read(cmdline_path) else { + return Vec::new(); + }; + // Re-validate after the read: PID reuse between identity check and read + // would otherwise leak an impostor's command line to callers. + if !self.live_identity() { + return Vec::new(); + } + split_nul_arguments(&content) + } + + pub fn kill(&self, signal: i32) -> bool { + // SAFETY: `self.pidfd` is an owned file descriptor returned by a successful + // `pidfd_open` call and remains open for the duration of this syscall. A null + // `siginfo_t` pointer is explicitly accepted by `pidfd_send_signal` and makes + // the kernel synthesize the same signal metadata as `kill(2)`. Flags are zero, + // which is the documented default behavior. + let ret = unsafe { + libc::syscall( + libc::SYS_pidfd_send_signal, + self.pidfd.as_raw_fd(), + signal, + ptr::null::<libc::siginfo_t>(), + 0, + ) + }; + ret == 0 + } + + pub fn group_id(&self) -> Option<i32> { + if self.status() != ProcessStatus::Running { + return None; + } + + // SAFETY: `self.pid` names the process currently referenced by `self.pidfd` + // unless it exits concurrently. If it exits, `getpgid` reports failure rather + // than dereferencing caller-owned memory. + let pgid = unsafe { libc::getpgid(self.pid) }; + if pgid > 0 { Some(pgid) } else { None } + } + + pub fn status(&self) -> ProcessStatus { + loop { + let mut pollfd = + libc::pollfd { fd: self.pidfd.as_raw_fd(), events: libc::POLLIN, revents: 0 }; + // SAFETY: `pollfd` points to one initialized `pollfd` element, and the pidfd + // remains open for the duration of the call. Timeout zero makes this a + // non-blocking readiness probe. + let ready = unsafe { libc::poll(&raw mut pollfd, 1, 0) }; + if ready < 0 { + // Retry on EINTR; for any other transient poll error treat the pidfd as + // still running. The pidfd is still owned and the kernel has not reported + // the process gone — a spurious `Exited` here makes every downstream + // signal/kill fall through silently. + if std::io::Error::last_os_error().raw_os_error() == Some(libc::EINTR) { + continue; + } + return ProcessStatus::Running; + } + if ready == 0 { + return ProcessStatus::Running; + } + if (pollfd.revents & (libc::POLLIN | libc::POLLHUP | libc::POLLERR | libc::POLLNVAL)) + != 0 + { + return ProcessStatus::Exited; + } + return ProcessStatus::Running; + } + } + + /// Walk the descendant tree in post-order (leaves first), de-duplicating + /// by PID so concurrent reparenting cannot trap us in a cycle. + pub fn descendants(&self) -> Vec<Self> { + let mut out = Vec::new(); + let mut visited = HashSet::new(); + visited.insert(self.pid); + self.descendants_into(&mut out, &mut visited); + out + } + + fn descendants_into(&self, out: &mut Vec<Self>, visited: &mut HashSet<i32>) { + for child in self.children() { + if visited.insert(child.pid) { + child.descendants_into(out, visited); + out.push(child); + } + } + } + + fn live_identity(&self) -> bool { + self.status() == ProcessStatus::Running + && read_start_time(self.pid) == Some(self.start_time) + } + } + + fn split_nul_arguments(content: &[u8]) -> Vec<String> { + content + .split(|byte| *byte == 0) + .filter(|part| !part.is_empty()) + .map(|part| String::from_utf8_lossy(part).into_owned()) + .collect() + } + + fn current_parent_pid(pid: i32) -> Option<i32> { + let status_path = format!("/proc/{pid}/status"); + let content = fs::read_to_string(status_path).ok()?; + content.lines().find_map(|line| { + line + .strip_prefix("PPid:") + .and_then(|ppid| ppid.trim().parse::<i32>().ok()) + }) + } + + fn read_start_time(pid: i32) -> Option<u64> { + // `/proc/[pid]/stat` field 22 is the process start time in clock ticks since + // boot. The comm field (between parens) may itself contain spaces and parens, + // so locate the *last* `)` and split the trailing whitespace-separated fields. + let stat_path = format!("/proc/{pid}/stat"); + let content = fs::read_to_string(stat_path).ok()?; + let last_paren = content.rfind(')')?; + let rest = &content[last_paren + 1..]; + rest.split_whitespace().nth(19)?.parse().ok() + } + + fn open_pidfd(pid: i32) -> Option<Arc<OwnedFd>> { + // SAFETY: `pidfd_open` takes the PID by value and does not read caller-owned + // memory. Flags are zero, which is valid. On success the returned descriptor is + // newly owned by this process and is immediately wrapped in `OwnedFd` below. + let fd = unsafe { libc::syscall(libc::SYS_pidfd_open, pid, 0) }; + if fd < 0 { + return None; + } + + // SAFETY: `fd` is non-negative and was just returned by `pidfd_open`, so it is + // an open descriptor owned by this process. `OwnedFd` takes sole ownership and + // will close it exactly once. + Some(Arc::new(unsafe { OwnedFd::from_raw_fd(fd as RawFd) })) + } + + /// Send `signal` to the process group `pgid`. + /// Returns true when the signal is delivered successfully. + pub fn kill_process_group(pgid: i32, signal: i32) -> bool { + // SAFETY: `kill` takes integer identifiers by value and does not access + // caller-owned memory. A negative PID is the POSIX process-group form. + unsafe { libc::kill(-pgid, signal) == 0 } + } + + /// Find processes whose `/proc/{pid}/exe` symlink resolves to exactly + /// `target`. + pub fn find_by_path(target: &str) -> Vec<Process> { + let mut matches = Vec::new(); + let Ok(entries) = fs::read_dir("/proc") else { + return matches; + }; + let target_os = OsStr::new(target); + for entry in entries.flatten() { + let name = entry.file_name(); + let Some(name_str) = name.to_str() else { + continue; + }; + let Ok(pid) = name_str.parse::<i32>() else { + continue; + }; + let exe_path = format!("/proc/{pid}/exe"); + let Ok(resolved) = fs::read_link(&exe_path) else { + continue; + }; + if resolved.as_os_str() == target_os + && let Some(process) = Process::from_pid(pid) + { + matches.push(process); + } + } + matches + } +} + +#[cfg(target_os = "macos")] +mod platform { + use std::{ + collections::{HashMap, HashSet}, + ptr, + }; + + use super::ProcessStatus; + + #[link(name = "proc", kind = "dylib")] + unsafe extern "C" { + fn proc_listallpids(buffer: *mut i32, buffersize: i32) -> i32; + fn proc_pidpath(pid: i32, buffer: *mut std::ffi::c_void, buffersize: u32) -> i32; + } + + /// macOS does not expose pidfds; identity is pinned via the kernel-reported + /// process start time so a recycled PID does not silently impersonate the + /// original target. + #[derive(Clone)] + pub struct Process { + pid: i32, + start_tvsec: u64, + start_tvusec: u64, + } + + impl Process { + pub fn from_pid(pid: i32) -> Option<Self> { + if pid <= 0 { + return None; + } + let info = read_bsdinfo(pid)?; + if i32::try_from(info.pbi_pid).ok()? != pid { + return None; + } + Some(Self { pid, start_tvsec: info.pbi_start_tvsec, start_tvusec: info.pbi_start_tvusec }) + } + + pub const fn pid(&self) -> i32 { + self.pid + } + + pub fn children(&self) -> Vec<Self> { + if self.live_bsdinfo().is_none() { + return Vec::new(); + } + // `proc_listchildpids` (the obvious choice) is broken on recent macOS + // kernels when queried for the *calling* process — it returns one byte of + // padding regardless of how many children the process actually has, so a + // process can never list its own descendants. Confirmed on darwin 25.4 + // from C, Rust, and Bun callers via `proc_listchildpids(getpid(), …)`, + // while `ps -P` and `pgrep -P` still see the same children. Walk the + // whole pid table via `proc_listallpids` and filter on `pbi_ppid` + // instead; this is the same approach we already use for `find_by_path` + // and that the Windows implementation uses via Toolhelp snapshots. + let tree = build_process_tree(); + Self::children_from_tree(self.pid, &tree) + } + + pub fn parent_pid(&self) -> Option<i32> { + let info = self.live_bsdinfo()?; + i32::try_from(info.pbi_ppid).ok().filter(|ppid| *ppid > 0) + } + + pub fn args(&self) -> Vec<String> { + if self.live_bsdinfo().is_none() { + return Vec::new(); + } + process_args(self.pid) + } + + pub fn kill(&self, signal: i32) -> bool { + // Re-validate identity right before signaling. There is no atomic + // "kill iff start_time matches" primitive on macOS, so a vanishingly small + // window remains between this check and the syscall — but matching against + // the recorded `(pid, start_tvsec, start_tvusec)` triple eliminates the + // PID-reuse race in every practical case. + if self.live_bsdinfo().is_none() { + return false; + } + // SAFETY: `kill` takes integer identifiers by value and does not access + // caller-owned memory. + unsafe { libc::kill(self.pid, signal) == 0 } + } + + pub fn group_id(&self) -> Option<i32> { + let info = self.live_bsdinfo()?; + i32::try_from(info.pbi_pgid).ok().filter(|pgid| *pgid > 0) + } + + /// Walk the descendant tree in post-order (leaves first), de-duplicating + /// by PID so concurrent reparenting cannot trap us in a cycle. + pub fn descendants(&self) -> Vec<Self> { + // One process-table snapshot per walk — building it inside the recursion + // would re-scan every pid for every visited node, producing an `O(N · D)` + // kernel call pattern. Mirrors the Windows implementation. + let tree = build_process_tree(); + let mut out = Vec::new(); + let mut visited = HashSet::new(); + visited.insert(self.pid); + Self::collect_descendants_from_tree(self.pid, &tree, &mut visited, &mut out); + out + } + + fn children_from_tree(parent: i32, tree: &HashMap<i32, Vec<i32>>) -> Vec<Self> { + let Some(child_pids) = tree.get(&parent) else { + return Vec::new(); + }; + child_pids + .iter() + .copied() + .filter_map(Self::from_pid) + .collect() + } + + fn collect_descendants_from_tree( + parent: i32, + tree: &HashMap<i32, Vec<i32>>, + visited: &mut HashSet<i32>, + out: &mut Vec<Self>, + ) { + let Some(child_pids) = tree.get(&parent) else { + return; + }; + for &child_pid in child_pids { + if !visited.insert(child_pid) { + continue; + } + let Some(child) = Self::from_pid(child_pid) else { + continue; + }; + // Post-order: grandchildren first, so leaf processes get signalled + // before their parents during tree termination. + Self::collect_descendants_from_tree(child_pid, tree, visited, out); + out.push(child); + } + } + + pub fn status(&self) -> ProcessStatus { + if self.live_bsdinfo().is_some() { + ProcessStatus::Running + } else { + ProcessStatus::Exited + } + } + + /// Returns the current `proc_bsdinfo` only if it still describes the same + /// process this reference was opened on — i.e. the start time has not + /// changed. + fn live_bsdinfo(&self) -> Option<libc::proc_bsdinfo> { + let info = read_bsdinfo(self.pid)?; + if info.pbi_start_tvsec == self.start_tvsec && info.pbi_start_tvusec == self.start_tvusec { + Some(info) + } else { + None + } + } + } + + /// Send `signal` to the process group `pgid`. + /// Returns true when the signal is delivered successfully. + pub fn kill_process_group(pgid: i32, signal: i32) -> bool { + // SAFETY: `kill` takes integer identifiers by value and does not access + // caller-owned memory. A negative PID is the POSIX process-group form. + unsafe { libc::kill(-pgid, signal) == 0 } + } + + const KERN_PROCARGS2: libc::c_int = 49; + + const PROC_PIDPATHINFO_MAXSIZE: usize = 4096; + + /// Snapshot every pid currently visible to `proc_listallpids`. macOS + /// silently truncates the second call to the supplied buffer size even + /// when the sizing query reports more bytes available, so the buffer is + /// padded well beyond the reported count. + fn snapshot_all_pids() -> Vec<i32> { + // SAFETY: Passing a null buffer with size 0 is the documented libproc query + // form for obtaining the byte count needed for all PIDs; libproc does not + // dereference the null pointer in this mode. + let bytes = unsafe { proc_listallpids(ptr::null_mut(), 0) }; + if bytes <= 0 { + return Vec::new(); + } + let count = (bytes as usize) / size_of::<i32>(); + let cap = count.saturating_mul(4).max(2048); + let mut buffer = vec![0i32; cap]; + // SAFETY: `buffer` is valid for `buffer.len() * size_of::<i32>()` bytes and + // is properly aligned for `i32`; libproc writes at most the supplied size. + let actual = + unsafe { proc_listallpids(buffer.as_mut_ptr(), (buffer.len() * size_of::<i32>()) as i32) }; + if actual <= 0 { + return Vec::new(); + } + let pid_count = ((actual as usize) / size_of::<i32>()).min(buffer.len()); + buffer.truncate(pid_count); + buffer + } + + /// Build a `ppid -> [pids]` map from a one-shot scan of `proc_listallpids`. + /// + /// Used as the foundation of `Process::children` and `Process::descendants` + /// on macOS where `proc_listchildpids` returns no children for self-queries. + pub(super) fn build_process_tree() -> HashMap<i32, Vec<i32>> { + let pids = snapshot_all_pids(); + let mut tree: HashMap<i32, Vec<i32>> = HashMap::with_capacity(pids.len() / 2); + for pid in pids { + if pid <= 0 { + continue; + } + let Some(info) = read_bsdinfo(pid) else { + continue; + }; + let Ok(ppid) = i32::try_from(info.pbi_ppid) else { + continue; + }; + if ppid <= 0 { + continue; + } + tree.entry(ppid).or_default().push(pid); + } + tree + } + + /// Find processes whose libproc-reported executable path equals `target`. + pub fn find_by_path(target: &str) -> Vec<Process> { + let pids = snapshot_all_pids(); + let mut path_buf = vec![0u8; PROC_PIDPATHINFO_MAXSIZE]; + let mut matches = Vec::new(); + for pid in pids { + if pid <= 0 { + continue; + } + // SAFETY: `path_buf` is valid for `path_buf.len()` bytes; libproc writes a + // NUL-terminated path no longer than the supplied capacity and returns the + // number of bytes written. + let len = unsafe { + proc_pidpath( + pid, + path_buf.as_mut_ptr().cast::<std::ffi::c_void>(), + path_buf.len() as u32, + ) + }; + if len <= 0 { + continue; + } + let path_bytes = &path_buf[..len as usize]; + let path_bytes = match path_bytes.iter().position(|byte| *byte == 0) { + Some(end) => &path_bytes[..end], + None => path_bytes, + }; + let Ok(path) = std::str::from_utf8(path_bytes) else { + continue; + }; + if path == target + && let Some(process) = Process::from_pid(pid) + { + matches.push(process); + } + } + matches + } + + fn read_bsdinfo(pid: i32) -> Option<libc::proc_bsdinfo> { + // SAFETY: `proc_bsdinfo` is a plain C data struct. Zero initialization is + // valid because every field is an integer or fixed-size integer array, and + // libproc fully overwrites the fields it reports on a successful call. + let mut info = unsafe { std::mem::zeroed::<libc::proc_bsdinfo>() }; + // SAFETY: `info` is a writable `proc_bsdinfo` buffer whose exact byte size is + // supplied to libproc. The PID, flavor, and arg are scalar values passed by + // value; libproc writes at most the supplied buffer size. + let actual = unsafe { + libc::proc_pidinfo( + pid, + libc::PROC_PIDTBSDINFO, + 0, + (&raw mut info).cast::<std::ffi::c_void>(), + size_of::<libc::proc_bsdinfo>() as i32, + ) + }; + if actual < size_of::<libc::proc_bsdinfo>() as i32 { + return None; + } + Some(info) + } + + fn process_args(pid: i32) -> Vec<String> { + let mut mib = [libc::CTL_KERN, KERN_PROCARGS2, pid]; + let mut size = 0usize; + // SAFETY: `mib` points to three initialized integers and the old-value buffer + // is null with a zero-length query, which is the documented `sysctl` sizing + // pattern. `size` is a valid out-parameter for the required byte count. + let sizing_ok = unsafe { + libc::sysctl( + mib.as_mut_ptr(), + mib.len() as u32, + ptr::null_mut(), + &raw mut size, + ptr::null_mut(), + 0, + ) + } == 0; + if !sizing_ok || size <= size_of::<libc::c_int>() { + return Vec::new(); + } + + let mut buffer = vec![0u8; size]; + // SAFETY: `mib` still points to three initialized integers. `buffer` is + // writable for `size` bytes, and `size` is provided as the in/out byte count. + let read_ok = unsafe { + libc::sysctl( + mib.as_mut_ptr(), + mib.len() as u32, + buffer.as_mut_ptr().cast::<std::ffi::c_void>(), + &raw mut size, + ptr::null_mut(), + 0, + ) + } == 0; + if !read_ok { + return Vec::new(); + } + buffer.truncate(size); + parse_macos_procargs(&buffer) + } + + fn parse_macos_procargs(buffer: &[u8]) -> Vec<String> { + // KERN_PROCARGS2 layout: `argc: i32 | exec_path: NUL-padded | argv[0..argc] | + // env[..]`. argc covers only argv, so we must skip the exec_path NUL padding + // and stop after exactly argc entries — otherwise environment variables leak + // into the arg list (each NUL-terminated env=value is indistinguishable from + // an arg). + let argc_size = size_of::<libc::c_int>(); + if buffer.len() <= argc_size { + return Vec::new(); + } + + let argc_bytes: [u8; 4] = match buffer[..argc_size].try_into() { + Ok(bytes) => bytes, + Err(_) => return Vec::new(), + }; + let argc = libc::c_int::from_ne_bytes(argc_bytes); + if argc <= 0 { + return Vec::new(); + } + + let mut offset = argc_size; + while offset < buffer.len() && buffer[offset] != 0 { + offset += 1; + } + while offset < buffer.len() && buffer[offset] == 0 { + offset += 1; + } + + let mut args = Vec::with_capacity(argc as usize); + while offset < buffer.len() && args.len() < argc as usize { + let end = buffer[offset..] + .iter() + .position(|byte| *byte == 0) + .map_or(buffer.len(), |position| offset + position); + if end == offset { + break; + } + args.push(String::from_utf8_lossy(&buffer[offset..end]).into_owned()); + offset = end + 1; + } + args + } +} +#[cfg(target_os = "windows")] +mod platform { + use std::{ + collections::{HashMap, HashSet}, + ffi::c_void, + mem, + sync::Arc, + }; + + use smallvec::SmallVec; + + use super::ProcessStatus; + + #[repr(C)] + #[allow(non_snake_case, reason = "Windows PROCESSENTRY32W field names must match Win32 ABI")] + struct PROCESSENTRY32W { + dwSize: u32, + cntUsage: u32, + th32ProcessID: u32, + th32DefaultHeapID: usize, + th32ModuleID: u32, + cntThreads: u32, + th32ParentProcessID: u32, + pcPriClassBase: i32, + dwFlags: u32, + szExeFile: [u16; 260], + } + + #[repr(C)] + struct ProcessBasicInformation { + exit_status: i32, + peb_base_address: usize, + affinity_mask: usize, + base_priority: i32, + unique_process_id: usize, + inherited_from_unique_process_id: usize, + } + + #[repr(C)] + #[derive(Clone, Copy)] + struct UnicodeString { + length: u16, + maximum_length: u16, + buffer: usize, + } + + #[repr(C)] + #[derive(Clone, Copy)] + struct PebPartial { + reserved1: [u8; 2], + being_debugged: u8, + reserved2: [u8; 1], + reserved3: [usize; 2], + loader: usize, + process_parameters: usize, + } + + #[repr(C)] + #[derive(Clone, Copy)] + struct UserProcessParametersPartial { + reserved1: [u8; 16], + reserved2: [usize; 10], + image_path_name: UnicodeString, + command_line: UnicodeString, + } + + #[repr(C)] + #[derive(Clone, Copy, Default)] + struct Filetime { + dw_low_date_time: u32, + dw_high_date_time: u32, + } + + type Handle = *mut c_void; + type NtStatus = i32; + const INVALID_HANDLE_VALUE: Handle = -1isize as Handle; + const PROCESS_QUERY_INFORMATION: u32 = 0x0400; + const PROCESS_VM_READ: u32 = 0x0010; + const PROCESS_BASIC_INFORMATION_CLASS: u32 = 0; + const STATUS_SUCCESS: NtStatus = 0; + const TH32CS_SNAPPROCESS: u32 = 0x00000002; + const PROCESS_TERMINATE: u32 = 0x0001; + const PROCESS_QUERY_LIMITED_INFORMATION: u32 = 0x1000; + const SYNCHRONIZE: u32 = 0x00100000; + const PROCESS_REFERENCE_ACCESS: u32 = + PROCESS_TERMINATE | PROCESS_QUERY_LIMITED_INFORMATION | SYNCHRONIZE; + const WAIT_OBJECT_0: u32 = 0; + + #[link(name = "kernel32")] + unsafe extern "system" { + fn CreateToolhelp32Snapshot(dwFlags: u32, th32ProcessID: u32) -> Handle; + fn Process32FirstW(hSnapshot: Handle, lppe: *mut PROCESSENTRY32W) -> i32; + fn Process32NextW(hSnapshot: Handle, lppe: *mut PROCESSENTRY32W) -> i32; + fn CloseHandle(hObject: Handle) -> i32; + fn OpenProcess(dwDesiredAccess: u32, bInheritHandle: i32, dwProcessId: u32) -> Handle; + fn TerminateProcess(hProcess: Handle, uExitCode: u32) -> i32; + fn QueryFullProcessImageNameW( + hProcess: Handle, + dwFlags: u32, + lpExeName: *mut u16, + lpdwSize: *mut u32, + ) -> i32; + fn WaitForSingleObject(hHandle: Handle, dwMilliseconds: u32) -> u32; + fn GetProcessTimes( + hProcess: Handle, + lpCreationTime: *mut Filetime, + lpExitTime: *mut Filetime, + lpKernelTime: *mut Filetime, + lpUserTime: *mut Filetime, + ) -> i32; + fn ReadProcessMemory( + hProcess: Handle, + lpBaseAddress: *const c_void, + lpBuffer: *mut c_void, + nSize: usize, + lpNumberOfBytesRead: *mut usize, + ) -> i32; + fn LocalFree(hMem: Handle) -> Handle; + } + + #[link(name = "shell32")] + unsafe extern "system" { + fn CommandLineToArgvW(lpCmdLine: *const u16, pNumArgs: *mut i32) -> *mut *mut u16; + } + + #[link(name = "ntdll")] + unsafe extern "system" { + fn NtQueryInformationProcess( + ProcessHandle: Handle, + ProcessInformationClass: u32, + ProcessInformation: *mut c_void, + ProcessInformationLength: u32, + ReturnLength: *mut u32, + ) -> NtStatus; + } + + struct OwnedHandle { + raw: isize, + } + + impl OwnedHandle { + fn from_raw(raw: Handle) -> Option<Self> { + if raw.is_null() || raw == INVALID_HANDLE_VALUE { + None + } else { + Some(Self { raw: raw as isize }) + } + } + + fn as_raw(&self) -> Handle { + self.raw as Handle + } + } + + impl Drop for OwnedHandle { + fn drop(&mut self) { + // SAFETY: `self.raw` was returned by a successful Win32 handle-producing + // function and stored only in this `OwnedHandle`. `Drop` runs once, so this + // closes the owned handle exactly once and no code uses it afterward. + let _ = unsafe { CloseHandle(self.as_raw()) }; + } + } + + #[derive(Clone)] + /// Stable Windows process reference backed by an owned process handle plus + /// the kernel-reported creation time, which pins identity even if the PID is + /// recycled while we hold the handle. + pub struct Process { + pid: i32, + handle: Arc<OwnedHandle>, + creation_time: u64, + } + + impl Process { + pub fn from_pid(pid: i32) -> Option<Self> { + if pid <= 0 { + return None; + } + let pid_u32 = u32::try_from(pid).ok()?; + let handle = open_process(pid_u32, PROCESS_REFERENCE_ACCESS)?; + let creation_time = process_creation_time(handle.as_raw())?; + Some(Self { pid, handle, creation_time }) + } + + pub const fn pid(&self) -> i32 { + self.pid + } + + pub fn parent_pid(&self) -> Option<i32> { + process_basic_information(self.handle.as_raw()) + .and_then(|info| i32::try_from(info.inherited_from_unique_process_id).ok()) + .filter(|pid| *pid > 0) + } + + pub fn args(&self) -> Vec<String> { + process_command_line(self) + .as_deref() + .map(split_windows_command_line) + .unwrap_or_default() + } + + pub fn children(&self) -> Vec<Self> { + let tree = build_process_tree(); + Self::children_from_tree(self.pid, &tree) + } + + /// Walk the entire descendant tree using a single Toolhelp snapshot. + /// + /// `children()` recursing per-node would re-snapshot the whole process + /// table for every visited descendant, making tree termination + /// `O(N · D)` snapshots. One snapshot per termination wave is enough. + pub fn descendants(&self) -> Vec<Self> { + let tree = build_process_tree(); + let Ok(root) = u32::try_from(self.pid) else { + return Vec::new(); + }; + let mut visited: HashSet<u32> = HashSet::new(); + visited.insert(root); + let mut out = Vec::new(); + Self::collect_descendants_from_tree(root, &tree, &mut visited, &mut out); + out + } + + fn children_from_tree(pid: i32, tree: &HashMap<u32, SmallVec<[u32; 4]>>) -> Vec<Self> { + let Ok(pid_u32) = u32::try_from(pid) else { + return Vec::new(); + }; + tree + .get(&pid_u32) + .into_iter() + .flatten() + .filter_map(|&child_pid| { + let child = Self::from_pid(i32::try_from(child_pid).ok()?)?; + (child.status() == ProcessStatus::Running).then_some(child) + }) + .collect() + } + + fn collect_descendants_from_tree( + parent: u32, + tree: &HashMap<u32, SmallVec<[u32; 4]>>, + visited: &mut HashSet<u32>, + out: &mut Vec<Self>, + ) { + let Some(children) = tree.get(&parent) else { + return; + }; + for &child_pid in children { + if !visited.insert(child_pid) { + continue; + } + let Ok(child_pid_i) = i32::try_from(child_pid) else { + continue; + }; + let Some(child) = Self::from_pid(child_pid_i) else { + continue; + }; + if child.status() != ProcessStatus::Running { + continue; + } + // Post-order: collect grandchildren first so leaves are signalled before + // their parents during tree termination. + Self::collect_descendants_from_tree(child_pid, tree, visited, out); + out.push(child); + } + } + + pub fn kill(&self, _signal: i32) -> bool { + // The handle pins the original kernel process object even after the PID is + // recycled, so `TerminateProcess` cannot accidentally hit a different + // process. SAFETY: `self.handle` is an owned process handle opened with + // `PROCESS_TERMINATE` access and remains valid for the duration of this + // call. The exit code is passed by value. + unsafe { TerminateProcess(self.handle.as_raw(), 1) != 0 } + } + + pub const fn group_id(&self) -> Option<i32> { + None + } + + pub fn status(&self) -> ProcessStatus { + // `WaitForSingleObject` on a process handle opened with `SYNCHRONIZE` is + // the definitive liveness probe: the handle becomes signalled iff the + // process has exited. This avoids the `STILL_ACTIVE == 259` pitfall in + // `GetExitCodeProcess`, where a process that legitimately exits with code + // 259 is indistinguishable from a still-running one. + // + // SAFETY: `self.handle` is an owned process handle opened with + // `SYNCHRONIZE` access. A zero timeout makes this a non-blocking probe. + let result = unsafe { WaitForSingleObject(self.handle.as_raw(), 0) }; + if result == WAIT_OBJECT_0 { + ProcessStatus::Exited + } else { + ProcessStatus::Running + } + } + } + + fn process_basic_information(handle: Handle) -> Option<ProcessBasicInformation> { + let mut info = ProcessBasicInformation { + exit_status: 0, + peb_base_address: 0, + affinity_mask: 0, + base_priority: 0, + unique_process_id: 0, + inherited_from_unique_process_id: 0, + }; + let mut returned = 0u32; + // SAFETY: `handle` is a valid process handle. `info` is writable for exactly + // `size_of::<ProcessBasicInformation>()` bytes, and `returned` is a valid + // optional out-parameter for the byte count. + let status = unsafe { + NtQueryInformationProcess( + handle, + PROCESS_BASIC_INFORMATION_CLASS, + (&raw mut info).cast::<c_void>(), + mem::size_of::<ProcessBasicInformation>() as u32, + &raw mut returned, + ) + }; + (status == STATUS_SUCCESS).then_some(info) + } + + fn process_command_line(process: &Process) -> Option<String> { + let pid_u32 = u32::try_from(process.pid).ok()?; + let read_handle = open_process(pid_u32, PROCESS_QUERY_INFORMATION | PROCESS_VM_READ)?; + // PID-reuse defense: `OpenProcess` resolves a PID to *whichever* process owns + // it right now, which need not be the one our original handle pinned. Compare + // the freshly opened handle's creation time against the recorded value to + // reject reads from an unrelated process that happens to share the PID. + if process_creation_time(read_handle.as_raw())? != process.creation_time { + return None; + } + let info = process_basic_information(read_handle.as_raw())?; + let peb: PebPartial = read_remote(read_handle.as_raw(), info.peb_base_address)?; + if peb.process_parameters == 0 { + return None; + } + let params: UserProcessParametersPartial = + read_remote(read_handle.as_raw(), peb.process_parameters)?; + read_remote_unicode_string(read_handle.as_raw(), params.command_line) + } + + fn process_creation_time(handle: Handle) -> Option<u64> { + let mut creation = Filetime::default(); + let mut exit = Filetime::default(); + let mut kernel = Filetime::default(); + let mut user = Filetime::default(); + // SAFETY: `handle` is a valid process handle opened with at least + // `PROCESS_QUERY_LIMITED_INFORMATION`. All four out-parameters point to + // initialized, writable `Filetime` values that live until the call returns. + let ok = unsafe { + GetProcessTimes(handle, &raw mut creation, &raw mut exit, &raw mut kernel, &raw mut user) + != 0 + }; + if !ok { + return None; + } + Some((u64::from(creation.dw_high_date_time) << 32) | u64::from(creation.dw_low_date_time)) + } + + fn read_remote<T: Copy>(handle: Handle, address: usize) -> Option<T> { + if address == 0 { + return None; + } + let mut value = mem::MaybeUninit::<T>::uninit(); + let mut bytes_read = 0usize; + // SAFETY: `handle` is opened with `PROCESS_VM_READ`. `address` comes from + // kernel-reported process structures for that same process. `value` points to + // uninitialized local storage large enough for `T`, and `bytes_read` is a valid + // out-parameter. The value is only assumed initialized after the OS reports a + // full-size successful read. + let ok = unsafe { + ReadProcessMemory( + handle, + address as *const c_void, + value.as_mut_ptr().cast::<c_void>(), + mem::size_of::<T>(), + &raw mut bytes_read, + ) != 0 + }; + if ok && bytes_read == mem::size_of::<T>() { + // SAFETY: The successful `ReadProcessMemory` call above initialized exactly + // `size_of::<T>()` bytes in `value`. + Some(unsafe { value.assume_init() }) + } else { + None + } + } + + fn read_remote_unicode_string(handle: Handle, value: UnicodeString) -> Option<String> { + if value.length == 0 || value.buffer == 0 || value.length % 2 != 0 { + return None; + } + let code_units = usize::from(value.length) / size_of::<u16>(); + let mut buffer = vec![0u16; code_units]; + let mut bytes_read = 0usize; + // SAFETY: `handle` is opened with `PROCESS_VM_READ`. `value.buffer` and + // `value.length` come from the remote process' own `UNICODE_STRING`. `buffer` + // is writable for exactly `value.length` bytes, and `bytes_read` is a valid + // out-parameter. The string is decoded only after a full successful read. + let ok = unsafe { + ReadProcessMemory( + handle, + value.buffer as *const c_void, + buffer.as_mut_ptr().cast::<c_void>(), + usize::from(value.length), + &raw mut bytes_read, + ) != 0 + }; + if ok && bytes_read == usize::from(value.length) { + Some(String::from_utf16_lossy(&buffer)) + } else { + None + } + } + + fn split_windows_command_line(command_line: &str) -> Vec<String> { + use std::os::windows::ffi::OsStringExt; + + let mut wide: Vec<u16> = command_line.encode_utf16().chain([0]).collect(); + let mut argc = 0i32; + // SAFETY: `wide` is a local, NUL-terminated UTF-16 buffer that remains alive + // for the duration of the call. `argc` is a valid out-parameter. The returned + // argv block is released with `LocalFree` below as required by + // `CommandLineToArgvW`. + let argv = unsafe { CommandLineToArgvW(wide.as_mut_ptr(), &raw mut argc) }; + if argv.is_null() || argc <= 0 { + return Vec::new(); + } + let argc = argc as usize; + // SAFETY: `CommandLineToArgvW` returned a non-null pointer to `argc` argument + // pointers, valid until freed with `LocalFree`. + let pointers = unsafe { std::slice::from_raw_parts(argv, argc) }; + let args = pointers + .iter() + .filter_map(|&arg| { + if arg.is_null() { + return None; + } + let mut len = 0usize; + // SAFETY: Each pointer in the argv block is a NUL-terminated UTF-16 + // string owned by the argv block and valid until `LocalFree` below. + while unsafe { *arg.add(len) } != 0 { + len += 1; + } + // SAFETY: The loop above found the terminating NUL, so the preceding + // `len` code units form a valid readable slice. + let slice = unsafe { std::slice::from_raw_parts(arg, len) }; + Some( + std::ffi::OsString::from_wide(slice) + .to_string_lossy() + .into_owned(), + ) + }) + .collect(); + // SAFETY: `argv` is the allocation returned by `CommandLineToArgvW` and has + // not been freed yet. No pointers into it are used after this call. + let _ = unsafe { LocalFree(argv.cast::<c_void>()) }; + args + } + + fn open_process(pid: u32, access: u32) -> Option<Arc<OwnedHandle>> { + // SAFETY: `OpenProcess` takes the PID and access mask by value and does not + // dereference caller-owned memory. Handle inheritance is disabled. Identity + // is established by the caller (typically `Process::from_pid`) capturing the + // creation time immediately after a successful open and re-checking it on + // every subsequent operation that re-resolves the PID. + let handle = unsafe { OpenProcess(access, 0, pid) }; + OwnedHandle::from_raw(handle).map(Arc::new) + } + + fn create_process_snapshot() -> Option<OwnedHandle> { + // SAFETY: The process snapshot API takes flags and a process ID by value and + // does not dereference caller-owned memory. PID zero requests all processes. + let snapshot = unsafe { CreateToolhelp32Snapshot(TH32CS_SNAPPROCESS, 0) }; + OwnedHandle::from_raw(snapshot) + } + + fn process_entry() -> PROCESSENTRY32W { + PROCESSENTRY32W { + dwSize: mem::size_of::<PROCESSENTRY32W>() as u32, + cntUsage: 0, + th32ProcessID: 0, + th32DefaultHeapID: 0, + th32ModuleID: 0, + cntThreads: 0, + th32ParentProcessID: 0, + pcPriClassBase: 0, + dwFlags: 0, + szExeFile: [0; 260], + } + } + + /// Build a map of `parent_pid` -> [`child_pids`] for all processes. + fn build_process_tree() -> HashMap<u32, SmallVec<[u32; 4]>> { + let mut tree: HashMap<u32, SmallVec<[u32; 4]>> = HashMap::new(); + let Some(snapshot) = create_process_snapshot() else { + return tree; + }; + + let mut entry = process_entry(); + // SAFETY: `snapshot` is a valid Toolhelp snapshot handle. `entry` points to a + // writable `PROCESSENTRY32W` whose `dwSize` field was initialized to the exact + // ABI size before the call. + if unsafe { Process32FirstW(snapshot.as_raw(), &raw mut entry) } == 0 { + return tree; + } + + loop { + tree + .entry(entry.th32ParentProcessID) + .or_default() + .push(entry.th32ProcessID); + + // SAFETY: `snapshot` remains a valid Toolhelp snapshot handle, and `entry` + // remains a writable `PROCESSENTRY32W` with its ABI size preserved. + if unsafe { Process32NextW(snapshot.as_raw(), &raw mut entry) } == 0 { + break; + } + } + + tree + } + + /// Process groups are not exposed on Windows. + /// Always returns `false`. + pub const fn kill_process_group(_pgid: i32, _signal: i32) -> bool { + false + } + + /// Find processes whose `QueryFullProcessImageNameW` result equals `target`. + pub fn find_by_path(target: &str) -> Vec<Process> { + use std::{ffi::OsString, os::windows::ffi::OsStringExt}; + + let mut matches = Vec::new(); + let Some(snapshot) = create_process_snapshot() else { + return matches; + }; + + let mut entry = process_entry(); + let mut buf = vec![0u16; 32_768]; + let target = OsString::from(target); + + // SAFETY: `snapshot` is a valid Toolhelp snapshot handle. `entry` points to a + // writable `PROCESSENTRY32W` whose `dwSize` field was initialized to the exact + // ABI size before the call. + if unsafe { Process32FirstW(snapshot.as_raw(), &raw mut entry) } == 0 { + return matches; + } + + loop { + let pid = entry.th32ProcessID; + if let Some(handle) = open_process(pid, PROCESS_QUERY_LIMITED_INFORMATION) { + let mut size = buf.len() as u32; + // SAFETY: `handle` was opened with query access and remains valid for the + // call. `buf` is writable for `size` UTF-16 code units, and `size` is a valid + // in/out parameter initialized to that capacity. + let ok = unsafe { + QueryFullProcessImageNameW(handle.as_raw(), 0, buf.as_mut_ptr(), &raw mut size) != 0 + }; + if ok { + let path = OsString::from_wide(&buf[..size as usize]); + if path == target + && let Some(process) = Process::from_pid(i32::try_from(pid).unwrap_or_default()) + { + matches.push(process); + } + } + } + + // SAFETY: `snapshot` remains a valid Toolhelp snapshot handle, and `entry` + // remains a writable `PROCESSENTRY32W` with its ABI size preserved. + if unsafe { Process32NextW(snapshot.as_raw(), &raw mut entry) } == 0 { + break; + } + } + + matches + } +} + +/// Stable process reference. +#[derive(Clone)] +pub struct Process { + inner: platform::Process, +} + +impl Process { + /// Open a stable process reference from a PID. + pub fn from_pid(pid: i32) -> Option<Self> { + platform::Process::from_pid(pid).map(Self::from_inner) + } + + /// Open stable process references whose executable path matches exactly. + pub fn from_path(path: String) -> Vec<Self> { + platform::find_by_path(&path) + .into_iter() + .map(Self::from_inner) + .collect() + } + + /// Operating-system process identifier for this process reference. + pub const fn pid(&self) -> i32 { + self.inner.pid() + } + + /// Parent process id for this process, when available. + pub fn ppid(&self) -> Option<i32> { + self.inner.parent_pid() + } + + /// Launch arguments for this process. + pub fn args(&self) -> Vec<String> { + self.inner.args() + } + + /// Send `signal` to this process and its descendants, children first. + /// + /// On Linux and macOS the signal is forwarded as-is. On Windows there is no + /// signal abstraction, so the `signal` argument is ignored and the entire + /// tree is hard-killed via `TerminateProcess`. Defaults to the POSIX + /// hard-kill signal. + pub fn kill_tree(&self, signal: Option<i32>) -> u32 { + self.signal_tree(signal.unwrap_or(KILL_SIGNAL)) + } + + /// Process group id for this process, when supported by the platform. + pub fn group_id(&self) -> Option<i32> { + self.inner.group_id() + } + + /// Direct children of this process as stable process references. + pub fn children(&self) -> Vec<Self> { + self + .inner + .children() + .into_iter() + .map(Self::from_inner) + .collect() + } + + /// Current status of this process reference. + pub fn status(&self) -> ProcessStatus { + self.inner.status() + } + + /// Gracefully terminate this process and its descendants. + /// + /// Sends `TERM_SIGNAL` to the optional process group, every live descendant, + /// and the root, then optionally waits up to `graceful_ms` for the tree to + /// exit before escalating to `KILL_SIGNAL`. Pass `graceful_ms < 0` to skip + /// the wait entirely (the polite signal is still emitted). Returns `true` + /// when the tree has exited by the end of the hard wave's wait window. + pub async fn terminate_tree( + &self, + group: bool, + graceful_ms: i32, + timeout_ms: u32, + ct: CancelToken, + ) -> Result<bool> { + self + .terminate_tree_impl(group, graceful_ms, timeout_ms, ct) + .await + } + + /// Wait until this process exits, optionally bounded by `timeout`. + pub async fn wait_for_exit(&self, timeout: Option<Duration>, ct: CancelToken) -> Result<bool> { + wait_for_exit(self, &[], timeout, ct).await + } +} + +impl Process { + const fn from_inner(inner: platform::Process) -> Self { + Self { inner } + } + + /// Walk the live descendant tree from scratch. Cheap and idempotent — call + /// it again before each signal wave so grandchildren spawned during a grace + /// period are not missed. + fn live_descendants(&self) -> Vec<Self> { + self + .inner + .descendants() + .into_iter() + .map(Self::from_inner) + .collect() + } + + fn signal_tree(&self, signal: i32) -> u32 { + let descendants = self.live_descendants(); + let mut signaled = 0u32; + // If self leads its own process group, also signal the group — this catches + // grandchildren reparented to init when their immediate parent died inside + // the descendant walk. + if let Some(pgid) = self.inner.group_id() + && pgid == self.inner.pid() + { + let _ = kill_process_group(pgid, signal); + } + for child in &descendants { + if child.inner.kill(signal) { + signaled += 1; + } + } + if self.inner.kill(signal) { + signaled += 1; + } + signaled + } + + async fn terminate_tree_impl( + &self, + group: bool, + graceful_ms: i32, + timeout_ms: u32, + ct: CancelToken, + ) -> Result<bool> { + if self.status() != ProcessStatus::Running { + return Ok(true); + } + + let process_group = if group { self.group_id() } else { None }; + + // Polite wave: SIGTERM the group, every live descendant, then the root. + if let Some(pgid) = process_group { + let _ = kill_process_group(pgid, TERM_SIGNAL); + } + let mut descendants = self.live_descendants(); + for child in &descendants { + let _ = child.inner.kill(TERM_SIGNAL); + } + let _ = self.inner.kill(TERM_SIGNAL); + + // Optional grace wait. A negative `graceful_ms` skips the wait entirely + // (we still emit the polite signal so cleanup handlers can run before KILL). + if graceful_ms >= 0 { + let exited = wait_for_exit( + self, + &descendants, + Some(Duration::from_millis(graceful_ms as u64)), + ct.clone(), + ) + .await?; + if exited { + return Ok(true); + } + } + + // Hard wave. Re-walk the tree so any grandchild spawned during the grace + // period — or any process re-parented to the root — is signalled too. + if let Some(pgid) = process_group { + let _ = kill_process_group(pgid, KILL_SIGNAL); + } + descendants = self.live_descendants(); + for child in &descendants { + let _ = child.inner.kill(KILL_SIGNAL); + } + let _ = self.inner.kill(KILL_SIGNAL); + + wait_for_exit(self, &descendants, Some(Duration::from_millis(u64::from(timeout_ms))), ct) + .await + } +} + +async fn wait_for_exit( + root: &Process, + descendants: &[Process], + timeout: Option<Duration>, + ct: CancelToken, +) -> Result<bool> { + ct.heartbeat()?; + if root.status() != ProcessStatus::Running + && descendants + .iter() + .all(|process| process.status() != ProcessStatus::Running) + { + return Ok(true); + } + + let poll_interval = Duration::from_millis(50); + let mut elapsed = Duration::ZERO; + while timeout.is_none_or(|limit| elapsed < limit) { + let sleep_for = + timeout.map_or(poll_interval, |limit| limit.saturating_sub(elapsed).min(poll_interval)); + if sleep_for.is_zero() { + break; + } + ct.heartbeat()?; + tokio::time::sleep(sleep_for).await; + elapsed += sleep_for; + + if root.status() != ProcessStatus::Running + && descendants + .iter() + .all(|process| process.status() != ProcessStatus::Running) + { + return Ok(true); + } + } + + Ok(false) +} + +/// Send `signal` to the process group `pgid`. +/// Returns false when process groups are unsupported on the platform. +#[allow(clippy::missing_const_for_fn, reason = "Dispatches to platform-specific implementation")] +pub fn kill_process_group(pgid: i32, signal: i32) -> bool { + // Defense in depth: refuse to deliver a signal to the harness's own + // process group. Doing so terminates the harness along with the targets. + // Higher layers (`add_new_descendants`) already filter pgids by descendant + // ownership; this catches any future caller that bypasses that filter. + if pgid <= 0 || is_self_process_group(pgid) { + return false; + } + platform::kill_process_group(pgid, signal) +} + +#[cfg(unix)] +fn is_self_process_group(pgid: i32) -> bool { + // SAFETY: `getpgid(0)` queries the calling process's pgid and does not access + // caller-owned memory. A return value <= 0 is treated as "unknown", which + // fails open so the actual signal call decides. + let self_pgid = unsafe { libc::getpgid(0) }; + self_pgid > 0 && self_pgid == pgid +} + +#[cfg(not(unix))] +const fn is_self_process_group(_pgid: i32) -> bool { + false +} + +/// POSIX `SIGTERM` / Windows polite termination sentinel. +pub const TERM_SIGNAL: i32 = 15; + +/// POSIX `SIGKILL` / Windows hard-termination sentinel. +pub const KILL_SIGNAL: i32 = 9; + +/// A collection of process groups and process trees scheduled for +/// termination together. +/// +/// Built incrementally from job records or PTY metadata, then signalled +/// in escalating waves (typically `TERM_SIGNAL` followed by +/// `KILL_SIGNAL` after a grace period). Process-group calls are no-ops +/// on platforms that do not expose process groups. +#[derive(Default)] +pub struct TerminationTargets { + pgids: Vec<i32>, + processes: Vec<Process>, + seen_pids: HashSet<i32>, +} + +impl TerminationTargets { + /// Create an empty target set. + pub fn new() -> Self { + Self::default() + } + + /// Record a process group id. Duplicates are ignored. + pub fn add_pgid(&mut self, pgid: i32) { + if pgid > 0 && !self.pgids.contains(&pgid) { + self.pgids.push(pgid); + } + } + + /// Record a pid. Duplicates are ignored. If the pid is alive, opens + /// a stable [`Process`] reference so the descendant tree can be + /// killed even if the original pid is reused later. + pub fn add_pid(&mut self, pid: i32) { + if self.seen_pids.insert(pid) + && let Some(process) = Process::from_pid(pid) + { + self.processes.push(process); + } + } + + /// True when no targets have been recorded. + pub const fn is_empty(&self) -> bool { + self.pgids.is_empty() && self.processes.is_empty() + } + + /// Send `signal` to every recorded target. Failures are swallowed: + /// targets routinely exit between collection and signalling, and + /// the caller's policy is "best effort". + pub fn signal(&self, signal: i32) { + for &pgid in &self.pgids { + let _ = kill_process_group(pgid, signal); + } + for process in &self.processes { + let _ = process.signal_tree(signal); + } + } +} + +#[must_use] +pub fn current_descendant_pids() -> HashSet<i32> { + Process::from_pid(i32::try_from(std::process::id()).unwrap_or_default()).map_or_else( + HashSet::new, + |process| { + process + .live_descendants() + .into_iter() + .map(|child| child.pid()) + .collect() + }, + ) +} + +pub fn add_new_descendants<S: std::hash::BuildHasher>( + targets: &mut TerminationTargets, + baseline: &HashSet<i32, S>, +) { + let self_pid = i32::try_from(std::process::id()).unwrap_or_default(); + let Some(process) = Process::from_pid(self_pid) else { + return; + }; + let descendants = process.live_descendants(); + let descendants_info: Vec<DescendantInfo> = descendants + .iter() + .map(|child| DescendantInfo { pid: child.pid(), pgid: child.group_id() }) + .collect(); + + let selection = select_termination_targets(&descendants_info, baseline); + for pgid in selection.pgids { + targets.add_pgid(pgid); + } + for pid in selection.pids { + targets.add_pid(pid); + } +} + +/// Light view of a descendant for target classification — just enough to +/// decide which pgids/pids belong in the kill set without holding any +/// platform-specific process handles. +#[derive(Debug, Clone, Copy)] +struct DescendantInfo { + pid: i32, + pgid: Option<i32>, +} + +/// Classified termination targets returned by [`select_termination_targets`]. +#[derive(Debug, Default)] +struct TargetSelection { + pgids: Vec<i32>, + pids: Vec<i32>, +} + +/// Pure target-classifier separated from process discovery so it is testable +/// without depending on the platform's process-listing primitives (libproc on +/// macOS, `/proc` on Linux). +/// +/// **Critical**: a `pgid` is only adopted when its leader is itself one of the +/// new descendants. Without that check, a descendant that inherited the +/// harness's pgid — any subprocess started via APIs that do not call `setpgid`, +/// such as a sibling LSP/MCP helper spawned outside of brush — would drag +/// `harness.pgid` into the kill set, and the subsequent +/// `kill(-harness.pgid, SIGTERM)` would terminate the harness alongside the +/// intended targets. Pids of new descendants are still tracked individually so +/// the descendant tree can be reaped via `signal_tree`. +fn select_termination_targets<S: std::hash::BuildHasher>( + descendants: &[DescendantInfo], + baseline: &HashSet<i32, S>, +) -> TargetSelection { + let new_descendant_pids: HashSet<i32> = descendants + .iter() + .map(|info| info.pid) + .filter(|pid| !baseline.contains(pid)) + .collect(); + + let mut selection = TargetSelection::default(); + let mut seen_pgids: HashSet<i32> = HashSet::new(); + for info in descendants { + if !new_descendant_pids.contains(&info.pid) { + continue; + } + if let Some(pgid) = info.pgid + && pgid > 0 + && new_descendant_pids.contains(&pgid) + && seen_pgids.insert(pgid) + { + selection.pgids.push(pgid); + } + selection.pids.push(info.pid); + } + selection +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Regression test for the cancellation-kills-harness bug. + /// + /// When the descendant walk harvested each descendant's `pgid` and pushed + /// it onto the kill list, a descendant that inherited the harness's pgid + /// — any subprocess started via APIs that do not call `setpgid`, such as a + /// sibling LSP/MCP helper — dragged `harness.pgid` into the kill set, and + /// the subsequent `kill(-harness.pgid, SIGTERM)` killed the harness. + /// + /// Encode the dangerous shape directly: a new descendant whose `pgid` + /// resolves to something the harness owns (not in the new descendant set) + /// must contribute its pid for individual cleanup but **must not** drag its + /// pgid into the group-signal list. + #[test] + fn select_targets_drops_inherited_harness_pgid() { + const HARNESS_PGID: i32 = 1000; + const BASELINE_HELPER_PID: i32 = 1500; + + // Harness pgid is *not* a new descendant; a baseline helper happens to + // lead a group that a new descendant inherited. Neither pgid is safe to + // signal as a group. + let descendants = [DescendantInfo { pid: 2000, pgid: Some(HARNESS_PGID) }, DescendantInfo { + pid: 2001, + pgid: Some(BASELINE_HELPER_PID), + }]; + let baseline: HashSet<i32> = std::iter::once(BASELINE_HELPER_PID).collect(); + + let selection = select_termination_targets(&descendants, &baseline); + + assert!( + selection.pgids.is_empty(), + "no pgid should be added when leaders live outside the new descendant set; got {:?}", + selection.pgids, + ); + assert_eq!( + selection.pids, + vec![2000, 2001], + "new descendant pids must still be tracked individually for tree cleanup", + ); + } + + #[test] + fn select_targets_adopts_owned_process_group() { + // A new descendant that *is* the group leader — brush's `NewProcessGroup` + // path — contributes both its pid and its pgid, so grandchildren in the + // same group get reaped in one signal wave. + let leader = DescendantInfo { pid: 3000, pgid: Some(3000) }; + let grandchild = DescendantInfo { pid: 3001, pgid: Some(3000) }; + let baseline: HashSet<i32> = HashSet::new(); + + let selection = select_termination_targets(&[leader, grandchild], &baseline); + + assert_eq!(selection.pgids, vec![3000]); + assert_eq!(selection.pids, vec![3000, 3001]); + } + + #[test] + fn select_targets_skips_baseline_descendants() { + let old = DescendantInfo { pid: 4000, pgid: Some(4000) }; + let fresh = DescendantInfo { pid: 4100, pgid: Some(4100) }; + let baseline: HashSet<i32> = std::iter::once(4000).collect(); + + let selection = select_termination_targets(&[old, fresh], &baseline); + + assert_eq!(selection.pgids, vec![4100]); + assert_eq!(selection.pids, vec![4100]); + } + + #[test] + fn select_targets_dedupes_shared_process_group() { + let a = DescendantInfo { pid: 5000, pgid: Some(5000) }; + let b = DescendantInfo { pid: 5001, pgid: Some(5000) }; + let c = DescendantInfo { pid: 5002, pgid: Some(5000) }; + let baseline: HashSet<i32> = HashSet::new(); + + let selection = select_termination_targets(&[a, b, c], &baseline); + + assert_eq!( + selection.pgids, + vec![5000], + "each pgid should be recorded exactly once even when many descendants share it", + ); + assert_eq!(selection.pids, vec![5000, 5001, 5002]); + } + + /// `kill_process_group` is the last line of defense: even if a future + /// caller manages to feed the harness's own pgid into the signal path, + /// this wrapper must refuse to deliver the signal. + #[cfg(unix)] + #[test] + fn kill_process_group_refuses_self_pgroup() { + // SAFETY: `getpgid(0)` queries the calling process and does not touch + // caller-owned memory. + let self_pgid = unsafe { libc::getpgid(0) }; + assert!(self_pgid > 0, "getpgid(0) failed"); + assert!( + !kill_process_group(self_pgid, TERM_SIGNAL), + "kill_process_group must refuse the harness pgid; otherwise the test process would have \ + been SIGTERMed", + ); + assert!( + !kill_process_group(0, TERM_SIGNAL), + "kill_process_group must reject non-positive pgids", + ); + } + + /// Regression test for the macOS `proc_listchildpids` brokenness: on + /// darwin 25.4+ the kernel returns no entries when a process queries its + /// own children via that API, so `Process::descendants` produced an empty + /// list and termination cleanup silently became a no-op. The replacement + /// path scans `proc_listallpids` and groups by `pbi_ppid`, which actually + /// works. Linux has always worked via `/proc`. + #[cfg(unix)] + #[test] + fn descendants_includes_freshly_spawned_child() { + use std::{process::Command, thread, time::Duration}; + + let mut child = Command::new("sleep") + .arg("10") + .spawn() + .expect("spawn sleep"); + let child_pid = i32::try_from(child.id()).expect("child pid fits in i32"); + + let self_pid = i32::try_from(std::process::id()).expect("self pid fits in i32"); + let harness = Process::from_pid(self_pid).expect("harness Process ref"); + + // Allow a few polling iterations so the kernel's process-table query + // settles on a loaded host. proc_listallpids reflects newly forked pids + // within milliseconds in practice; 1s is a comfortable upper bound. + let mut found = false; + for _ in 0..40 { + if harness + .live_descendants() + .iter() + .any(|descendant| descendant.pid() == child_pid) + { + found = true; + break; + } + thread::sleep(Duration::from_millis(25)); + } + + let _ = child.kill(); + let _ = child.wait(); + + assert!( + found, + "freshly spawned child pid {child_pid} must appear in `live_descendants` so the \ + cancellation cleanup can reach it; this regressed on macOS when the walk relied on the \ + broken `proc_listchildpids`", + ); + } +} diff --git a/crates/pi-shell/src/shell.rs b/crates/pi-shell/src/shell.rs new file mode 100644 index 000000000..2b93a65f9 --- /dev/null +++ b/crates/pi-shell/src/shell.rs @@ -0,0 +1,1508 @@ +//! Runtime-agnostic brush shell execution. + +#[cfg(windows)] +use std::collections::HashSet; +use std::{ + collections::HashMap, + fs, + io::{self, Write}, + str, + sync::Arc, + time::Duration, +}; + +use anyhow::{Error, Result}; +use brush_builtins::{BuiltinSet, default_builtins}; +use brush_core::{ + ExecutionContext, ExecutionControlFlow, ExecutionExitCode, ExecutionResult, ProcessGroupPolicy, + ProfileLoadBehavior, RcLoadBehavior, Shell as BrushShell, ShellValue, ShellVariable, SourceInfo, + builtins, + env::EnvironmentScope, + openfiles::{self, OpenFile, OpenFiles}, +}; +use clap::Parser; +#[cfg(not(unix))] +use tokio::io::AsyncReadExt as _; +use tokio::{ + sync::{Mutex as TokioMutex, mpsc}, + time, +}; +use tokio_util::sync::CancellationToken; + +#[cfg(windows)] +use crate::windows::configure_windows_path; +use crate::{ + cancel::{AbortReason, AbortToken, CancelToken}, + minimizer, process, +}; + +struct ShellSessionCore { + shell: BrushShell, +} + +#[derive(Clone, Default)] +struct ShellAbortState(Arc<TokioMutex<Option<AbortToken>>>); + +impl ShellAbortState { + async fn set(&self, abort_token: AbortToken) { + *self.0.lock().await = Some(abort_token); + } + + async fn clear(&self) { + *self.0.lock().await = None; + } + + async fn abort(&self) { + let abort_token = self.0.lock().await.clone(); + if let Some(abort_token) = abort_token { + abort_token.abort(AbortReason::Signal); + } + } +} + +#[derive(Clone)] +struct ShellConfig { + session_env: Option<HashMap<String, String>>, + snapshot_path: Option<String>, + minimizer: Option<minimizer::MinimizerConfig>, +} + +#[derive(Debug, Clone, Default)] +pub struct ShellOptions { + pub session_env: Option<HashMap<String, String>>, + pub snapshot_path: Option<String>, + pub minimizer: Option<minimizer::MinimizerOptions>, +} + +struct ShellRunConfig { + command: String, + cwd: Option<String>, + env: Option<HashMap<String, String>>, + minimizer: Option<minimizer::MinimizerConfig>, +} + +#[derive(Debug, Clone, Default)] +pub struct ShellRunOptions { + pub command: String, + pub cwd: Option<String>, + pub env: Option<HashMap<String, String>>, + pub timeout_ms: Option<u32>, +} + +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct MinimizerResult { + pub filter: String, + pub text: String, + pub original_text: String, + pub input_bytes: u32, + pub output_bytes: u32, +} + +#[derive(Debug, Clone, serde::Serialize, serde::Deserialize)] +pub struct ShellRunResult { + pub exit_code: Option<i32>, + pub cancelled: bool, + pub timed_out: bool, + pub minimized: Option<MinimizerResult>, +} + +#[derive(Debug, Clone, Default)] +pub struct ShellExecuteOptions { + pub command: String, + pub cwd: Option<String>, + pub env: Option<HashMap<String, String>>, + pub session_env: Option<HashMap<String, String>>, + pub timeout_ms: Option<u32>, + pub snapshot_path: Option<String>, + pub minimizer: Option<minimizer::MinimizerOptions>, +} + +pub type ShellExecuteResult = ShellRunResult; + +pub struct Shell { + session: Arc<TokioMutex<Option<ShellSessionCore>>>, + abort_state: ShellAbortState, + config: ShellConfig, +} + +impl Shell { + #[must_use] + pub fn new(options: Option<ShellOptions>) -> Self { + let config = match options { + None => ShellConfig { session_env: None, snapshot_path: None, minimizer: None }, + Some(opt) => { + let minimizer = opt + .minimizer + .as_ref() + .map(minimizer::MinimizerConfig::from_options); + ShellConfig { + session_env: opt.session_env, + snapshot_path: opt.snapshot_path, + minimizer, + } + }, + }; + Self { + session: Arc::new(TokioMutex::new(None)), + abort_state: ShellAbortState::default(), + config, + } + } + + pub async fn run( + &self, + options: ShellRunOptions, + on_chunk: Option<mpsc::UnboundedSender<String>>, + mut cancel_token: CancelToken, + ) -> Result<ShellRunResult> { + let run_config = ShellRunConfig { + command: options.command, + cwd: options.cwd, + env: options.env, + minimizer: self.config.minimizer.clone(), + }; + run_shell_session( + self.session.clone(), + self.abort_state.clone(), + self.config.clone(), + run_config, + on_chunk, + &mut cancel_token, + ) + .await + } + + pub async fn abort(&self) { + self.abort_state.abort().await; + } +} + +pub async fn execute_shell( + options: ShellExecuteOptions, + on_chunk: Option<mpsc::UnboundedSender<String>>, + cancel_token: CancelToken, +) -> Result<ShellExecuteResult> { + let minimizer = options + .minimizer + .as_ref() + .map(minimizer::MinimizerConfig::from_options); + let config = ShellConfig { + session_env: options.session_env, + snapshot_path: options.snapshot_path, + minimizer: minimizer.clone(), + }; + let run_config = + ShellRunConfig { command: options.command, cwd: options.cwd, env: options.env, minimizer }; + run_shell_oneshot(config, run_config, on_chunk, cancel_token).await +} + +async fn run_shell_session( + session: Arc<TokioMutex<Option<ShellSessionCore>>>, + abort_state: ShellAbortState, + config: ShellConfig, + run_config: ShellRunConfig, + on_chunk: Option<mpsc::UnboundedSender<String>>, + ct: &mut CancelToken, +) -> Result<ShellRunResult> { + let tokio_cancel = CancellationToken::new(); + + let mut run_task = tokio::spawn({ + let session = session.clone(); + let abort_state = abort_state.clone(); + let tokio_cancel = tokio_cancel.clone(); + let at = ct.emplace_abort_token(); + async move { + let mut session_guard = session.lock().await; + + let session = match &mut *session_guard { + Some(session) => session, + None => session_guard.insert(create_session(&config).await?), + }; + abort_state.set(at).await; + run_shell_command(session, &run_config, on_chunk, tokio_cancel).await + } + }); + + let res = tokio::select! { + res = &mut run_task => res, + reason = ct.wait() => { + tokio_cancel.cancel(); + let graceful = time::timeout(Duration::from_secs(2), &mut run_task).await; + if graceful.is_err() { + run_task.abort(); + let _ = run_task.await; + } + abort_state.clear().await; + // Use try_lock to avoid deadlocking if another task holds the session. + // If we can't acquire the lock, the session will be cleaned up when the + // holding task finishes. + if let Ok(mut guard) = session.try_lock() { + *guard = None; + } + return Ok(ShellRunResult { + exit_code: None, + cancelled: matches!(reason, AbortReason::Signal), + timed_out: matches!(reason, AbortReason::Timeout), + minimized: None, + }); + } + }; + let res = + res.unwrap_or_else(|err| Err(Error::msg(format!("Shell execution task failed: {err}")))); + abort_state.clear().await; + + let keepalive = res.as_ref().is_ok_and(|pair| session_keepalive(&pair.0)); + if !keepalive { + *session.lock().await = None; + } + let (exec, minimized) = res?; + Ok(ShellRunResult { + exit_code: Some(exit_code(&exec)), + cancelled: false, + timed_out: false, + minimized, + }) +} + +async fn run_shell_oneshot( + config: ShellConfig, + run_config: ShellRunConfig, + on_chunk: Option<mpsc::UnboundedSender<String>>, + ct: CancelToken, +) -> Result<ShellExecuteResult> { + let tokio_cancel = CancellationToken::new(); + + let mut task = tokio::spawn({ + let tokio_cancel = tokio_cancel.clone(); + async move { + let mut session = create_session(&config).await?; + run_shell_command(&mut session, &run_config, on_chunk, tokio_cancel).await + } + }); + + let run_result = tokio::select! { + result = &mut task => result, + reason = ct.wait() => { + tokio_cancel.cancel(); + let graceful = time::timeout(Duration::from_secs(2), &mut task).await; + if graceful.is_err() { + task.abort(); + let _ = task.await; + } + return Ok(ShellExecuteResult { + exit_code: None, + cancelled: matches!(reason, AbortReason::Signal), + timed_out: matches!(reason, AbortReason::Timeout), + minimized: None, + }); + }, + }; + + let res = run_result + .unwrap_or_else(|err| Err(Error::msg(format!("Shell execution task failed: {err}")))); + let (exec, minimized) = res?; + Ok(ShellExecuteResult { + exit_code: Some(exit_code(&exec)), + cancelled: false, + timed_out: false, + minimized, + }) +} + +fn null_file() -> Result<OpenFile> { + openfiles::null().map_err(|err| Error::msg(format!("Failed to create null file: {err}"))) +} + +const fn exit_code(result: &ExecutionResult) -> i32 { + match result.exit_code { + ExecutionExitCode::Success => 0, + ExecutionExitCode::GeneralError => 1, + ExecutionExitCode::InvalidUsage => 2, + ExecutionExitCode::Unimplemented => 99, + ExecutionExitCode::CannotExecute => 126, + ExecutionExitCode::NotFound => 127, + ExecutionExitCode::Interrupted => 130, + ExecutionExitCode::BrokenPipe => 141, + ExecutionExitCode::Custom(code) => code as i32, + } +} + +#[cfg(windows)] +const fn normalize_env_key(key: &str) -> &str { + if key.eq_ignore_ascii_case("PATH") { + "PATH" + } else { + key + } +} + +#[cfg(not(windows))] +const fn normalize_env_key(key: &str) -> &str { + key +} + +#[cfg(windows)] +fn merge_path_values(existing: &str, incoming: &str) -> String { + let mut merged = Vec::new(); + let mut seen = HashSet::new(); + push_unique_paths(&mut merged, &mut seen, existing); + push_unique_paths(&mut merged, &mut seen, incoming); + + std::env::join_paths(merged.iter()) + .map_or_else(|_| merged.join(";"), |paths| paths.to_string_lossy().into_owned()) +} + +#[cfg(windows)] +fn push_unique_paths(merged: &mut Vec<String>, seen: &mut HashSet<String>, value: &str) { + for segment in std::env::split_paths(value) { + let segment_str = segment.to_string_lossy().into_owned(); + let normalized = normalize_path_segment(&segment_str); + if normalized.is_empty() { + continue; + } + if seen.insert(normalized) { + merged.push(segment_str); + } + } +} + +#[cfg(windows)] +fn normalize_path_segment(segment: &str) -> String { + let trimmed = segment.trim().trim_matches('"'); + if trimmed.is_empty() { + return String::new(); + } + + let mut normalized = std::path::PathBuf::new(); + for component in std::path::Path::new(trimmed).components() { + normalized.push(component.as_os_str()); + } + + normalized.to_string_lossy().to_ascii_lowercase() +} + +#[cfg(not(windows))] +fn merge_path_values(_existing: &str, incoming: &str) -> String { + incoming.to_string() +} + +async fn create_session(config: &ShellConfig) -> Result<ShellSessionCore> { + let mut shell = BrushShell::builder() + .do_not_inherit_env(true) + .profile(ProfileLoadBehavior::Skip) + .rc(RcLoadBehavior::Skip) + .builtins(default_builtins(BuiltinSet::BashMode)) + .build() + .await + .map_err(|err| Error::msg(format!("Failed to initialize shell: {err}")))?; + + if let Some(exec_builtin) = shell.builtin_mut("exec") { + exec_builtin.disabled = true; + } + if let Some(suspend_builtin) = shell.builtin_mut("suspend") { + suspend_builtin.disabled = true; + } + shell.register_builtin("sleep", builtins::builtin::<SleepCommand, _>()); + shell.register_builtin("timeout", builtins::builtin::<TimeoutCommand, _>()); + + let mut merged_path: Option<String> = None; + for (key, value) in std::env::vars() { + let normalized_key = normalize_env_key(&key); + if should_skip_env_var(normalized_key) { + continue; + } + if normalized_key == "PATH" { + merged_path = Some(match merged_path { + Some(existing) => merge_path_values(&existing, &value), + None => value, + }); + continue; + } + let mut var = ShellVariable::new(ShellValue::String(value)); + var.export(); + shell + .env_mut() + .set_global(normalized_key, var) + .map_err(|err| Error::msg(format!("Failed to set env: {err}")))?; + } + + #[cfg(windows)] + if merged_path.is_none() + && let Some(value) = std::env::var_os("Path").or_else(|| std::env::var_os("PATH")) + { + merged_path = Some(value.to_string_lossy().into_owned()); + } + + if let Some(path_value) = merged_path { + let mut var = ShellVariable::new(ShellValue::String(path_value)); + var.export(); + shell + .env_mut() + .set_global("PATH", var) + .map_err(|err| Error::msg(format!("Failed to set env: {err}")))?; + } + + if let Some(env) = config.session_env.as_ref() { + for (key, value) in env { + let normalized_key = normalize_env_key(key); + if should_skip_env_var(normalized_key) { + continue; + } + let mut var = ShellVariable::new(ShellValue::String(value.clone())); + var.export(); + shell + .env_mut() + .set_global(normalized_key, var) + .map_err(|err| Error::msg(format!("Failed to set env: {err}")))?; + } + } + + #[cfg(windows)] + configure_windows_path(&mut shell)?; + + if let Some(snapshot_path) = config.snapshot_path.as_ref() { + source_snapshot(&mut shell, snapshot_path).await?; + } + + Ok(ShellSessionCore { shell }) +} + +async fn source_snapshot(shell: &mut BrushShell, snapshot_path: &str) -> Result<()> { + let mut params = shell.default_exec_params(); + let source_info = SourceInfo::from("pi-natives:snapshot"); + params.set_fd(OpenFiles::STDIN_FD, null_file()?); + params.set_fd(OpenFiles::STDOUT_FD, null_file()?); + params.set_fd(OpenFiles::STDERR_FD, null_file()?); + + let escaped = snapshot_path.replace('\'', "'\\''"); + let command = format!("source '{escaped}'"); + shell + .run_string(command, &source_info, ¶ms) + .await + .map_err(|err| Error::msg(format!("Failed to source snapshot: {err}")))?; + Ok(()) +} + +async fn run_shell_command( + session: &mut ShellSessionCore, + options: &ShellRunConfig, + on_chunk: Option<mpsc::UnboundedSender<String>>, + cancel_token: CancellationToken, +) -> Result<(ExecutionResult, Option<MinimizerResult>)> { + if let Some(cwd) = options.cwd.as_deref() { + session + .shell + .set_working_dir(cwd) + .map_err(|err| Error::msg(format!("Failed to set cwd: {err}")))?; + } + + let mut env_scope_pushed = false; + if let Some(env) = options.env.as_ref() { + session + .shell + .env_mut() + .push_scope(EnvironmentScope::Command); + env_scope_pushed = true; + for (key, value) in env { + let normalized_key = normalize_env_key(key); + if should_skip_env_var(normalized_key) { + continue; + } + let mut var = ShellVariable::new(ShellValue::String(value.clone())); + var.export(); + if let Err(err) = + session + .shell + .env_mut() + .add(normalized_key, var, EnvironmentScope::Command) + { + let _ = session.shell.env_mut().pop_scope(EnvironmentScope::Command); + return Err(Error::msg(format!("Failed to set env: {err}"))); + } + } + } + + let minimizer_mode = if let Some(config) = options.minimizer.as_ref() { + minimizer::engine::mode_for(&options.command, config) + } else { + minimizer::engine::MinimizerMode::None + }; + let should_minimize = !matches!(minimizer_mode, minimizer::engine::MinimizerMode::None); + let max_capture_bytes = if let Some(config) = options.minimizer.as_ref() { + config.max_capture_bytes as usize + } else { + 0 + }; + + let (reader_file, writer_file) = pipe_to_files("output")?; + + let stdout_file = OpenFile::from( + writer_file + .try_clone() + .map_err(|err| Error::msg(format!("Failed to clone pipe: {err}")))?, + ); + let stderr_file = OpenFile::from(writer_file); + + let mut params = session.shell.default_exec_params(); + params.set_fd(OpenFiles::STDIN_FD, null_file()?); + params.set_fd(OpenFiles::STDOUT_FD, stdout_file); + params.set_fd(OpenFiles::STDERR_FD, stderr_file); + params.process_group_policy = ProcessGroupPolicy::NewProcessGroup; + params.set_cancel_token(cancel_token.clone()); + let baseline_descendants = process::current_descendant_pids(); + let reader_cancel = CancellationToken::new(); + let (activity_tx, mut activity_rx) = mpsc::channel::<()>(1); + // Stream every raw chunk to the caller live, regardless of whether + // minimization is enabled. When minimization actually transforms the + // output, we propagate the replacement text via `MinimizerResult.text` + // so the caller can swap their accumulated buffer for the minimized + // version without losing intermediate progress updates. + let reader_callback = on_chunk; + let mut reader_handle = tokio::spawn({ + let reader_cancel = reader_cancel.clone(); + async move { + if should_minimize { + let output = read_output_buffered( + reader_file, + reader_callback, + reader_cancel, + activity_tx, + max_capture_bytes, + ) + .await; + Result::<OutputRead>::Ok(OutputRead::Buffered(output)) + } else { + Box::pin(read_output(reader_file, reader_callback, reader_cancel, activity_tx)).await; + Result::<OutputRead>::Ok(OutputRead::Streaming) + } + } + }); + let cancel_bridge = tokio::spawn({ + let cancel_token = cancel_token.clone(); + let reader_cancel = reader_cancel.clone(); + async move { + cancel_token.cancelled().await; + reader_cancel.cancel(); + } + }); + let process_cancel_bridge = tokio::spawn({ + let cancel_token = cancel_token.clone(); + let baseline_descendants = baseline_descendants.clone(); + async move { + cancel_token.cancelled().await; + // Rescan-and-signal loop. Each pass picks up grandchildren spawned + // during the previous wave's grace period, then exits early as soon + // as no descendants remain. The first wave is SIGTERM so well-behaved + // programs get a chance to clean up; subsequent waves escalate to + // SIGKILL. Cheaper than the previous 20 Hz tracker loop and avoids + // the constant kernel chatter when no cancellation ever happens. + const WAVES: u32 = 3; + for wave in 0..WAVES { + let mut targets = process::TerminationTargets::new(); + process::add_new_descendants(&mut targets, &baseline_descendants); + if targets.is_empty() { + return; + } + let signal = if wave == 0 { + process::TERM_SIGNAL + } else { + process::KILL_SIGNAL + }; + targets.signal(signal); + if wave + 1 < WAVES { + let pause = if wave == 0 { + Duration::from_millis(75) + } else { + Duration::from_millis(150) + }; + time::sleep(pause).await; + } + } + } + }); + let source_info = SourceInfo::from("pi-natives:command"); + let result = session + .shell + .run_string(options.command.clone(), &source_info, ¶ms) + .await; + + if cancel_token.is_cancelled() { + terminate_background_jobs(&session.shell); + } + + if env_scope_pushed { + session + .shell + .env_mut() + .pop_scope(EnvironmentScope::Command) + .map_err(|err| Error::msg(format!("Failed to pop env scope: {err}")))?; + } + + drop(params); + + // The foreground command can complete while background jobs keep the + // stdout/stderr pipe open. Don't hang forever waiting for EOF; drain output + // for a short period, then cancel. + const POST_EXIT_IDLE: Duration = Duration::from_millis(250); + const POST_EXIT_MAX: Duration = Duration::from_secs(2); + const READER_SHUTDOWN_TIMEOUT: Duration = Duration::from_millis(250); + + let mut reader_finished = false; + let mut reader_output = None; + let mut idle_timer = Box::pin(time::sleep(POST_EXIT_IDLE)); + let mut max_timer = Box::pin(time::sleep(POST_EXIT_MAX)); + + loop { + tokio::select! { + res = &mut reader_handle => { + if let Ok(Ok(output)) = res { + reader_output = Some(output); + } + reader_finished = true; + break; + } + msg = activity_rx.recv() => { + if msg.is_none() { + break; + } + idle_timer.as_mut().reset(time::Instant::now() + POST_EXIT_IDLE); + } + () = &mut idle_timer => break, + () = &mut max_timer => break, + } + } + + if !reader_finished { + reader_cancel.cancel(); + if let Ok(res) = time::timeout(READER_SHUTDOWN_TIMEOUT, &mut reader_handle).await { + if let Ok(output) = res + && let Ok(output) = output + { + reader_output = Some(output); + } + } else { + reader_handle.abort(); + let _ = reader_handle.await; + } + } + cancel_bridge.abort(); + let _ = cancel_bridge.await; + if cancel_token.is_cancelled() { + // Cancel fired — the bridge is actively running its rescan-and-signal + // loop. Let it run to completion so all three waves get a chance to + // reach stragglers; aborting here would cut the kill loop short. + let _ = process_cancel_bridge.await; + } else { + // Happy path — the bridge is still parked on `cancel_token.cancelled()` + // and would never exit on its own. Tear it down. + process_cancel_bridge.abort(); + let _ = process_cancel_bridge.await; + } + + let result = result.map_err(|err| Error::msg(format!("Shell execution failed: {err}")))?; + let mut minimized_out: Option<MinimizerResult> = None; + if let Some(OutputRead::Buffered(output)) = reader_output + && let Some(config) = options.minimizer.as_ref() + && !output.exceeded + { + let minimized = match minimizer_mode { + minimizer::engine::MinimizerMode::WholeCommand => { + minimizer::apply(&options.command, &output.text, exit_code(&result), config) + }, + minimizer::engine::MinimizerMode::None => { + minimizer::MinimizerOutput::passthrough(&output.text) + }, + }; + if minimized.changed + && let Some(original) = minimized.original_text + { + let output_bytes = u32::try_from(minimized.text.len()).unwrap_or(u32::MAX); + minimized_out = Some(MinimizerResult { + filter: minimized.filter.to_string(), + text: minimized.text, + original_text: original, + input_bytes: u32::try_from(minimized.input_bytes).unwrap_or(u32::MAX), + output_bytes, + }); + } + } + Ok((result, minimized_out)) +} + +fn terminate_background_jobs(shell: &BrushShell) { + let mut targets = process::TerminationTargets::new(); + for job in &shell.jobs().jobs { + if let Some(pgid) = job.process_group_id() { + targets.add_pgid(pgid); + } + if let Some(pid) = job.representative_pid() { + targets.add_pid(pid); + } + } + if targets.is_empty() { + // Pure descendant cleanup is handled by `process_cancel_bridge` while + // the cancel was still in flight. Here we only signal brush's own + // job-tracked targets — pgids of background-group leaders that may have + // already exited (so the descendant walk would no longer find them as + // new descendants, but their group still holds live grandchildren). + return; + } + + targets.signal(process::TERM_SIGNAL); + tokio::spawn(async move { + time::sleep(Duration::from_millis(150)).await; + targets.signal(process::KILL_SIGNAL); + }); +} +fn should_skip_env_var(key: &str) -> bool { + if key.starts_with("BASH_FUNC_") && key.ends_with("%%") { + return true; + } + + matches!( + key, + "BASH_ENV" + | "ENV" + | "HISTFILE" + | "HISTTIMEFORMAT" + | "HISTCMD" + | "PS0" + | "PS1" + | "PS2" + | "PS4" + | "BRUSH_PS_ALT" + | "READLINE_LINE" + | "READLINE_POINT" + | "BRUSH_VERSION" + | "BASH" + | "BASHOPTS" + | "BASH_ALIASES" + | "BASH_ARGV0" + | "BASH_CMDS" + | "BASH_SOURCE" + | "BASH_SUBSHELL" + | "BASH_VERSINFO" + | "BASH_VERSION" + | "SHELLOPTS" + | "SHLVL" + | "SHELL" + | "COMP_WORDBREAKS" + | "DIRSTACK" + | "EPOCHREALTIME" + | "EPOCHSECONDS" + | "FUNCNAME" + | "GROUPS" + | "IFS" + | "LINENO" + | "MACHTYPE" + | "OSTYPE" + | "OPTERR" + | "OPTIND" + | "PIPESTATUS" + | "PPID" + | "PWD" + | "OLDPWD" + | "RANDOM" + | "SRANDOM" + | "SECONDS" + | "UID" + | "EUID" + | "HOSTNAME" + | "HOSTTYPE" + ) +} + +const fn session_keepalive(result: &ExecutionResult) -> bool { + match result.next_control_flow { + ExecutionControlFlow::Normal => true, + ExecutionControlFlow::BreakLoop { .. } => false, + ExecutionControlFlow::ContinueLoop { .. } => false, + ExecutionControlFlow::ReturnFromFunctionOrScript => false, + ExecutionControlFlow::ExitShell => false, + } +} + +enum OutputRead { + Streaming, + Buffered(BufferedOutput), +} + +struct BufferedOutput { + text: String, + exceeded: bool, +} + +async fn read_output( + reader: fs::File, + on_chunk: Option<mpsc::UnboundedSender<String>>, + cancel_token: CancellationToken, + activity: mpsc::Sender<()>, +) { + const REPLACEMENT: &str = "\u{FFFD}"; + const BUF: usize = 65536; + let mut buf = vec![0u8; BUF + 4]; // +4 for max UTF-8 char + let mut it = 0; + + #[cfg(unix)] + let Ok(reader) = register_nonblocking_pipe(reader) else { + return; + }; + #[cfg(not(unix))] + let reader = tokio::fs::File::from_std(reader); + #[cfg(not(unix))] + tokio::pin!(reader); + + loop { + #[cfg(unix)] + let n = { + let Ok(mut readiness) = (tokio::select! { + ready = reader.readable() => ready, + () = cancel_token.cancelled() => break, + }) else { + break; + }; + match readiness.try_io(|inner| read_nonblocking(inner.get_ref(), &mut buf[it..BUF])) { + Ok(Ok(0)) => break, + Ok(Ok(n)) => n, + Ok(Err(e)) if e.kind() == io::ErrorKind::Interrupted => continue, + Ok(Err(_)) => break, + Err(_would_block) => continue, + } + }; + #[cfg(not(unix))] + let n = { + let read_future = reader.read(&mut buf[it..BUF]); + tokio::pin!(read_future); + match tokio::select! { + res = &mut read_future => res, + () = cancel_token.cancelled() => break, + } { + Ok(0) => break, // EOF + Ok(n) => n, + Err(e) if e.kind() == io::ErrorKind::Interrupted => continue, + Err(_) => break, + } + }; + if n > 0 { + let _ = activity.try_send(()); + } + it += n; + + // Consume as much of `pending` as is decodable *right now*. + while it > 0 { + let pending = &buf[..it]; + match str::from_utf8(pending) { + Ok(text) => { + emit_chunk(text, on_chunk.as_ref()); + it = 0; + break; + }, + Err(err) => { + let p = err.valid_up_to(); + if p > 0 { + // SAFETY: [..p] is guaranteed valid UTF-8 by valid_up_to(). + let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; + emit_chunk(text, on_chunk.as_ref()); + // copy p..it to the beginning of the buffer + buf.copy_within(p..it, 0); + it -= p; + } + + match err.error_len() { + Some(p) => { + // Invalid byte sequence: emit replacement and drop those bytes. + emit_chunk(REPLACEMENT, on_chunk.as_ref()); + // copy p..it to the beginning of the buffer + buf.copy_within(p..it, 0); + it -= p; + // continue loop in case more bytes remain after the + // invalid sequence + }, + None => { + // Incomplete UTF-8 sequence at end: keep bytes for next read. + break; + }, + } + }, + } + } + } + + // Flush whatever is left at EOF (including an incomplete final sequence). + for chunk in buf[..it].utf8_chunks() { + let valid = chunk.valid(); + if !valid.is_empty() { + emit_chunk(valid, on_chunk.as_ref()); + } + if !chunk.invalid().is_empty() { + emit_chunk(REPLACEMENT, on_chunk.as_ref()); + } + } +} + +async fn read_output_buffered( + reader: fs::File, + on_chunk: Option<mpsc::UnboundedSender<String>>, + cancel_token: CancellationToken, + activity: mpsc::Sender<()>, + max_capture_bytes: usize, +) -> BufferedOutput { + const REPLACEMENT: &str = "\u{FFFD}"; + const BUF: usize = 65536; + let mut buf = vec![0u8; BUF]; + let mut captured = Vec::new(); + let mut exceeded = false; + // Pending bytes from a prior read that ended mid-UTF-8 sequence. We hold + // them back so we emit only valid UTF-8 to the streaming callback while + // still capturing every byte into `captured` for post-processing. + let mut pending = Vec::<u8>::new(); + + #[cfg(unix)] + let Ok(reader) = register_nonblocking_pipe(reader) else { + return BufferedOutput { text: String::new(), exceeded: true }; + }; + #[cfg(not(unix))] + let reader = tokio::fs::File::from_std(reader); + #[cfg(not(unix))] + tokio::pin!(reader); + + loop { + #[cfg(unix)] + let n = { + let Ok(mut readiness) = (tokio::select! { + ready = reader.readable() => ready, + () = cancel_token.cancelled() => break, + }) else { + break; + }; + match readiness.try_io(|inner| read_nonblocking(inner.get_ref(), &mut buf)) { + Ok(Ok(0)) => break, + Ok(Ok(n)) => n, + Ok(Err(e)) if e.kind() == io::ErrorKind::Interrupted => continue, + Ok(Err(_)) => break, + Err(_would_block) => continue, + } + }; + #[cfg(not(unix))] + let n = { + let read_future = reader.read(&mut buf); + tokio::pin!(read_future); + match tokio::select! { + res = &mut read_future => res, + () = cancel_token.cancelled() => break, + } { + Ok(0) => break, + Ok(n) => n, + Err(e) if e.kind() == io::ErrorKind::Interrupted => continue, + Err(_) => break, + } + }; + if n > 0 { + let _ = activity.try_send(()); + } + // Once `exceeded`, the post-process minimizer is bypassed (see the + // `!output.exceeded` gate at the call site), so further appends just + // grow `captured` without serving any purpose. Stop accumulating to + // bound peak memory on commands that produce very large output. + if !exceeded { + if captured.len().saturating_add(n) > max_capture_bytes { + exceeded = true; + } else { + captured.extend_from_slice(&buf[..n]); + } + } + + // Stream whatever is validly decodable *right now* to the callback, + // carrying incomplete trailing UTF-8 bytes over to the next iteration. + if let Some(cb) = on_chunk.as_ref() { + pending.extend_from_slice(&buf[..n]); + while !pending.is_empty() { + match str::from_utf8(&pending) { + Ok(text) => { + emit_chunk(text, Some(cb)); + pending.clear(); + break; + }, + Err(err) => { + let p = err.valid_up_to(); + if p > 0 { + // SAFETY: [..p] is valid UTF-8 per valid_up_to(). + let text = unsafe { str::from_utf8_unchecked(&pending[..p]) }; + emit_chunk(text, Some(cb)); + pending.drain(..p); + } + match err.error_len() { + Some(skip) => { + emit_chunk(REPLACEMENT, Some(cb)); + pending.drain(..skip); + }, + None => break, + } + }, + } + } + } + } + + // Flush any trailing bytes the streaming decoder held back at EOF. + if let Some(cb) = on_chunk.as_ref() { + for chunk in pending.utf8_chunks() { + let valid = chunk.valid(); + if !valid.is_empty() { + emit_chunk(valid, Some(cb)); + } + if !chunk.invalid().is_empty() { + emit_chunk(REPLACEMENT, Some(cb)); + } + } + } + + BufferedOutput { text: String::from_utf8_lossy(&captured).into_owned(), exceeded } +} + +#[cfg(unix)] +fn register_nonblocking_pipe(reader: fs::File) -> io::Result<tokio::io::unix::AsyncFd<fs::File>> { + set_nonblocking(&reader)?; + tokio::io::unix::AsyncFd::new(reader) +} + +#[cfg(unix)] +fn set_nonblocking<T: std::os::fd::AsRawFd>(file: &T) -> io::Result<()> { + let fd = file.as_raw_fd(); + // SAFETY: `fd` is owned by `file` and remains valid for the duration of + // these `fcntl` calls. + let flags = unsafe { libc::fcntl(fd, libc::F_GETFL) }; + if flags < 0 { + return Err(io::Error::last_os_error()); + } + if flags & libc::O_NONBLOCK != 0 { + return Ok(()); + } + + // SAFETY: `fd` remains valid here and we are only toggling `O_NONBLOCK`. + let result = unsafe { libc::fcntl(fd, libc::F_SETFL, flags | libc::O_NONBLOCK) }; + if result < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(()) + } +} + +#[cfg(unix)] +fn read_nonblocking<T: std::os::fd::AsRawFd>(file: &T, buf: &mut [u8]) -> io::Result<usize> { + // SAFETY: `buf` is writable for `buf.len()` bytes, and the raw fd obtained + // from `file` stays valid for the duration of the syscall. + let read = unsafe { libc::read(file.as_raw_fd(), buf.as_mut_ptr().cast(), buf.len()) }; + if read < 0 { + Err(io::Error::last_os_error()) + } else { + Ok(read as usize) + } +} + +fn emit_chunk(text: &str, callback: Option<&mpsc::UnboundedSender<String>>) { + if let Some(callback) = callback { + let _ = callback.send(text.to_string()); + } +} + +fn pipe_to_files(label: &str) -> Result<(fs::File, fs::File)> { + let (r, w) = + os_pipe::pipe().map_err(|err| Error::msg(format!("Failed to create {label} pipe: {err}")))?; + + #[cfg(unix)] + let (r, w): (fs::File, fs::File) = { + use std::os::unix::io::{FromRawFd, IntoRawFd}; + let r = r.into_raw_fd(); + let w = w.into_raw_fd(); + // SAFETY: We just obtained these fds from os_pipe and own them exclusively. + unsafe { (FromRawFd::from_raw_fd(r), FromRawFd::from_raw_fd(w)) } + }; + + #[cfg(windows)] + let (r, w): (fs::File, fs::File) = { + use std::os::windows::io::{FromRawHandle, IntoRawHandle}; + let r = r.into_raw_handle(); + let w = w.into_raw_handle(); + // SAFETY: We just obtained these handles from os_pipe and own them exclusively. + unsafe { (FromRawHandle::from_raw_handle(r), FromRawHandle::from_raw_handle(w)) } + }; + + Ok((r, w)) +} + +#[derive(Parser)] +#[command(disable_help_flag = true)] +struct SleepCommand { + #[arg(required = true)] + durations: Vec<String>, +} + +impl builtins::Command for SleepCommand { + type Error = brush_core::Error; + + fn execute<SE: brush_core::ShellExtensions>( + &self, + context: ExecutionContext<'_, SE>, + ) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send { + let durations = self.durations.clone(); + async move { + if context.is_cancelled() { + return Ok(ExecutionExitCode::Interrupted.into()); + } + let mut total = Duration::from_millis(0); + for duration in &durations { + let Some(parsed) = parse_duration(duration) else { + let _ = writeln!(context.stderr(), "sleep: invalid time interval '{duration}'"); + return Ok(ExecutionResult::new(1)); + }; + total += parsed; + } + let sleep = time::sleep(total); + tokio::pin!(sleep); + if let Some(cancel_token) = context.cancel_token() { + tokio::select! { + () = &mut sleep => Ok(ExecutionResult::success()), + () = cancel_token.cancelled() => Ok(ExecutionExitCode::Interrupted.into()), + } + } else { + sleep.await; + Ok(ExecutionResult::success()) + } + } + } +} + +#[derive(Parser)] +#[command(disable_help_flag = true)] +struct TimeoutCommand { + #[arg(required = true)] + duration: String, + #[arg(required = true, num_args = 1.., trailing_var_arg = true)] + command: Vec<String>, +} + +impl builtins::Command for TimeoutCommand { + type Error = brush_core::Error; + + fn execute<SE: brush_core::ShellExtensions>( + &self, + context: ExecutionContext<'_, SE>, + ) -> impl Future<Output = std::result::Result<ExecutionResult, brush_core::Error>> + Send { + let duration = self.duration.clone(); + let command = self.command.clone(); + async move { + if context.is_cancelled() { + return Ok(ExecutionExitCode::Interrupted.into()); + } + let Some(timeout) = parse_duration(&duration) else { + let _ = writeln!(context.stderr(), "timeout: invalid time interval '{duration}'"); + return Ok(ExecutionResult::new(125)); + }; + if command.is_empty() { + let _ = writeln!(context.stderr(), "timeout: missing command"); + return Ok(ExecutionResult::new(125)); + } + + let child_cancel = CancellationToken::new(); + let mut params = context.params.clone(); + params.process_group_policy = ProcessGroupPolicy::NewProcessGroup; + params.set_cancel_token(child_cancel.clone()); + + let mut command_line = String::new(); + for (idx, arg) in command.iter().enumerate() { + if idx > 0 { + command_line.push(' '); + } + command_line.push_str("e_arg(arg)); + } + + let cancel_token = context.cancel_token(); + let source_info = SourceInfo::from("pi-natives:timeout"); + let run_future = context + .shell + .run_string(command_line, &source_info, ¶ms); + tokio::pin!(run_future); + + if let Some(cancel_token) = cancel_token { + tokio::select! { + result = &mut run_future => result, + () = time::sleep(timeout) => { + child_cancel.cancel(); + // Wait briefly for the child to exit after cancellation. + let _ = time::timeout(Duration::from_secs(2), &mut run_future).await; + Ok(ExecutionResult::new(124)) + }, + () = cancel_token.cancelled() => { + child_cancel.cancel(); + Ok(ExecutionExitCode::Interrupted.into()) + }, + } + } else { + tokio::select! { + result = &mut run_future => result, + () = time::sleep(timeout) => { + child_cancel.cancel(); + // Wait briefly for the child to exit after cancellation. + let _ = time::timeout(Duration::from_secs(2), &mut run_future).await; + Ok(ExecutionResult::new(124)) + }, + } + } + } + } +} +fn parse_duration(input: &str) -> Option<Duration> { + let trimmed = input.trim(); + if trimmed.is_empty() { + return None; + } + let (number, multiplier) = match trimmed.chars().last()? { + 's' => (&trimmed[..trimmed.len() - 1], 1.0), + 'm' => (&trimmed[..trimmed.len() - 1], 60.0), + 'h' => (&trimmed[..trimmed.len() - 1], 3600.0), + 'd' => (&trimmed[..trimmed.len() - 1], 86400.0), + ch if ch.is_ascii_alphabetic() => return None, + _ => (trimmed, 1.0), + }; + let value = number.parse::<f64>().ok()?; + if value.is_sign_negative() { + return None; + } + let millis = value * multiplier * 1000.0; + if !millis.is_finite() || millis < 0.0 { + return None; + } + Some(Duration::from_millis(millis.round() as u64)) +} + +fn quote_arg(arg: &str) -> String { + if arg.is_empty() { + return "''".to_string(); + } + let safe = arg + .chars() + .all(|ch| ch.is_ascii_alphanumeric() || matches!(ch, '-' | '_' | '.' | '/' | ':' | '+')); + if safe { + return arg.to_string(); + } + let escaped = arg.replace('\'', "'\"'\"'"); + format!("'{escaped}'") +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Truth-table coverage for `brush_core::commands::child_session_action`. + /// + /// Lives in `pi-natives` because the brush-core crate is excluded from the + /// workspace (vendored upstream) and cannot be tested standalone — its tokio + /// dependency only resolves the `net` feature via feature-unification with + /// other workspace members. + mod child_session_action { + use brush_core::commands::{ChildSessionAction, child_session_action}; + + /// Interactive brush, leading its own pgroup, terminal stdin: foreground. + #[test] + fn interactive_with_terminal_stdin_takes_foreground() { + assert_eq!(child_session_action(true, true, false), ChildSessionAction::TakeForeground,); + // Terminal foregrounding wins even when this is the first stage of a + // pipeline; no detach is attempted. + assert_eq!(child_session_action(true, true, true), ChildSessionAction::TakeForeground,); + } + + /// Brush leading a new pgroup with non-terminal stdin detaches only when + /// it is not part of a multi-command pipeline. Pipeline leaders must stay + /// in the parent session so later stages can join their process group. + #[test] + fn non_terminal_stdin_leading_new_pgroup_detaches_unless_pipeline() { + assert_eq!(child_session_action(true, false, false), ChildSessionAction::DetachSession,); + assert_eq!(child_session_action(true, false, true), ChildSessionAction::None,); + } + + /// Non-interactive brush, terminal stdin, no pipeline: nothing to do. + #[test] + fn non_interactive_with_terminal_stdin_does_nothing() { + assert_eq!(child_session_action(false, true, false), ChildSessionAction::None,); + } + + /// Non-interactive brush, terminal stdin, joining a pipeline pgroup: + /// nothing to do (parent already wired pgroup membership). + #[test] + fn non_interactive_terminal_stdin_in_pipeline_does_nothing() { + assert_eq!(child_session_action(false, true, true), ChildSessionAction::None,); + } + + /// **Embedded host bug fix.** Non-interactive brush, non-terminal stdin, + /// no pipeline pgroup: detach so the child cannot SIGTTIN/SIGTTOU the + /// host. This is the case that regressed before this fix and is the + /// motivating bug for PR #895. + #[test] + fn embedded_host_with_non_terminal_stdin_detaches() { + assert_eq!(child_session_action(false, false, false), ChildSessionAction::DetachSession,); + } + + /// **Pipeline carve-out.** Non-interactive brush, non-terminal stdin + /// (pipe), and a multi-command pipeline: MUST NOT detach. For the first + /// external stage, `setsid()` puts the process-group leader into a + /// different session, so later stages fail to join its group with + /// EPERM. For later stages, `setsid()` would either fail with EPERM or + /// move the child into a new session, breaking the pipeline's shared + /// process group and job-control signal propagation. + #[test] + fn pipeline_stage_does_not_detach() { + assert_eq!(child_session_action(false, false, true), ChildSessionAction::None,); + } + } + + /// End-to-end verification that brush, when embedded as a non-interactive + /// library (`interactive: false`, exactly what `create_session` produces), + /// spawns external commands in a **separate session** from the host. + /// + /// The truth-table tests in `child_session_action` cover the decision in + /// isolation. This test covers the wiring: it boots a real `BrushShell`, + /// runs a child that prints its PID then sleeps, and asks the kernel for + /// that PID's session via `getsid(2)` while the child is still alive. + /// Pre-fix (`new_pg=false` skipped `detach_session`), the child inherited + /// the host's session, so `getsid(child_pid) == getsid(0)`. Post-fix, + /// `setsid` ran and the child is its own session leader + /// (`getsid(child_pid) == child_pid`). + #[cfg(unix)] + #[tokio::test(flavor = "multi_thread")] + async fn embedded_external_command_runs_in_its_own_session() { + use std::io::Read as _; + + // SAFETY: `getsid(0)` only queries the current process session; the return + // value is checked. + let host_sid = unsafe { libc::getsid(0) }; + assert!(host_sid > 0, "getsid(0) failed: {}", std::io::Error::last_os_error()); + + // Build the same kind of session pi-natives uses in production. + let config = ShellConfig { session_env: None, snapshot_path: None, minimizer: None }; + let mut session = create_session(&config).await.expect("create_session"); + + // Output pipe shared between the brush child and a concurrent reader. The + // reader runs on a blocking thread because `os_pipe` reads are blocking. + let (mut reader, writer) = pipe_to_files("e2e").expect("pipe"); + let stdout_file = OpenFile::from(writer.try_clone().expect("clone")); + let stderr_file = OpenFile::from(writer); + + let mut params = session.shell.default_exec_params(); + params.set_fd(OpenFiles::STDIN_FD, null_file().expect("null stdin")); + params.set_fd(OpenFiles::STDOUT_FD, stdout_file); + params.set_fd(OpenFiles::STDERR_FD, stderr_file); + + // (pid_tx, pid_rx) — reader task signals the test as soon as it has the PID. + let (pid_tx, pid_rx) = tokio::sync::oneshot::channel::<i32>(); + let reader_handle = tokio::task::spawn_blocking(move || { + let mut buf = Vec::new(); + // Read just enough to capture the PID line. The child sleeps after + // printing so the pipe will not back-pressure. + let mut chunk = [0u8; 64]; + let mut pid_tx = Some(pid_tx); + while let Ok(n) = reader.read(&mut chunk) + && n > 0 + { + buf.extend_from_slice(&chunk[..n]); + if pid_tx.is_some() + && let Some(line_end) = buf.iter().position(|&byte| byte == b'\n') + && let Ok(line) = std::str::from_utf8(&buf[..line_end]) + && let Ok(pid) = line.trim().parse::<i32>() + { + let _ = pid_tx + .take() + .expect("pid sender should be present") + .send(pid); + } + } + buf + }); + + // Run brush in the background so we can call `getsid(child_pid)` while + // the child is still alive. + let shell_handle = tokio::spawn(async move { + let source_info = SourceInfo::from("pi-natives:test"); + // `printf '%d\n' "$$"` then `sleep 0.5`. Long enough for our `getsid`. + let exec = session + .shell + .run_string("/bin/sh -c 'printf \"%d\\n\" \"$$\"; sleep 0.5'", &source_info, ¶ms) + .await + .expect("run_string"); + drop(params); + (session, exec) + }); + + let child_pid = time::timeout(Duration::from_secs(5), pid_rx) + .await + .expect("timed out waiting for child PID") + .expect("reader closed pid channel without sending"); + assert!(child_pid > 0, "got non-positive child pid: {child_pid}"); + + // Snapshot the child's session ID immediately, while the child is still + // in `sleep`. POSIX guarantees `getsid` against a live PID returns the + // session of that process. + // SAFETY: `child_pid` is a positive PID from the child; errors are reported via + // the checked return value. + let child_sid = unsafe { libc::getsid(child_pid) }; + assert!( + child_sid > 0, + "getsid({child_pid}) failed: {} (child may have already exited)", + std::io::Error::last_os_error(), + ); + + // Drain the brush task and the pipe reader. + let (_session, exec) = time::timeout(Duration::from_secs(5), shell_handle) + .await + .expect("shell timed out") + .expect("shell task panicked"); + assert!( + matches!(exec.exit_code, ExecutionExitCode::Success), + "unexpected exit: {}", + exit_code(&exec), + ); + let _ = time::timeout(Duration::from_secs(2), reader_handle).await; + + assert_ne!( + child_sid, host_sid, + "child PID {child_pid} inherited host session {host_sid}; setsid() did not run — the \ + embedded-host bug is back", + ); + assert_eq!( + child_sid, child_pid, + "child PID {child_pid} should be its own session leader after setsid", + ); + } + + #[tokio::test] + async fn abort_state_signals_cancel_token() { + let abort_state = ShellAbortState::default(); + let mut cancel_token = CancelToken::default(); + let abort_token = cancel_token.emplace_abort_token(); + + abort_state.set(abort_token).await; + abort_state.abort().await; + + let reason = time::timeout(Duration::from_millis(100), cancel_token.wait()) + .await + .expect("cancel token should be signalled"); + assert!(matches!(reason, AbortReason::Signal)); + } + + #[cfg(unix)] + #[tokio::test] + async fn read_output_stops_when_cancelled_before_pipe_eof() { + let (reader, _writer) = pipe_to_files("test").expect("test pipe should be created"); + let cancel = CancellationToken::new(); + let (activity_tx, _activity_rx) = mpsc::channel(1); + let handle = tokio::spawn(read_output(reader, None, cancel.clone(), activity_tx)); + + time::sleep(Duration::from_millis(10)).await; + cancel.cancel(); + + time::timeout(Duration::from_millis(100), handle) + .await + .expect("reader task should stop after cancellation") + .expect("reader task should not panic"); + } +} diff --git a/crates/pi-natives/src/shell/windows.rs b/crates/pi-shell/src/windows.rs similarity index 100% rename from crates/pi-natives/src/shell/windows.rs rename to crates/pi-shell/src/windows.rs diff --git a/docs/tools/ask.md b/docs/tools/ask.md new file mode 100644 index 000000000..ae9ce9c35 --- /dev/null +++ b/docs/tools/ask.md @@ -0,0 +1,85 @@ +# ask + +> Prompts the interactive user for one or more choices or free-form answers. + +## Source +- Entry: `packages/coding-agent/src/tools/ask.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/ask.md` +- Key collaborators: + - `packages/coding-agent/src/config/settings-schema.ts` — `ask.timeout` / `ask.notify` defaults + - `packages/coding-agent/src/modes/theme/theme.ts` — checkbox and tree glyphs for TUI rendering + - `packages/coding-agent/src/tui.ts` — status-line rendering + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `questions` | `Question[]` | Yes | One or more questions. Empty arrays are rejected by schema and also guarded at runtime. | + +### `Question` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `id` | `string` | Yes | Stable identifier used in multi-question results. | +| `question` | `string` | Yes | Prompt text shown to the user. | +| `options` | `{ label: string }[]` | Yes | Explicit options. The UI always appends `Other (type your own)`; callers must not include it. | +| `multi` | `boolean` | No | Enables multi-select mode. Default: `false`. | +| `recommended` | `number` | No | Zero-based recommended option index. In single-select mode the label gets ` (Recommended)` appended in the UI. | + +## Outputs +- Single-shot result. +- `content[0].text` is plain text: + - single question: `User selected: ...` and/or `User provided custom input: ...` + - multiple questions: `User answers:` followed by one line per `id` +- `details`: + - single question: `{ question, options, multi, selectedOptions, customInput? }` + - multiple questions: `{ results: QuestionResult[] }`, where each item includes `id`, `question`, `options`, `multi`, `selectedOptions`, and optional `customInput` +- Cancellation and headless cases throw instead of returning a structured success result. + +## Flow +1. `AskTool.createIf()` only registers the tool when `session.hasUI` is true; headless sessions never get it. +2. `execute()` requires `context.ui`; if missing it aborts the context and throws `ToolAbortError("Ask tool requires interactive mode")`. +3. It reads `ask.timeout` from settings, converts seconds to milliseconds, and disables timeout entirely while plan mode is enabled (`packages/coding-agent/src/tools/ask.ts`). +4. If `ask.notify` is not `off`, it sends a terminal notification: `Waiting for input`. +5. For each question, `askSingleQuestion()` drives either: + - single-select list + optional editor for `Other` + - multi-select checkbox loop + `Done selecting` sentinel + optional editor for `Other` +6. In multi-question mode, left/right arrow handlers enable back/forward navigation between questions and preserve prior selections. +7. If a timeout fires before any selection/custom input, the tool auto-selects the recommended option, or the first option when no valid `recommended` index exists. +8. If the user cancels without timeout, `execute()` aborts the tool context and throws `ToolAbortError("Ask tool was cancelled by the user")`. +9. On success it formats human-readable text plus structured `details`; the TUI renderer uses `details` for rich display. + +## Modes / Variants +- Single question: returns flattened `details` fields for one question. +- Multiple questions: returns `details.results[]` and allows back/forward navigation across questions. +- Single-select: one option or custom input. +- Multi-select: toggled checkbox list, `Done selecting` sentinel only when forward navigation is not active. + +## Side Effects +- User-visible prompts / interactive UI + - Opens a selection dialog via `context.ui.select(...)`. + - Opens a text editor dialog via `context.ui.editor(...)` for `Other`. + - Sends a terminal notification unless `ask.notify=off`. +- Session state + - Reads plan-mode state to disable timeouts. + - Calls `context.abort()` on headless use or user cancellation. +- Background work / cancellation + - Wraps UI waits in `untilAborted(...)` so abort signals interrupt pending dialogs. + +## Limits & Caps +- `questions` must contain at least 1 item (`askSchema` in `packages/coding-agent/src/tools/ask.ts`). +- `ask.timeout` default is `30` seconds; `0` disables timeout (`packages/coding-agent/src/config/settings-schema.ts`). +- Prompt guidance says provide 2-5 options, but code does not enforce that (`packages/coding-agent/src/prompts/tools/ask.md`). +- Timeout only applies to the option picker; once the user chooses `Other`, the editor has no timeout (`packages/coding-agent/src/prompts/tools/ask.md`). + +## Errors +- Missing interactive UI: throws `ToolAbortError("Ask tool requires interactive mode")`. +- User cancels picker/editor without timeout: throws `ToolAbortError("Ask tool was cancelled by the user")`. +- Abort signal during input: converted to `ToolAbortError("Ask input was cancelled")`. +- Empty `questions` at runtime returns a text error payload instead of throwing: `Error: questions must not be empty`. + +## Notes +- `recommended` is only a UI hint; invalid indexes are ignored. +- In single-select mode the returned `selectedOptions` value strips the appended ` (Recommended)` suffix. +- Multi-select results preserve selection order by `Set` insertion order, not original option order after arbitrary toggles. +- Option labels and prompt text are returned verbatim in `details`; the tool does not interpret them beyond UI affordances like `Other` and ` (Recommended)`. diff --git a/docs/tools/ast-edit.md b/docs/tools/ast-edit.md new file mode 100644 index 000000000..cc96c42cc --- /dev/null +++ b/docs/tools/ast-edit.md @@ -0,0 +1,121 @@ +# ast_edit + +> Preview and apply structural rewrites over source files via native ast-grep. + +## Source +- Entry: `packages/coding-agent/src/tools/ast-edit.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/ast-edit.md` +- Key collaborators: + - `crates/pi-natives/src/ast.rs` — native rewrite planning and file mutation + - `crates/pi-natives/src/language/mod.rs` — language aliases and extension inference + - `packages/coding-agent/src/tools/path-utils.ts` — path/glob parsing and multi-path resolution + - `packages/coding-agent/src/tools/resolve.ts` — preview/apply queueing + - `packages/coding-agent/src/tools/render-utils.ts` — parse-error dedupe and display caps + - `packages/coding-agent/src/utils/file-display-mode.ts` — hashline vs line-number diff references + - `packages/coding-agent/src/hashline/hash.ts` — stable hashline diff anchors + - `packages/natives/native/index.d.ts` — JS-visible native binding contract + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `ops` | `{ pat: string; out: string }[]` | Yes | One or more rewrite rules. `pat` must be non-empty. Duplicate `pat` values fail before native execution. Empty `out` deletes the matched node. | +| `paths` | `string[]` | Yes | One or more files, directories, globs, or internal URLs with backing files. Empty entries are rejected. Globs are forbidden for internal URLs. | + +Shared AST pattern grammar and language catalog: see [`ast_grep`](./ast-grep.md#inputs). + +- `ast_edit` uses the same `$NAME`, `$_`, `$$$NAME`, and `$$$` metavariable semantics. +- The tool prompt adds rewrite-specific constraints: + - metavariable names must be uppercase and must stand for whole AST nodes, + - captures from `pat` are substituted into `out`, + - each rewrite is a 1:1 structural substitution; one capture cannot expand into multiple sibling nodes unless the grammar itself permits that expansion at that position. + +## Outputs +- Single-shot preview result from `ast_edit` itself. +- Model-facing `content` is one text block showing proposed edits, grouped by file for directory/multi-file runs. + - Each change renders as two lines: `-REF|before` and `+REF|after` in hashline mode, or `-LINE:COLUMN before` / `+LINE:COLUMN after` when hashlines are off. + - Only the first line of each `before`/`after` snippet is shown, truncated to 120 characters in the wrapper. + - `Limit reached; narrow paths.` and formatted parse issues are appended when applicable. +- If no rewrites match, text is `No replacements made` plus formatted parse issues when present. +- `details` includes aggregate preview metadata: + - `totalReplacements`, `filesTouched`, `filesSearched`, `applied`, `limitReached` + - optional `parseErrors`, `scopePath`, `files`, `fileReplacements`, `displayContent`, `meta` +- The tool always previews first (`applied: false` in the direct result). Actual file writes happen only later through `resolve(action: "apply", ...)`. +- When preview produced replacements, `ast_edit` also queues a pending `resolve` action. Successful apply returns a separate `resolve` result, not another `ast_edit` result. + +## Flow +1. `AstEditTool.execute()` validates each op in `packages/coding-agent/src/tools/ast-edit.ts`: + - empty `pat` fails, + - at least one op is required, + - duplicate `pat` values fail, + - ops are converted to a `Record<pattern, replacement>`. +2. The wrapper reads `PI_MAX_AST_FILES` via `$envpos(..., 1000)` and uses that as the native `maxFiles` cap for both preview and apply. +3. Path normalization, internal URL handling, missing-path partitioning, and multi-path resolution follow the same `path-utils.ts` flow as `ast_grep`. +4. The wrapper stats the resolved base path to decide whether to render grouped directory output. +5. `runAstEditOnce(...)` always runs native `astEdit(...)` with `dryRun: true` and `failOnParseError: false` on the first pass. +6. Native `ast_edit` in `crates/pi-natives/src/ast.rs`: + - normalizes the rewrite map and sorts rules by pattern string, + - resolves strictness (`smart` by default), + - collects candidate files from a file or gitignore-aware directory scan, + - infers a single language for the whole call unless `lang` was supplied, + - compiles every rewrite pattern for that language, + - parses each file, skips files with syntax-error trees, collects `replace_by(...)` edits for every match, enforces replacement and file caps, and returns textual before/after slices plus source ranges. +7. The TS wrapper deduplicates parse errors, groups changes by file, and renders preview diff lines. +8. If preview found replacements and `applied` is false, `queueResolveHandler(...)` registers a forced `resolve` action and injects a `resolve-reminder` steering message. +9. On `resolve(action: "apply")`, the queued callback reruns the same rewrite set with `dryRun: false`, recomputes counts, and rejects the apply as an error if the live result no longer matches the preview (`stalePreview`). +10. On a non-stale apply, the callback returns `Applied N replacements in M files.`; on discard, `resolve` returns a discard message without mutating files. + +## Modes / Variants +- Single file: preview or apply against one file. +- Directory + optional glob: native scan walks the directory, then filters by compiled glob. +- Multiple explicit paths/globs: wrapper unions them into one synthetic scope or runs per-target native calls when paths only meet at root. +- Internal URL inputs: only supported when the router resolves them to a backing file path. +- Preview mode: always the direct `ast_edit` tool result. +- Apply mode: only reachable through the queued `resolve` callback after a preview. +- Hashline output mode vs plain line/column mode: controlled by `resolveFileDisplayMode()`. + +## Side Effects +- Filesystem + - Preview reads files and scans directories. + - Apply rewrites files in place with `std::fs::write(...)`, but only when the computed output differs from the original source. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Queues a one-shot forced `resolve` tool choice through `queueResolveHandler(...)`. + - Adds a `resolve-reminder` steering message. +- User-visible prompts / interactive UI + - Direct `ast_edit` results are previews. + - Follow-up apply/discard is exposed through the hidden `resolve` tool. +- Background work / cancellation + - Native preview/apply work runs on a blocking worker via `task::blocking(...)`. + - Cancellation and optional native timeout are cooperative through `CancelToken::heartbeat()`. + +## Limits & Caps +- File cap exposed by the wrapper: `PI_MAX_AST_FILES`, default `1000`, in `packages/coding-agent/src/tools/ast-edit.ts`. +- Native `maxFiles` and `maxReplacements` are both clamped to at least `1` when provided in `crates/pi-natives/src/ast.rs`. +- The wrapper never sets `maxReplacements`; native behavior therefore defaults to effectively unbounded replacements for a run. +- Parse issues are rendered with at most `PARSE_ERRORS_LIMIT = 20` lines in `packages/coding-agent/src/tools/render-utils.ts`; `details.parseErrors` is deduplicated but not capped. +- Directory scans use `include_hidden: true`, `use_gitignore: true`, and skip `node_modules` unless the glob text explicitly mentions `node_modules` in `crates/pi-natives/src/ast.rs`. +- No separate glob-expansion count cap exists. Candidate count is whatever the resolved path/glob expands to after gitignore filtering, then native `maxFiles` stops mutations after the configured number of touched files. +- Preview text truncates each rendered `before` and `after` first line to 120 characters in `packages/coding-agent/src/tools/ast-edit.ts`. + +## Errors +- TS wrapper throws `ToolError` for empty patterns, duplicate rewrite patterns, empty path entries, unsupported internal-URL globs, internal URLs without `sourcePath`, and missing paths. +- Native code returns hard errors for: + - inability to infer one language across all candidates when `lang` is absent, + - unsupported explicit `lang`, + - bad glob compilation or unreadable search roots, + - overlapping computed edits (`Overlapping replacements detected; refine pattern to avoid ambiguous edits`), + - out-of-bounds edit ranges or non-UTF-8 replacement text, + - write failures during apply, + - cancellation or timeout. +- With `failOnParseError: false` (the wrapper always uses this), pattern compile failures and file parse failures become `parseErrors` instead of aborting the whole run. +- If every rewrite pattern fails to compile, native `ast_edit` returns a successful zero-replacement result with `parseErrors` populated. +- Files containing tree-sitter error nodes are skipped for rewriting; they do not get partial edits. +- Apply can fail after a successful preview if the preview becomes stale. The resolve callback compares replacement totals and per-file counts and returns an error result rather than applying a mismatched preview silently. + +## Notes +- `ast_edit` does not expose the native `lang`, `strictness`, `selector`, `maxReplacements`, `failOnParseError`, or `timeoutMs` fields to the model. The runtime fixes the call shape to a preview-first, smart-strictness, best-effort parse mode. +- Because the wrapper does not expose `lang`, mixed-language rewrites only succeed when every candidate infers to the same canonical language. This is stricter than `ast_grep`. +- Idempotency is not enforced syntactically. A rewrite like `foo($A) -> foo($A)` previews zero changes because output equals input; a rewrite that keeps matching its own output may still produce replacements on repeated calls. +- Rewrites are accumulated per file, then applied from the end of the file backward after an overlap check. Independent matches can coexist; overlapping matches abort the run. +- Native rewrite rule order is by pattern-string sort, not by the original `ops` array order, because `normalize_rewrite_map(...)` sorts the `(pattern, rewrite)` pairs. +- Preview/apply parity is validated only by totals and per-file counts, not by a byte-for-byte diff of every replacement payload. \ No newline at end of file diff --git a/docs/tools/ast-grep.md b/docs/tools/ast-grep.md new file mode 100644 index 000000000..97282d6ef --- /dev/null +++ b/docs/tools/ast-grep.md @@ -0,0 +1,112 @@ +# ast_grep + +> Structural code search over supported source files via native ast-grep. + +## Source +- Entry: `packages/coding-agent/src/tools/ast-grep.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/ast-grep.md` +- Key collaborators: + - `crates/pi-natives/src/ast.rs` — native scan, parse, match engine + - `crates/pi-natives/src/language/mod.rs` — language aliases and extension inference + - `packages/coding-agent/src/tools/path-utils.ts` — path/glob parsing and multi-path resolution + - `packages/coding-agent/src/tools/render-utils.ts` — parse-error dedupe and display caps + - `packages/coding-agent/src/tools/match-line-format.ts` — anchor-prefixed match rendering + - `packages/coding-agent/src/utils/file-display-mode.ts` — hashline vs line-number output mode + - `packages/natives/native/index.d.ts` — JS-visible native binding contract + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `pat` | `string` | Yes | Single AST pattern. The wrapper trims it and rejects empty strings. | +| `paths` | `string[]` | Yes | One or more files, directories, globs, or internal URLs with backing files. Empty entries are rejected. Globs are forbidden for internal URLs. | +| `skip` | `number` | No | Match offset. Defaults to `0`, then `Math.floor(...)`; negatives and non-finite values fail. | + +Pattern grammar and language support exposed to the model: +- `$NAME` — capture one AST node. +- `$_` — match one AST node without binding. +- `$$$NAME` — capture zero or more AST nodes; ast-grep stops lazily at the next satisfiable node. +- `$$$` — match zero or more AST nodes without binding. +- Metavariable names must be uppercase and must stand for whole AST nodes, not partial tokens or string fragments. +- Reusing the same metavariable requires identical code at each occurrence. +- Patterns must parse as one valid AST node for the inferred target language. +- Supported canonical languages come from `SupportLang::all_langs()` in `crates/pi-natives/src/language/mod.rs`: `astro`, `bash`, `c`, `cmake`, `cpp`, `csharp`, `dart`, `clojure`, `css`, `diff`, `dockerfile`, `elixir`, `erlang`, `go`, `graphql`, `haskell`, `hcl`, `html`, `ini`, `java`, `javascript`, `json`, `just`, `julia`, `kotlin`, `lua`, `make`, `markdown`, `nix`, `objc`, `ocaml`, `odin`, `perl`, `php`, `powershell`, `protobuf`, `python`, `r`, `regex`, `ruby`, `rust`, `scala`, `solidity`, `sql`, `starlark`, `svelte`, `swift`, `toml`, `tlaplus`, `tsx`, `typescript`, `verilog`, `vue`, `xml`, `yaml`, `zig`. + +## Outputs +- Single-shot tool result. +- Model-facing `content` is one text block: + - grouped by file for directory/multi-file searches, + - match lines rendered as `*LINE+HASH|text` in hashline mode or `*LINE|text` otherwise, + - continuation lines for multi-line matches rendered with a leading space, + - optional `meta: NAME=value` lines when ast-grep captured metavariables. +- If no matches are found, text is `No matches found` or `No matches found. Parse issues mean the query may be mis-scoped; narrow paths before concluding absence.` plus formatted parse issues. +- If the wrapper truncates visible results, the text ends with `Result limit reached; narrow paths or increase limit.` +- `details` includes counts and metadata, not full match payloads: + - `matchCount`, `fileCount`, `filesSearched`, `limitReached` + - optional `parseErrors`, `scopePath`, `files`, `fileMatches`, `displayContent`, `meta` +- Native ranges (`byteStart`, `byteEnd`, `startLine`, `startColumn`, `endLine`, `endColumn`) exist only inside the native result; the wrapper does not emit them directly to the model. + +## Flow +1. `AstGrepTool.execute()` validates `pat`, normalizes `skip`, and normalizes each `paths` entry in `packages/coding-agent/src/tools/ast-grep.ts`. +2. Internal URLs are resolved through `session.internalRouter`; entries without `sourcePath` fail, and internal-URL globs fail early. +3. For multiple path inputs, `partitionExistingPaths()` drops missing bases only when at least one surviving base remains; if all bases are missing the call fails. +4. `parseSearchPath()` splits a single path into `basePath` plus optional `glob`. `resolveExplicitSearchPaths()` collapses multiple inputs into a common base plus a brace-union glob, or separate `targets` when the only common base is a filesystem root. +5. The wrapper stats the resolved base path to decide whether output should be grouped as a directory result. +6. Execution dispatches to either: + - one native `astGrep(...)` call for a single resolved base, or + - `runMultiTargetAstGrep(...)`, which calls the native binding once per target, rebases paths back to the common root, sorts globally, then applies `skip` and the wrapper limit. +7. Native `ast_grep` in `crates/pi-natives/src/ast.rs`: + - normalizes and deduplicates patterns, + - resolves a `MatchStrictness` (`smart` by default), + - collects candidate files from a file or gitignore-aware directory scan, + - infers language per candidate from extension unless `lang` was provided, + - compiles the pattern separately for each language present, + - reads each file, reports syntax-error trees as parse issues, runs `find_all`, and optionally captures metavariable bindings. +8. Native results are sorted by path and source position, then paged by `offset`/`limit`. +9. The TS wrapper normalizes parse-error strings, deduplicates them, groups matches by formatted path, renders anchor lines, appends limit/parse notices, and returns `toolResult(...).text(...).done()`. + +## Modes / Variants +- Single file: native path is the file; output is a flat list of rendered match lines. +- Directory + optional glob: native scan walks the directory, then filters by compiled glob. +- Multiple explicit paths/globs: wrapper unions them into one synthetic scope or runs per-target native calls when paths only meet at root. +- Internal URL inputs: only supported when the router can resolve them to a backing file path. +- Hashline output mode vs plain line-number mode: controlled by `resolveFileDisplayMode()`; hashline mode requires the edit tool and non-raw, mutable sources. + +## Side Effects +- Filesystem + - Stats input paths in the TS wrapper. + - Native code reads matched files and scans directories through `fs_cache`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - None beyond normal tool transcript/result metadata. +- Background work / cancellation + - Native work runs on a blocking worker via `task::blocking(...)`. + - Cancellation and optional native timeout are cooperative through `CancelToken::heartbeat()`. + +## Limits & Caps +- Wrapper-visible result cap: `DEFAULT_AST_LIMIT = 50` in `packages/coding-agent/src/tools/ast-grep.ts`. + - Single-target calls rely on the native default limit of 50 in `crates/pi-natives/src/ast.rs`. + - Multi-target calls fetch `skip + 50 + 1` matches per target, then re-page after global sort. +- Native `limit` is clamped to at least `1`; omitted `offset` defaults to `0` in `crates/pi-natives/src/ast.rs`. +- Parse issues are rendered with at most `PARSE_ERRORS_LIMIT = 20` lines in `packages/coding-agent/src/tools/render-utils.ts`; `details.parseErrors` itself is only deduplicated, not capped. +- Directory scans use `include_hidden: true`, `use_gitignore: true`, and skip `node_modules` unless the glob text explicitly mentions `node_modules` in `crates/pi-natives/src/ast.rs`. +- No hard file-count cap is applied by the wrapper or native `ast_grep`; candidate count is whatever the resolved path/glob expands to after gitignore filtering. +- Multi-path union deduplicates identical path inputs before resolution in `resolveExplicitSearchPaths()`. + +## Errors +- TS wrapper throws `ToolError` for empty patterns, invalid `skip`, empty path entries, unsupported internal-URL globs, internal URLs without `sourcePath`, and missing paths. +- Native code returns hard errors for: + - unsupported explicit `lang`, + - inability to infer language for a candidate when `lang` is not supplied, + - invalid AST pattern compilation for every relevant language, + - unreadable search roots or bad glob compilation, + - cancellation (`Aborted: Signal`) or timeout (`Aborted: Timeout`). +- File-level parse failures and many per-language pattern compile failures are non-fatal: they are accumulated in `parseErrors` and surfaced alongside successful matches. +- `no matches` is not an error, even when parse issues were recorded. + +## Notes +- `pat` is always wrapped into a one-element `patterns` array by the TS tool; the model cannot send multiple patterns through `ast_grep` even though the native binding supports it. +- `ast_grep` can search mixed-language trees because native compilation happens per discovered language, but the prompt still tells the model to keep calls single-language when possible to reduce parse noise. +- Pattern compilation is per language present in the candidate set. One pattern can succeed for some languages and generate per-file parse errors for others in the same run. +- A file with tree-sitter error nodes still gets searched; the syntax warning is additive, not a skip condition. +- For glob semantics, `*.ts` matches only direct children while `**/*.ts` recurses; this is covered by native tests in `crates/pi-natives/src/ast.rs`. +- Output anchors are intended for follow-up tools, but the exact anchor format depends on session edit mode (`hashline` vs line-number mode). \ No newline at end of file diff --git a/docs/tools/bash.md b/docs/tools/bash.md new file mode 100644 index 000000000..843c82067 --- /dev/null +++ b/docs/tools/bash.md @@ -0,0 +1,149 @@ +# bash + +> Execute a shell command in the session workspace, with optional PTY or background-job handling. + +## Source +- Entry: `packages/coding-agent/src/tools/bash.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/bash.md` +- Key collaborators: + - `packages/coding-agent/src/tools/bash-interactive.ts` — PTY/TUI execution path. + - `packages/coding-agent/src/tools/bash-interceptor.ts` — blocks tool-better shell patterns. + - `packages/coding-agent/src/tools/bash-skill-urls.ts` — expands internal URLs to paths. + - `packages/coding-agent/src/exec/bash-executor.ts` — non-PTY shell execution. + - `packages/coding-agent/src/session/streaming-output.ts` — tail buffer, truncation, artifact spill. + - `packages/coding-agent/src/tools/tool-timeouts.ts` — timeout clamp bounds. + - `packages/coding-agent/src/config/settings-schema.ts` — default interceptor rules. + - `docs/bash-tool-runtime.md` — deeper executor/runtime notes; use as the companion doc for shell-session internals. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `command` | `string` | Yes | Shell command text to execute. A leading `cd <path> && ...` is rewritten into `cwd` only when `cwd` was omitted. | +| `env` | `Record<string, string>` | No | Extra environment variables. Keys must match `^[A-Za-z_][A-Za-z0-9_]*$` or the tool throws. Values also go through internal-URL expansion. | +| `timeout` | `number` | No | Timeout in seconds. Default `300`; clamped to `1..3600` by `clampTimeout("bash", ...)`. | +| `cwd` | `string` | No | Working directory, resolved against `session.cwd` via `resolveToCwd`. Must exist and be a directory. | +| `pty` | `boolean` | No | Request PTY mode. Default `false`. PTY is used only when `pty: true`, `PI_NO_PTY !== "1"`, and the tool context has a UI. | +| `async` | `boolean` | No | Background execution request. Present only when `async.enabled` is true for the session. Returns immediately with a job id instead of waiting. | + +## Outputs +The tool returns a single `text` content block plus optional `details`. + +- Success, foreground: + - `content[0].text`: command output, or `(no output)` when the command produced nothing. + - `details.timeoutSeconds`: effective timeout after clamping. + - `details.requestedTimeoutSeconds`: only present when the requested timeout was clamped. + - `details.meta.truncation`: present when output was truncated in memory; includes `artifactId` when full output spilled to an artifact. +- Success, background start (`async: true` or auto-background): + - `content[0].text`: optional preview tail, timeout notice if any, then `Background job <id> started: <label>` with follow-up instructions. + - `details.async`: `{ state: "running", jobId, type: "bash" }`. +- Background progress / completion: + - delivered through `onUpdate` / async job manager, not the initial return. + - running updates contain tail text and `details.async.state: "running"` only after the job is considered backgrounded. + - completion/failure updates carry final text and `details.async.state: "completed" | "failed"`. +- Failure: + - the tool throws `ToolError` / `ToolAbortError`; non-zero exits are surfaced as errors, not success results. + +Stdout and stderr are merged before the model sees them. Non-zero exit codes are appended to the thrown error text as `Command exited with code <n>`. + +## Flow +1. `BashTool.execute()` in `packages/coding-agent/src/tools/bash.ts` reads `command`, normalizes `env`, and defaults `timeout` to `300`. +2. If `cwd` is absent, it rewrites a leading `cd <path> && ...` into the structured `cwd` field and strips that prefix from `command`. +3. If `async: true` is requested while `async.enabled` is off, it throws `ToolError` before any execution. +4. If `bashInterceptor.enabled` is on, `checkBashInterception()` runs against both the original command and the `cd`-stripped command. A matching enabled rule throws before URL expansion or execution. +5. `expandInternalUrls()` rewrites supported internal URLs inside `command`, each `env` value, and protocol-looking `cwd` values. Command/env replacements are shell-escaped unless `noEscape` is requested by the caller path. +6. `resolveToCwd()` resolves `cwd` against `session.cwd`; `fs.stat()` verifies that the target exists and is a directory. +7. `clampTimeout("bash", requestedTimeoutSec)` enforces `TOOL_TIMEOUTS.bash` (`default: 300`, `min: 1`, `max: 3600`). When clamped, `#buildCompletedResult()` / `#buildBackgroundStartResult()` append a notice line. +8. Execution path splits: + 1. `async: true` -> `#startManagedBashJob()` registers a session async job and returns immediately. + 2. Non-PTY with `bash.autoBackground.enabled` and an async job manager -> starts a managed job, waits up to `min(thresholdMs, timeoutMs - 1000)`, and either returns the completed result or converts the run into a background job. + 3. Otherwise runs foreground execution. +9. Foreground non-PTY calls `executeBash()` from `packages/coding-agent/src/exec/bash-executor.ts`. +10. Foreground PTY calls `runInteractiveBashPty()` from `packages/coding-agent/src/tools/bash-interactive.ts`. +11. Both paths allocate an output artifact first when `session.allocateOutputArtifact` is available. The artifact path/id are passed into the sink so large output can spill to disk. +12. `executeBash()` loads shell settings, optional shell snapshot, and shell minimizer settings, then runs via a persistent native `Shell` session or one-shot `executeShell()`. `docs/bash-tool-runtime.md` covers that path in detail. +13. `runInteractiveBashPty()` creates a `PtySession`, overlays an xterm-backed console UI, forwards user key input into the PTY, captures output through `OutputSink`, and kills the PTY on dismiss/dispose. +14. On completion, `#buildCompletedResult()` formats `(no output)` when needed, attaches truncation metadata from the `OutputSink` summary, and re-checks exit status / timeout / cancellation before returning. +15. On non-zero exit, timeout, missing exit status, or cancellation, `#buildResultText()` throws with the captured output included in the error message. + +## Modes / Variants +1. Foreground non-PTY + - Default path. + - Uses `executeBash()`. + - Streams tail-only updates through `streamTailUpdates()` and `TailBuffer(DEFAULT_MAX_BYTES)`. +2. Foreground PTY + - Requires `pty: true`, UI context, and `PI_NO_PTY !== "1"`. + - Uses `runInteractiveBashPty()` and a `PtySession` overlay. + - Supports interactive input; `Esc` kills the session from the overlay. +3. Explicit background job + - Requires `async: true` and `async.enabled`. + - Registers a job with `session.asyncJobManager` and returns `{ state: "running", jobId }` immediately. +4. Auto-backgrounded non-PTY job + - Requires `bash.autoBackground.enabled`, no PTY, and an async job manager. + - Starts like a foreground managed job, then backgrounds it when it outlives the wait window. +5. Intercepted command + - No subprocess created. + - Returns a `ToolError` pointing the model at `read`, `search`, `find`, `edit`, or `write`. + +## Side Effects +- Filesystem + - Validates `cwd` with `fs.stat()`. + - May allocate and write artifact files for full output (`bash`) and minimizer-preserved raw output (`bash-original`). + - `expandInternalUrls(..., { ensureLocalParentDirs: true })` creates parent directories for `local://` paths before execution. +- Subprocesses / native bindings + - Non-PTY uses native shell execution via `@oh-my-pi/pi-natives` (`Shell.run()` or `executeShell()`). + - PTY uses native `PtySession.start()`. +- Session state + - Reads session settings for async, auto-background, interceptor, tool availability, and shell configuration. + - Registers jobs with `session.asyncJobManager` for explicit/auto background runs. + - Uses `session.getSessionId()` to isolate shell reuse and async session keys. + - Uses `session.allocateOutputArtifact()` for spill files. +- User-visible prompts / interactive UI + - PTY mode opens a TUI overlay titled `Console` and forwards input to the PTY. + - Background start messages direct the agent to `job` and `read jobs://<id>`. +- Background work / cancellation + - Async and auto-background jobs continue after the initial tool return. + - Cancellation aborts the native run; PTY overlay dismissal also kills the PTY. + +## Limits & Caps +- Default timeout: `300s` (`TOOL_TIMEOUTS.bash.default` in `packages/coding-agent/src/tools/tool-timeouts.ts`). +- Timeout clamp: `1..3600s` (`TOOL_TIMEOUTS.bash.min/max`). +- Auto-background default threshold: `60_000ms` (`DEFAULT_AUTO_BACKGROUND_THRESHOLD_MS` in `packages/coding-agent/src/tools/bash.ts`), further capped to `timeoutMs - 1000` by `#resolveAutoBackgroundWaitMs()`. +- Hard kill grace beyond requested timeout in non-PTY executor: `5_000ms` (`HARD_TIMEOUT_GRACE_MS` in `packages/coding-agent/src/exec/bash-executor.ts`). +- In-memory output tail cap: `50 * 1024` bytes (`DEFAULT_MAX_BYTES` in `packages/coding-agent/src/session/streaming-output.ts`). Once exceeded, the sink keeps only the tail window in memory. +- Streaming callback throttle in `executeBash()`: `50ms` between `onChunk` calls when streaming is enabled. +- TUI collapsed preview: `10` visual lines (`BASH_DEFAULT_PREVIEW_LINES`) when rendered inline in the agent UI; this is a renderer cap, not a tool output cap. + +## Errors +- Input validation: + - invalid env key -> `ToolError("Invalid bash env name: <key>")`. + - async requested while disabled -> `ToolError("Async bash execution is disabled...")`. + - missing async job manager -> `ToolError("Async job manager unavailable for this session.")`. + - missing/bad `cwd` -> `ToolError("Working directory does not exist: ...")` or `ToolError("Working directory is not a directory: ...")`. +- Interceptor: + - matched command -> `ToolError` with `Blocked: <rule.message>` and the original command. + - invalid interceptor regexes are silently skipped by `compileRules()`. +- Internal URL expansion: + - unsupported scheme, unknown skill, path traversal, missing router support, or router resolution failures all throw `ToolError` from `packages/coding-agent/src/tools/bash-skill-urls.ts`. +- Execution: + - non-zero exit -> thrown `ToolError` containing captured output plus `Command exited with code <n>`. + - missing exit code -> thrown `ToolError` with `Command failed: missing exit status`. + - timeout -> thrown `ToolError`; PTY uses `Command timed out after <n> seconds`, non-PTY executor returns cancelled output that `BashTool` converts to an error. + - user abort -> `ToolAbortError` when the caller signal is aborted. +- Artifact allocation / artifact save failures are swallowed in `saveBashOriginalArtifact()` and `OutputSink.#createFileSink()`; execution continues without that artifact. + +## Notes +- `strict = true` and `concurrency = "exclusive"` are set on `BashTool`; the tool does not run concurrently with another bash tool call in the same session. +- `command` and `env` URL expansions shell-escape replacements; `cwd` expansion uses `noEscape: true` because it becomes a filesystem path argument, not shell text. +- `checkBashInterception()` blocks only when the matching rule's `tool` name is present in `ctx.toolNames`; missing tools disable their corresponding rule. +- Default interceptor rules come from `DEFAULT_BASH_INTERCEPTOR_RULES` in `packages/coding-agent/src/config/settings-schema.ts`: + - `cat|head|tail|less|more` -> `read` + - `grep|rg|ripgrep|ag|ack` -> `search` + - `find|fd|locate` with name/type/glob flags -> `find` + - `sed -i`, `perl -i`, `awk -i inplace` -> `edit` + - `echo|printf|cat <<` with redirection -> `write` +- PTY mode is ignored in non-UI contexts and when `PI_NO_PTY=1`; the tool silently falls back to non-PTY execution. +- Non-PTY runs merge `NON_INTERACTIVE_ENV` with `env`; PTY runs also prepend `NON_INTERACTIVE_ENV` before custom env values. +- When the shell minimizer rewrites output inside `executeBash()`, the visible output is replaced with minimized text and a `[raw output: artifact://<id>]` footer may be appended if `onMinimizedSave` persisted the original text. +- The TUI renderer parses partial JSON to recover `env` assignments early in streaming previews; that behavior is display-only. +- For executor internals that are not tool-specific — shell session reuse keys, snapshots, prefix handling, and native timeout behavior — see `docs/bash-tool-runtime.md`. diff --git a/docs/tools/browser.md b/docs/tools/browser.md new file mode 100644 index 000000000..89639a4e8 --- /dev/null +++ b/docs/tools/browser.md @@ -0,0 +1,229 @@ +# browser + +> Open, reuse, close, and script Puppeteer tabs against headless Chromium or CDP-attached apps. + +## Source +- Entry: `packages/coding-agent/src/tools/browser.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/browser.md` +- Key collaborators: + - `packages/coding-agent/src/tools/browser/tab-supervisor.ts` — global tab registry; worker lifecycle; run/close coordination. + - `packages/coding-agent/src/tools/browser/tab-worker.ts` — executes `run` code; implements the `tab` helper API. + - `packages/coding-agent/src/tools/browser/tab-worker-entry.ts` — worker-thread transport bootstrap. + - `packages/coding-agent/src/tools/browser/registry.ts` — browser-handle registry keyed by browser kind. + - `packages/coding-agent/src/tools/browser/launch.ts` — Puppeteer loading, Chromium resolution/download, headless launch, stealth injection. + - `packages/coding-agent/src/tools/browser/attach.ts` — CDP attach/reuse, target picking, spawned-app process handling. + - `packages/coding-agent/src/tools/browser/tab-protocol.ts` — worker init/run/result message schema. + - `packages/coding-agent/src/tools/browser/readable.ts` — `tab.extract()` readability extraction. + - `packages/coding-agent/src/tools/browser/render.ts` — TUI rendering for `open`/`close` status lines and `run` JS cells. + - `packages/coding-agent/src/tools/puppeteer/00_stealth_tampering.txt` — mask patched functions/descriptors as native. + - `packages/coding-agent/src/tools/puppeteer/01_stealth_activity.txt` — synthesize visibility/focus/scroll activity. + - `packages/coding-agent/src/tools/puppeteer/02_stealth_hairline.txt` — fix Modernizr hairline detection. + - `packages/coding-agent/src/tools/puppeteer/03_stealth_botd.txt` — spoof `navigator.webdriver`, `window.chrome`, and Chrome fingerprint surfaces. + - `packages/coding-agent/src/tools/puppeteer/04_stealth_iframe.txt` — patch iframe `contentWindow`/`srcdoc` behavior. + - `packages/coding-agent/src/tools/puppeteer/05_stealth_webgl.txt` — spoof WebGL vendor/renderer/precision. + - `packages/coding-agent/src/tools/puppeteer/06_stealth_screen.txt` — normalize screen/viewport/device-pixel-ratio values. + - `packages/coding-agent/src/tools/puppeteer/07_stealth_fonts.txt` — spoof local fonts and perturb canvas text rendering. + - `packages/coding-agent/src/tools/puppeteer/08_stealth_audio.txt` — spoof audio latency/sample-rate and perturb offline rendering. + - `packages/coding-agent/src/tools/puppeteer/09_stealth_locale.txt` — force locale/languages/timezone/date strings. + - `packages/coding-agent/src/tools/puppeteer/10_stealth_plugins.txt` — synthesize `navigator.plugins`/`navigator.mimeTypes`. + - `packages/coding-agent/src/tools/puppeteer/11_stealth_hardware.txt` — spoof `navigator.hardwareConcurrency`. + - `packages/coding-agent/src/tools/puppeteer/12_stealth_codecs.txt` — spoof media codec support. + - `packages/coding-agent/src/tools/puppeteer/13_stealth_worker.txt` — carry UA/platform spoofing into `Worker`/`SharedWorker`. + +## Inputs + +### Shared fields + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `action` | `"open" \| "close" \| "run"` | Yes | Dispatches to the open/close/run path. | +| `name` | `string` | No | Tab id. Defaults to `"main"`. Tabs live in a process-global map, so the same name is reused across later calls and in-process subagents until closed. | +| `timeout` | `number` | No | Tool wall-clock timeout in seconds. Defaults to `30`; clamped to the browser tool range before execution. | + +### `action: "open"` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `url` | `string` | No | Navigate after the tab is ready. Existing reusable tabs also navigate when `url` is supplied. | +| `viewport` | `{ width: number; height: number; scale?: number }` | No | Requested viewport. For headless launch this becomes the initial viewport; for a page it is applied with `page.setViewport()`. `scale` maps to Puppeteer `deviceScaleFactor`. | +| `wait_until` | `"load" \| "domcontentloaded" \| "networkidle0" \| "networkidle2"` | No | Navigation wait condition. Defaults to `"networkidle2"` where omitted. | +| `dialogs` | `"accept" \| "dismiss"` | No | Installs a page `dialog` handler that auto-accepts or auto-dismisses dialogs. Omitted means no handler. | +| `app` | `{ path?: string; cdp_url?: string; args?: string[]; target?: string }` | No | Selects browser kind. No `app` uses the session `browser.headless` setting. `app.path` is resolved against the session cwd and used as the executable path for spawn/attach reuse. `app.cdp_url` connects to an existing CDP endpoint. `args` are appended only when spawning `app.path`. `target` is only used for attached/spawned-app page selection. | + +### `action: "close"` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `all` | `boolean` | No | Close every known tab. Omitted closes only `name`. | +| `kill` | `boolean` | No | When a tab release drops a spawned-app browser handle to refcount 0, also terminate its process tree. Has no effect on headless shutdown and only disconnects connected CDP browsers. | + +### `action: "run"` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `code` | `string` | Yes | Async-function body executed in a VM context with `page`, `browser`, `tab`, `display`, `assert`, `wait`, `console`, timers, `URL`, `TextEncoder`, `TextDecoder`, and `Buffer` in scope. | + +## Outputs +The tool returns one result per call; no streaming partial output is emitted from the browser implementation itself. + +- `open`: text content with `Opened` or `Reused`, browser description, URL, and optional title. `details` includes `action`, `name`, `browser`, `url`, `viewport`, and the same text in `details.result`. +- `close`: text content with either `Closed ...` or `No tab named ...`. `details` includes `action`, `name`, and `details.result`. +- `run`: ordered `content` array built as: + 1. every `display(value)` call in execution order, + 2. final return value, JSON-stringified unless already a string, + 3. or `Ran code on tab "..."` if nothing else was produced. +- `display(value)` coercion in `packages/coding-agent/src/tools/browser/tab-worker.ts`: + - `{ type: "image", data: string, mimeType: string }` becomes image content, + - `string` becomes text content, + - other values become pretty JSON text when serializable, else `String(value)`. +- `tab.screenshot()` also appends text plus an image content item unless `silent: true`; `details.screenshots` records persisted screenshot metadata `{ dest, mimeType, bytes, width, height }`. +- `run` `details` includes `action`, `name`, current `browser`/`url` when the tab exists, optional `screenshots`, and `details.result` containing only the concatenated text outputs. + +## Flow +1. `BrowserTool.execute()` (`packages/coding-agent/src/tools/browser.ts`) abort-checks, clamps `timeout` via `clampTimeout("browser", ...)`, defaults `name` to `"main"`, and dispatches on `action`. +2. `open` resolves browser kind with `resolveBrowserKind()`: + - `app.cdp_url` → `{ kind: "connected" }` after trimming trailing slashes. + - `app.path` → `{ kind: "spawned" }` after resolving against session cwd. + - otherwise → `{ kind: "headless", headless: session.settings.get("browser.headless") }`. +3. `open` rejects reusing the same tab name across different browser kinds (`sameBrowserKind()`); callers must close first. +4. `open` acquires a browser handle through `acquireBrowser()` (`packages/coding-agent/src/tools/browser/registry.ts`): + - existing connected handle is reused by browser-kind key; + - stale disconnected handles are disposed and recreated; + - headless launches via `launchHeadlessBrowser()`; + - `connected` waits for `${cdpUrl}/json/version`, then `puppeteer.connect()`; + - `spawned` first tries `findReusableCdp()`, else kills same-path processes, allocates a free loopback port, spawns the executable with `--remote-debugging-port=<port>`, waits for CDP, then connects. +5. `open` acquires a tab through `acquireTab()` (`packages/coding-agent/src/tools/browser/tab-supervisor.ts`): + - same-name + same-browser + alive tab is reused unless `dialogs` changed; + - same-name but different browser handle, dead state, or changed dialog policy forces release and recreation; + - reusing with a new `url` navigates by issuing `await tab.goto(...)` through the worker. +6. New tabs build a `WorkerInitPayload` in `buildInitPayload()`: + - headless mode sends `url`, `waitUntil`, `viewport`, `dialogs`, and timeout; + - attach mode resolves a page with `pickElectronTarget()`, gets its target id, and sends `targetId` plus `dialogs`. +7. `acquireTab()` spawns a dedicated Bun `Worker` from `tab-worker-entry.ts`; if that fails it falls back to inline execution in the main thread (`spawnInlineWorker()`), preserving behavior but losing protection against synchronous infinite loops. +8. `WorkerCore.#init()` (`packages/coding-agent/src/tools/browser/tab-worker.ts`) connects back to the browser websocket endpoint. Headless mode opens a new page, applies stealth patches, applies viewport, installs dialog handling if requested, and optionally navigates. Attach mode resolves the requested target page and optionally installs dialog handling. +9. On success the worker sends `ready` with `{ url, title, viewport, targetId }`; the supervisor stores a `TabSession`, increments browser-handle refcount with `holdBrowser()`, and keeps the tab in a process-global `Map<string, TabSession>`. +10. `run` requires non-empty `code`, looks up the tab with `getTab()`, then delegates to `runInTab()`. +11. `runInTabWithSnapshot()` rejects dead tabs and concurrent runs (`Tab ... is busy`), captures session cwd plus optional `browser.screenshotDir`, registers an abort hook, sends a `run` message to the worker, and races the result against `timeoutMs + 750` ms. Timeouts force-kill the tab worker and, for headless tabs, close the orphaned page target. +12. `WorkerCore.#run()` creates a VM context, exposes the raw Puppeteer `page`/`browser` plus a synthetic `tab` API, and executes `(async () => { ...code... })()` via `vm.runInContext()`. +13. The `tab` helper API implemented in `#createTabApi()` is: + - `tab.name: string` + - `tab.page: Page` + - `tab.signal?: AbortSignal` + - `tab.url(): string` + - `tab.title(): Promise<string>` + - `tab.goto(url, { waitUntil? })` + - `tab.observe({ includeAll?, viewportOnly? })` + - `tab.screenshot({ selector?, fullPage?, save?, silent? })` + - `tab.extract(format = "markdown")` + - `tab.click(selector)` + - `tab.type(selector, text)` + - `tab.fill(selector, value)` + - `tab.press(key, { selector? })` + - `tab.scroll(deltaX, deltaY)` + - `tab.drag(from, to)` + - `tab.waitFor(selector)` + - `tab.evaluate(fn, ...args)` + - `tab.scrollIntoView(selector)` + - `tab.select(selector, ...values)` + - `tab.uploadFile(selector, ...filePaths)` + - `tab.waitForUrl(pattern, { timeout? })` + - `tab.waitForResponse(pattern, { timeout? })` + - `tab.id(n)` +14. Selector handling in `normalizeSelector()` accepts plain CSS and Puppeteer query handlers, and rewrites legacy Playwright-style prefixes `p-text/`, `p-xpath/`, `p-pierce/`, `p-aria/`; other `p-*` prefixes throw a `ToolError`. +15. `tab.observe()` clears the element cache, takes a Puppeteer accessibility snapshot, filters to interactive nodes unless `includeAll`, optionally filters to viewport-visible nodes, assigns numeric ids, caches `ElementHandle`s, and returns URL/title/viewport/scroll metadata plus `elements`. +16. `tab.id(n)` resolves the cached `ElementHandle`, verifies `el.isConnected`, and throws a stale-id error after cache invalidation if the DOM changed or the cache was cleared. +17. `tab.goto()` clears the cached element ids before navigating. Any new `tab.observe()` also clears and rebuilds the cache. +18. `tab.click()` uses a custom retry loop for `text/...` selectors to find an actionable visible match; other selectors use `page.locator(...).click()` with the run timeout. +19. `tab.screenshot()` captures either the whole page or a selector PNG, downsizes a copy for model output, chooses a persistence path, writes the image to disk, records metadata, and optionally emits text + image display entries. +20. `display()` calls accumulate in an array. After code finishes, the worker posts `{ displays, returnValue, screenshots }`; `BrowserTool.#run()` appends the return value as trailing text content when not `undefined`. +21. `close` releases one tab or all tabs via `releaseTab()` / `releaseAllTabs()`. Each tab aborts pending runs, asks the worker to close, waits up to `750` ms for a `closed` ack, terminates the worker, decrements browser refcount, and disposes the browser handle when refcount reaches zero. + +## Modes / Variants +- **Action dispatch** + - `open` — acquire/reuse browser + tab. + - `close` — release one tab or all tabs. + - `run` — execute JS inside the tab worker. +- **Browser kind** + - **Headless**: launches local Chromium with Puppeteer, applies stealth patches, and creates a fresh page per tab. + - **Spawned app (`app.path`)**: reuses an existing CDP-enabled process for that executable when possible; otherwise kills same-path processes, spawns the executable with remote debugging enabled, then attaches. No stealth patches are injected. + - **Connected browser (`app.cdp_url`)**: attaches to an already-running CDP endpoint. No process ownership; close only disconnects. +- **Target selection for attached/spawned browsers** + - With `app.target`, `pickElectronTarget()` returns the first page whose URL or title contains the case-insensitive substring. + - Without `app.target`, it skips titles/URLs matching `request handler|devtools|background page|background host|service worker` and otherwise falls back to the first page. +- **Worker mode** + - **Dedicated worker**: normal path; user code runs off the main thread and can be aborted even when it blocks synchronously. + - **Inline fallback**: activated when Bun worker spawn fails; behavior matches, but synchronous infinite loops on user code cannot be interrupted. +- **Dialog policy** + - No `dialogs` field: no auto-handler. + - `accept`/`dismiss`: page `dialog` events are handled automatically. + - Changing dialog policy on an existing live tab forces tab recreation instead of mutating the worker in place. +- **Screenshot persistence** + - `save` provided: persist full-resolution PNG at the resolved cwd-relative or absolute path. + - `browser.screenshotDir` session setting set: persist full-resolution PNG under that directory with a timestamped filename. + - Neither set: persist the resized image to a temp-file path under the OS temp dir. + +## Side Effects +- Filesystem + - `loadPuppeteer()` writes `{}` to `<puppeteer-safe-dir>/package.json` before importing `puppeteer-core`. + - First headless launch may download Chromium into the Puppeteer cache directory returned by `getPuppeteerDir()`. + - `tab.screenshot()` creates parent directories and writes image files. + - `tab.uploadFile()` resolves supplied paths against the session cwd. +- Network + - CDP attach paths poll `http://127.0.0.1:<port>/json/version` or the supplied `cdp_url` `/json/version`. + - Headless/browser-attach sessions create CDP websocket connections. + - Headless first-use Chromium download uses `@puppeteer/browsers`. + - User `page` / `tab` operations perform normal browser network traffic. +- Subprocesses / native bindings + - Headless mode launches Chromium through Puppeteer. + - `app.path` mode may spawn the target executable via `Bun.spawn()`. + - `killExistingByPath()` / `gracefulKillTreeOnce()` use `@oh-my-pi/pi-natives` process inspection/termination. + - Worker mode uses Bun `Worker`; fallback mode does not. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Browser handles are cached in a process-global `Map` keyed by browser kind in `packages/coding-agent/src/tools/browser/registry.ts`. + - Tabs are cached in a process-global `Map` keyed by `name` in `packages/coding-agent/src/tools/browser/tab-supervisor.ts`. + - `run` captures session cwd and optional `browser.screenshotDir` for screenshot/save path resolution. + - `restartForModeChange()` drops only headless tabs. +- User-visible prompts / interactive UI + - None beyond normal tool output. Dialog auto-handling is invisible unless it fails and emits debug logs. +- Background work / cancellation + - `open`, `run`, CDP waits, and browser actions thread through abort signals. + - A timed-out `run` aborts the worker execution path and can tear down the tab. + +## Limits & Caps +- Tool timeout clamp: default `30` s, min `1` s, max `30` s (`TOOL_TIMEOUTS.browser` in `packages/coding-agent/src/tools/tool-timeouts.ts`). +- Supervisor grace period around init/run/close: `750` ms (`GRACE_MS` in `packages/coding-agent/src/tools/browser/tab-supervisor.ts`). +- Puppeteer protocol timeout for launch/connect operations: `60_000` ms (`BROWSER_PROTOCOL_TIMEOUT_MS` in `packages/coding-agent/src/tools/browser/launch.ts`). +- Connected-browser CDP readiness wait: `5_000` ms before `puppeteer.connect()` (`packages/coding-agent/src/tools/browser/registry.ts`). +- Spawned-app CDP readiness wait after spawn: `30_000` ms (`packages/coding-agent/src/tools/browser/registry.ts`). +- CDP polling cadence: 150 ms in `waitForCdp()` (`packages/coding-agent/src/tools/browser/attach.ts`). +- Headless default viewport: `1365x768` at `deviceScaleFactor: 1.25` (`DEFAULT_VIEWPORT` in `packages/coding-agent/src/tools/browser/launch.ts`). +- Screenshot model-attachment resize cap: `maxWidth 1024`, `maxHeight 1024`, `maxBytes 150 * 1024`, `jpegQuality 70` (`packages/coding-agent/src/tools/browser/tab-worker.ts`). +- `tab.waitForUrl()` polling interval: `200` ms (`packages/coding-agent/src/tools/browser/tab-worker.ts`). +- Drag simulation uses `12` mouse-move steps (`packages/coding-agent/src/tools/browser/tab-worker.ts`). + +## Errors +- `BrowserTool.execute()` converts DOM-style `AbortError` into `ToolAbortError`; other errors propagate. +- `run` hard-fails on missing code: `Missing required parameter 'code' for action 'run'.` +- `open` fails when reusing a name across browser kinds: `Tab "..." is bound to a different browser (...). Close it first.` +- `runInTabWithSnapshot()` fails when the tab is absent/dead (`Tab "..." is not alive. Reopen it.`) or already running (`Tab "..." is busy`). +- Worker init failures and run failures are serialized through `RunErrorPayload`; `ToolError` and abort state are reconstructed on the host side by `errorFromPayload()`. +- Attached-target mismatches surface as: + - `No page targets available on the attached browser` + - `No page target matched "...". Available pages:\n...` + - `Target ... is no longer available on the attached browser` +- Spawned-app path validation requires an absolute executable path, not an app bundle path. +- Spawn/attach failures are wrapped into `ToolError`s such as `Timed out waiting for CDP endpoint ...`, `Failed to attach to ...`, or `Connected to ... but puppeteer.connect failed: ...`. +- `tab` helper errors are user-visible `ToolError`s, including unsupported selector prefix, stale/unknown element id, invalid drag target, missing upload files, non-`<select>` for `tab.select()`, non-file-input for `tab.uploadFile()`, and screenshot selector misses. +- On run timeout, the worker reports `Browser code execution timed out after <ms>ms`; the supervisor may escalate to `Browser code execution hung past grace; tab killed` if the worker does not respond after the grace window. + +## Notes +- `loadPuppeteer()` and `loadPuppeteerInWorker()` temporarily redirect `cwd` to a safe Puppeteer directory before importing `puppeteer-core`, because Puppeteer probes the current working directory during module load. +- Headless launch prefers a detected system Chrome/Chromium, then `PUPPETEER_EXECUTABLE_PATH`, and only then downloads Chromium. +- Headless launch always passes `--no-sandbox`, `--disable-setuid-sandbox`, `--disable-blink-features=AutomationControlled`, and a `--window-size=...` matching the initial viewport. It also ignores Puppeteer default args `--disable-extensions`, `--disable-default-apps`, and `--disable-component-extensions-with-background-pages`. +- Proxy-related env vars only affect headless launch: `PUPPETEER_PROXY`, `PUPPETEER_PROXY_BYPASS_LOOPBACK`, and `PUPPETEER_PROXY_IGNORE_CERT_ERRORS`. +- Stealth patches are applied only in headless mode. Spawned or externally connected browsers are intentionally left untouched. +- `applyStealthPatches()` also strips Puppeteer's `//# sourceURL=__puppeteer_evaluation_script__` suffix from CDP `Runtime.evaluate` / `Runtime.callFunctionOn` payloads. +- `tab.extract()` reads `page.content()`, runs Readability first, then falls back to `main article`/`article`/`main`/`[role='main']`/`body`, and returns `null` if neither extraction path yields content. +- `close(all: true, kill: false)` disconnects from spawned/connected browsers when the last tab closes but leaves spawned app processes running. +- Headless orphan cleanup is best-effort: if a worker dies before closing its page, the supervisor searches browser targets by `targetId` and closes that page. +- Console methods inside `run` do not appear in tool output; they are forwarded as debug/warn/error logs through the worker transport. \ No newline at end of file diff --git a/docs/tools/calc.md b/docs/tools/calc.md new file mode 100644 index 000000000..2f43b7c9a --- /dev/null +++ b/docs/tools/calc.md @@ -0,0 +1,71 @@ +# calc + +> Evaluates one or more arithmetic expressions and returns formatted numeric results. + +## Source +- Entry: `packages/coding-agent/src/tools/calculator.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/calculator.md` +- Key collaborators: + - `packages/coding-agent/src/tui.ts` — status lines and tree-list rendering + - `packages/coding-agent/src/tools/render-utils.ts` — preview limits and formatting helpers + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `calculations` | `Calculation[]` | Yes | Batch of expressions to evaluate in order. | + +### `Calculation` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `expression` | `string` | Yes | Arithmetic expression string. | +| `prefix` | `string` | Yes | Prepended verbatim to the rendered numeric result. | +| `suffix` | `string` | Yes | Appended verbatim to the rendered numeric result. | + +## Outputs +- Single-shot result. +- `content[0].text` is the newline-joined `prefix + value + suffix` string for each calculation. +- `details.results` is an array of `{ expression, value, output }`. +- On renderer fallback, if `details` is missing but `content[0].text` exists, the TUI tries to pair each output line with the original expressions from call args. + +## Flow +1. `execute()` wraps evaluation in `untilAborted(...)`. +2. For each entry, `evaluateExpression(...)` tokenizes the expression, parses it with a recursive-descent parser, rejects non-finite outputs, and normalizes `-0` to `0`. +3. `tokenizeExpression(...)` accepts whitespace, parentheses, operators, and number literals; any other character throws immediately. +4. `ExpressionParser` applies precedence in this order: `+ -`, `* / %`, unary `+ -`, exponentiation `**`, parentheses/literals. +5. Exponentiation is right-associative (`2 ** 3 ** 2` parses as `2 ** (3 ** 2)`). +6. Each numeric result is formatted with `String(value)` and wrapped with the provided `prefix` and `suffix`. +7. The tool returns text output plus structured `details`. + +## Side Effects +- Background work / cancellation + - Supports abort via `untilAborted(...)`. +- Session state + - None. +- Filesystem / Network / Subprocesses + - None. + +## Limits & Caps +- Supported operators: `+`, `-`, `*`, `/`, `%`, `**` (`packages/coding-agent/src/tools/calculator.ts`). +- Supported numeric literals: + - decimal integers/floats, including leading-dot forms like `.5` + - scientific notation like `1e10`, `2.5E-3` + - hexadecimal `0x...` + - binary `0b...` + - octal `0o...` +- Results must be finite; `Infinity` and `NaN` are rejected. +- The renderer collapses long result lists using `PREVIEW_LIMITS.COLLAPSED_ITEMS` from `packages/coding-agent/src/tools/render-utils.ts`. + +## Errors +- Invalid characters: e.g. `Invalid character "x" in expression`. +- Malformed numbers: invalid prefixed literal, invalid exponent, invalid number. +- Syntax errors: `Unexpected token in expression`, `Unexpected end of expression`, `Missing closing parenthesis`, `Expression is empty`. +- Non-finite arithmetic: `Expression result is not a finite number`. +- Any evaluation error aborts the whole batch; the tool does not return partial successes. + +## Notes +- Despite the schema example showing `sqrt(16)`, the parser does not support functions, identifiers, units, or constants; only numeric literals, operators, and parentheses are accepted. +- Precision is plain JavaScript `number` semantics throughout, including floating-point rounding behavior. +- `/` and `%` use JavaScript numeric operators directly; there is no integer-only mode or unit handling. +- Unary operators bind tighter than `*`/`/`/`%` but looser than exponentiation because unary parsing delegates to `#parsePower()`. diff --git a/docs/tools/checkpoint.md b/docs/tools/checkpoint.md new file mode 100644 index 000000000..0dcdbe881 --- /dev/null +++ b/docs/tools/checkpoint.md @@ -0,0 +1,84 @@ +# checkpoint + +> Mark the current top-level conversation state so later `rewind` can collapse exploratory context into a report. + +## Source +- Entry: `packages/coding-agent/src/tools/checkpoint.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/checkpoint.md` +- Key collaborators: + - `packages/coding-agent/src/session/agent-session.ts` — captures the active checkpoint after tool success. + - `packages/coding-agent/src/session/session-manager.ts` — persists the normal session entry stream; not the active checkpoint marker. + - `packages/coding-agent/src/tools/index.ts` — registers the tool and gates it behind `checkpoint.enabled`. + - `packages/coding-agent/src/config/settings-schema.ts` — defines the disabled-by-default feature flag. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `goal` | `string` | Yes | Investigation goal. Required by the TypeBox schema and echoed in the tool result. | + +## Outputs +The tool returns a single text result plus structured details: + +- text body: + - `Checkpoint created.` + - `Goal: <goal>` + - `Run your investigation, then call rewind with a concise report.` +- `details`: + - `goal: string` + - `startedAt: string` — ISO timestamp created inside `CheckpointTool.execute()` + +No checkpoint ID, artifact URI, job handle, file path, or restore token is returned. + +## Flow +1. `CheckpointTool.createIf()` in `packages/coding-agent/src/tools/checkpoint.ts` returns `null` for subagents by checking `session.taskDepth`; only top-level sessions can see the tool. +2. `CheckpointTool.execute()` rejects subagent calls again with `ToolError("Checkpoint not available in subagents.")`. +3. It rejects nested checkpoints with `ToolError("Checkpoint already active.")` when `session.getCheckpointState?.()` is already set. +4. It creates `startedAt = new Date().toISOString()` and returns a normal `toolResult()` payload. The tool itself does not persist anything. +5. On the later `tool_execution_end` event, `AgentSession` in `packages/coding-agent/src/session/agent-session.ts` detects successful `checkpoint` execution and captures three in-memory fields: + - `checkpointMessageCount` — current `agent.state.messages.length`, after the checkpoint tool result has already been appended + - `checkpointEntryId` — `sessionManager.getEntries().at(-1)?.id ?? null`, i.e. the last persisted session entry ID at checkpoint time + - `startedAt` — copied from tool details or regenerated +6. `AgentSession` stores that object in its private `#checkpointState` field and clears `#pendingRewindReport`. + +## Side Effects +- Session state (transcript, memory, jobs, checkpoints, registries) + - Sets `AgentSession.#checkpointState` in memory. + - Records the checkpoint boundary as a message count plus a session entry ID. + - Enables the later yield guard: if a checkpoint is active and no rewind report is pending, `#enforceRewindBeforeYield()` injects a developer-role warning and schedules another turn. +- User-visible prompts / interactive UI + - The tool result tells the model to call `rewind` after the investigation. + - If the agent tries to `yield` first, `AgentSession` injects: + +```text +<system-warning> +You are in an active checkpoint. You MUST call rewind with your investigation findings before yielding. Do NOT yield without completing the checkpoint. +</system-warning> +``` + +## Limits & Caps +- Availability is gated by `checkpoint.enabled`, default `false`, in `packages/coding-agent/src/config/settings-schema.ts`. +- The tool is registered as discoverable in `packages/coding-agent/src/tools/index.ts`. +- Only one active checkpoint is allowed per top-level session. +- Checkpoint state is not persisted as a dedicated session entry. If the process exits, a resumed session can reload the conversation history, but not the live `#checkpointState` guard. +- Session persistence still applies to the ordinary checkpoint tool call message. Global session persistence truncation is `MAX_PERSIST_CHARS = 500_000` in `packages/coding-agent/src/session/session-manager.ts`. + +## Errors +- `ToolError("Checkpoint not available in subagents.")` — thrown for subagent sessions. +- `ToolError("Checkpoint already active.")` — thrown when a prior checkpoint has not been rewound or cleared. +- The tool body has no local `try/catch`; unexpected exceptions propagate. + +## Notes +- Despite the summary string `Create a git-based checkpoint to save and restore session state`, the implementation does not call git and does not snapshot filesystem state. +- Captured state is conversation/session metadata only: + - in-memory message count + - session entry ID in the session tree + - timestamp +- Not captured: + - working tree contents + - staged changes + - artifacts + - blob-store contents + - SQLite history rows from `packages/coding-agent/src/session/history-storage.ts` + - auth or agent records from `packages/coding-agent/src/session/agent-storage.ts` +- If the turn ends with `stopReason === "aborted"` while a checkpoint is active, `AgentSession` clears `#checkpointState` and `#pendingRewindReport` instead of preserving a half-finished checkpoint. diff --git a/docs/tools/debug.md b/docs/tools/debug.md new file mode 100644 index 000000000..dee1293a7 --- /dev/null +++ b/docs/tools/debug.md @@ -0,0 +1,289 @@ +# debug + +> Drive one DAP debug session; adjacent debug UI code reuses the same subsystem for logs, raw SSE capture, reports, profiling, and system diagnostics. + +## Source +- Entry: `packages/coding-agent/src/tools/debug.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/debug.md` +- Key collaborators: + - `packages/coding-agent/src/dap/session.ts` — session lifecycle, breakpoint/state cache + - `packages/coding-agent/src/dap/client.ts` — adapter process/socket transport, DAP message loop + - `packages/coding-agent/src/dap/config.ts` — adapter resolution and auto-selection + - `packages/coding-agent/src/dap/defaults.json` — built-in adapter definitions + - `packages/coding-agent/src/dap/types.ts` — request/response/capability shapes + - `packages/coding-agent/src/tools/tool-timeouts.ts` — per-tool timeout clamp + - `packages/coding-agent/src/debug/index.ts` — interactive debug selector menu + - `packages/coding-agent/src/debug/log-viewer.ts` — recent-log TUI viewer + - `packages/coding-agent/src/debug/raw-sse.ts` — raw SSE TUI viewer + - `packages/coding-agent/src/debug/raw-sse-buffer.ts` — bounded SSE capture buffer + - `packages/coding-agent/src/debug/profiler.ts` — CPU/heap profiling helpers + - `packages/coding-agent/src/debug/report-bundle.ts` — `.tar.gz` report bundling, log source, cache cleanup + - `packages/coding-agent/src/debug/system-info.ts` — system snapshot collection and env redaction + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `action` | `"launch" \| "attach" \| "set_breakpoint" \| "remove_breakpoint" \| "set_instruction_breakpoint" \| "remove_instruction_breakpoint" \| "data_breakpoint_info" \| "set_data_breakpoint" \| "remove_data_breakpoint" \| "continue" \| "step_over" \| "step_in" \| "step_out" \| "pause" \| "evaluate" \| "stack_trace" \| "threads" \| "scopes" \| "variables" \| "disassemble" \| "read_memory" \| "write_memory" \| "modules" \| "loaded_sources" \| "custom_request" \| "output" \| "terminate" \| "sessions"` | Yes | Dispatch key for the tool switch in `packages/coding-agent/src/tools/debug.ts`. | +| `program` | `string` | No | Launch target path. Required for `launch`. Resolved relative to `cwd` if provided, otherwise session cwd. | +| `args` | `string[]` | No | Program argv for `launch`. | +| `adapter` | `string` | No | Explicit adapter name. Otherwise `selectLaunchAdapter()` / `selectAttachAdapter()` auto-pick from `packages/coding-agent/src/dap/config.ts`. | +| `cwd` | `string` | No | Launch/attach working directory. Defaults to session cwd. | +| `file` | `string` | No | Source file path for source breakpoints. | +| `line` | `number` | No | Source line for source breakpoints. | +| `function` | `string` | No | Function breakpoint name. Mutually exclusive with `file`+`line` in breakpoint actions. | +| `name` | `string` | No | Data breakpoint info target name. Required for `data_breakpoint_info`. | +| `condition` | `string` | No | Conditional expression for source/function/instruction/data breakpoints. | +| `hit_condition` | `string` | No | Hit-count condition for instruction/data breakpoints. | +| `expression` | `string` | No | Expression or raw debugger command. Required for `evaluate`. | +| `context` | `string` | No | Evaluate context. Defaults to `"repl"`. Passed through as DAP evaluate context. | +| `frame_id` | `number` | No | Frame selector for `evaluate`, `scopes`, `data_breakpoint_info`. `scopes` and `evaluate` default to the current stopped frame when omitted. | +| `scope_id` | `number` | No | Variables reference from a scope. Accepted by `variables`; also used as a fallback variables reference for `data_breakpoint_info`. | +| `variable_ref` | `number` | No | Variables reference for `variables`; preferred over `scope_id` when both are present. | +| `pid` | `number` | No | Local process id for `attach`. `attach` requires `pid` or `port`. | +| `port` | `number` | No | Remote attach port. If no adapter is forced, attach prefers `debugpy` when `port` is present. | +| `host` | `string` | No | Remote attach host for `attach`. | +| `levels` | `number` | No | Max stack frames for `stack_trace`. | +| `memory_reference` | `string` | No | Memory reference/address for `disassemble`, `read_memory`, `write_memory`. `disassemble` also accepts it via `instruction_reference` fallback logic in `resolveDisassemblyReference()`. | +| `instruction_reference` | `string` | No | Instruction breakpoint reference; required for instruction breakpoint actions. | +| `instruction_count` | `number` | No | Required for `disassemble`. | +| `instruction_offset` | `number` | No | Instruction offset for `disassemble`. | +| `count` | `number` | No | Byte count for `read_memory`. Required there. | +| `data` | `string` | No | Base64 payload for `write_memory`. Required there. | +| `data_id` | `string` | No | Data breakpoint id. Required for `set_data_breakpoint` / `remove_data_breakpoint`. | +| `access_type` | `"read" \| "write" \| "readWrite"` | No | Access filter for `set_data_breakpoint`. | +| `command` | `string` | No | Custom DAP request command. Required for `custom_request`. | +| `arguments` | `Record<string, unknown>` | No | Custom DAP request body for `custom_request`. | +| `offset` | `number` | No | Offset for instruction breakpoints, disassembly, memory read, memory write. | +| `resolve_symbols` | `boolean` | No | `disassemble` symbol-resolution flag. | +| `allow_partial` | `boolean` | No | `write_memory` partial-write allowance. | +| `start_module` | `number` | No | Modules pagination start index for `modules`. | +| `module_count` | `number` | No | Modules pagination count for `modules`. | +| `timeout` | `number` | No | Per-request timeout in seconds. Default `30`, clamped to `5..300`. | + +### Action-specific requirements +- `launch`: `program` +- `attach`: `pid` or `port` +- `set_breakpoint` / `remove_breakpoint`: `function`, or `file` + `line` +- `set_instruction_breakpoint` / `remove_instruction_breakpoint`: `instruction_reference` +- `data_breakpoint_info`: `name` +- `set_data_breakpoint` / `remove_data_breakpoint`: `data_id` +- `evaluate`: `expression` +- `variables`: `variable_ref` or `scope_id` +- `disassemble`: capability `supportsDisassembleRequest`, plus `instruction_count` +- `read_memory`: capability `supportsReadMemoryRequest`, plus `memory_reference` and `count` +- `write_memory`: capability `supportsWriteMemoryRequest`, plus `memory_reference` and `data` +- `modules`: capability `supportsModulesRequest` +- `loaded_sources`: capability `supportsLoadedSourcesRequest` +- `custom_request`: `command` + +### Interactive selector values +`packages/coding-agent/src/debug/index.ts` also exposes a fixed UI-only selector with values `open-artifacts`, `performance`, `work`, `dump`, `memory`, `logs`, `system`, `raw-sse`, `transcript`, `clear-cache`. These are not model-callable through `debugSchema`; they are local TUI menu routes. + +## Outputs +The agent tool returns a standard `toolResult()` payload from `packages/coding-agent/src/tools/debug.ts`: +- `content`: one text block. Every action renders human-readable text; there is no structured JSON block in `content`. +- `details.action`: echoed action. +- `details.success`: always initialized `true`; failures surface by throwing before a result is returned. +- `details.snapshot`: present for actions that operate on or create a session, using `DapSessionSummary` from `packages/coding-agent/src/dap/types.ts`. +- Action-specific `details` fields: + - `launch` / `attach`: `adapter` + - breakpoint actions: `breakpoints`, `functionBreakpoints`, `instructionBreakpoints`, `dataBreakpoints` + - `data_breakpoint_info`: `dataBreakpointInfo` + - `continue` / `step_*`: `state`, `timedOut` + - `threads`: `threads` + - `stack_trace`: `stackFrames` + - `scopes`: `scopes` + - `variables`: `variables` + - `evaluate`: `evaluation` + - `disassemble`: `disassembly` + - `read_memory`: `memoryAddress`, `memoryData`, `unreadableBytes` + - `write_memory`: `bytesWritten` + - `modules`: `modules` + - `loaded_sources`: `sources` + - `custom_request`: `customBody` + - `output`: `output` + - `sessions`: `sessions` + +Streaming/UI behavior: +- The tool renderer merges call and result (`mergeCallAndResult: true`) and renders inline. +- `debug.ts` itself does not emit progress updates through `_onUpdate`; result delivery is single-shot. +- The interactive selector is UI-driven instead of model-driven. It swaps TUI components, appends status lines to the chat pane, opens files in external viewers, or writes archives/temp files. + +Side-channel artifacts outside the model tool result: +- `createReportBundle()` writes `omp-report-<timestamp>.tar.gz` under the reports dir and returns the filesystem path to the UI handler. +- `#handleWorkReport()` writes `/tmp/work-profile-<Date.now()>.svg` before opening it. +- `RawSseViewerComponent` and `DebugLogViewerComponent` can copy captured text to the clipboard. + +## Flow +1. Tool registration is conditional: `DebugTool.createIf()` in `packages/coding-agent/src/tools/debug.ts` returns `null` unless `session.settings.get("debug.enabled")` is true. `packages/coding-agent/src/tools/index.ts` wires the factory and rechecks the same setting in tool filtering. +2. `DebugTool.execute()` clamps `params.timeout` through `clampTimeout("debug", params.timeout)` and composes the caller `AbortSignal` with `AbortSignal.timeout(...)`. +3. `launch` and `attach` resolve cwd/program paths, select an adapter in `packages/coding-agent/src/dap/config.ts`, then delegate to `dapSessionManager.launch()` / `.attach()`. +4. `DapSessionManager.launch()` / `.attach()` enforce the single-session rule with `#ensureLaunchSlot()`, spawn the adapter through `DapClient.spawn()`, register listeners, send `initialize`, cache capabilities, start listening for an initial stop event before sending `launch`/`attach`, then complete the `initialized` → `configurationDone` handshake in `#completeConfigurationHandshake()`. +5. `DapClient.spawn()` starts the adapter detached with `NON_INTERACTIVE_ENV`. Most adapters use stdio; socket-mode adapters (`dlv`) use `#spawnSocketUnix()` on Linux or `#spawnSocketClientAddr()` on macOS/other. +6. `#registerSession()` in `packages/coding-agent/src/dap/session.ts` installs reverse-request handlers: + - `runInTerminal`: spawns the requested debuggee command detached via `ptree.spawn()` and returns `{ processId }` + - `startDebugging`: logs the child-session request and returns `{}`; it does not create nested sessions + - events: `output`, `initialized`, `stopped`, `continued`, `exited`, `terminated` update cached session state +7. Operational actions (`set_breakpoint`, `evaluate`, `threads`, `read_memory`, `custom_request`, and similar) call `dapSessionManager` methods. Most flow through `#sendRequestWithConfig()`, which first sends `configurationDone` when required, then sends the DAP request, then updates `lastUsedAt`. +8. Breakpoint actions maintain local cached breakpoint sets in `DapSessionManager` and remap adapter responses back onto those cached records. +9. `continue` and the three step actions clear cached stop state, subscribe for `stopped`/`terminated`/`exited` before sending the DAP request, then `#awaitStopOutcome()` either returns the new stopped location or reports that the program is still running after timeout. +10. `pause` sends DAP `pause`, waits for a stopped event if needed, and reuses cached stop state if the program was already stopped. +11. `stack_trace`, `scopes`, `variables`, and `evaluate` default to the current stopped thread/frame when the caller omits ids and cached state is available. +12. `output` reads the in-memory output ring from `DapSessionManager.getOutput()`. `terminate` sends `terminate` when supported, always attempts `disconnect`, marks the session terminated, and disposes the client. +13. `sessions` reads the manager’s current map and formats all summaries. Although the manager stores a map, only one active session can exist because new launch/attach calls are blocked until the active one is terminated or cleaned up. +14. The interactive selector in `packages/coding-agent/src/debug/index.ts` builds a `SelectList` of fixed values and dispatches each to a handler: + - `performance`: `startCpuProfile()`, wait for Enter/Escape, stop profiling, read a 30-second work profile with `getWorkProfile(30)`, then bundle via `createReportBundle()` + - `work`: read `getWorkProfile(30)`, write a temp SVG, open it externally + - `dump`: create a report bundle immediately + - `memory`: force GC, call `Bun.generateHeapSnapshot("v8")`, then bundle + - `logs`: build a `DebugLogSource` and mount `DebugLogViewerComponent` + - `raw-sse`: resolve a `RawSseDebugBuffer` from the session and mount `RawSseViewerComponent` + - `system`: call `collectSystemInfo()` and render `formatSystemInfo()` into the chat pane + - `open-artifacts`: open the current session artifact directory if it exists + - `transcript`: delegates to `ctx.handleDebugTranscriptCommand()` + - `clear-cache`: show confirmation, then remove artifact directories older than 30 days with `clearArtifactCache()` + +## Modes / Variants +- **Availability gate** + - Tool hidden when `debug.enabled` is false. +- **Adapter selection** + - `launch`: explicit `adapter` wins; otherwise `selectLaunchAdapter()` ranks available adapters by extension match, root-marker match, then native-debugger preference (`gdb`, `lldb-dap`) for extensionless binaries. + - `attach`: explicit `adapter` wins; otherwise remote `port` prefers `debugpy`, then native debuggers, then first available adapter. +- **Transport** + - stdio adapters: direct `stdin`/`stdout` framing. + - socket adapters: Unix domain socket on Linux; TCP callback on macOS/other. +- **DAP agent-tool actions** + - `launch` — spawn adapter, initialize session, maybe stop on entry; returns formatted session snapshot and `details.adapter`. + - `attach` — connect to a live process or remote port; same output shape as `launch`. + - `set_breakpoint` — source or function breakpoint add/update; returns the current breakpoint list for that target. + - `remove_breakpoint` — source or function breakpoint removal; returns the remaining breakpoint list. + - `set_instruction_breakpoint` / `remove_instruction_breakpoint` — require `supportsInstructionBreakpoints`; return current instruction breakpoint list. + - `data_breakpoint_info` — require `supportsDataBreakpoints`; asks the adapter for a `dataId`, access types, and description for `name`. + - `set_data_breakpoint` / `remove_data_breakpoint` — require `supportsDataBreakpoints`; return the cached data-breakpoint list. + - `continue` / `step_over` / `step_in` / `step_out` — return text describing whether execution stopped, terminated, or kept running, plus `details.state` and `details.timedOut`. + - `pause` — interrupts a running target and returns a stopped snapshot. + - `evaluate` — adapter expression evaluation; defaults context to `repl`. + - `stack_trace` — fetches frames for the resolved thread. + - `threads` — fetches current threads. + - `scopes` — frame scopes for an explicit `frame_id` or the current stopped frame. + - `variables` — variables for `variable_ref` or `scope_id`. + - `disassemble` — require `supportsDisassembleRequest`; disassembles around a memory reference. + - `read_memory` — require `supportsReadMemoryRequest`; returns address, base64 data, unreadable-byte count. + - `write_memory` — require `supportsWriteMemoryRequest`; writes base64 data and reports bytes written. + - `modules` — require `supportsModulesRequest`; optional pagination via `start_module` / `module_count`. + - `loaded_sources` — require `supportsLoadedSourcesRequest`; returns loaded source descriptors. + - `custom_request` — sends any DAP request name with arbitrary arguments. + - `output` — dumps captured stdout/stderr/console text from the session cache. + - `terminate` — disconnects and disposes the active session; returns `No debug session to terminate.` when none exists. + - `sessions` — lists all cached session summaries. +- **Interactive selector routes (UI-only)** + - `logs` — loads today’s log tail and optional older daily log files into `DebugLogViewerComponent`; supports copy, range selection, pid filtering, load-older. + - `raw-sse` — live view over the session’s `RawSseDebugBuffer`; supports tail-follow, scrolling, copy-all. + - `performance` — CPU profile + 30-second work profile + report bundle. + - `memory` — heap snapshot + report bundle. + - `dump` — report bundle without profiler artifacts. + - `work` — standalone work-profile flamegraph export/open. + - `system` — formatted OS/arch/CPU/memory/version/cwd/shell/terminal dump. + - `open-artifacts` / `transcript` / `clear-cache` — artifact directory open, transcript export, artifact-cache pruning. + +## Side Effects +- Filesystem + - Resolves program/file/cwd paths against the session cwd. + - Report creation writes `.tar.gz` bundles and may read the session JSONL, artifact files, subagent session JSONLs, and log files. + - Work-profile export writes `/tmp/work-profile-<timestamp>.svg`. + - Log source reads daily log files from the logs dir. + - Artifact-cache cleanup removes session artifact directories older than the cutoff. + - `resolveRawSseDebugBuffer()` may attach a non-enumerable `rawSseDebugBuffer` property to the owner object. +- Network + - Socket-mode adapters bind/connect local sockets. + - Remote attach may connect through the adapter to a remote debug port. +- Subprocesses / native bindings + - Spawns debugger adapters (`gdb`, `lldb-dap`, `python -m debugpy.adapter`, `dlv`, and others from `defaults.json`) detached. + - Reverse DAP `runInTerminal` requests spawn the debuggee detached via `ptree.spawn()`. + - `getWorkProfile(30)` comes from `@oh-my-pi/pi-natives`. + - CPU profiling uses `node:inspector/promises`; heap snapshots use `Bun.generateHeapSnapshot("v8")`; raw/log viewers sanitize text via `@oh-my-pi/pi-natives`. + - `openPath()` launches the OS default file/browser handler for artifact dirs and SVGs. + - Log/raw-SSE viewers can call `copyToClipboard()`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - `DapSessionManager` keeps session summaries, breakpoints, threads, stack frames, stop location, output capture, capabilities, and last-used timestamps in memory. + - Active-session id is global to the singleton `dapSessionManager`. + - `RawSseDebugBuffer` stores recent SSE events per owner/session. + - The tool is `exclusive`; concurrent debug tool calls are blocked by the scheduler. +- User-visible prompts / interactive UI + - Debug selector shows confirmation before cache deletion. + - Performance profiling temporarily hijacks editor Enter/Escape handlers until profiling stops. + - Log/raw-SSE viewers replace the editor pane with custom components. +- Background work / cancellation + - Every DAP request accepts an `AbortSignal`; timeouts and caller cancellation abort the active request, not the whole session lifetime. + - `DapSessionManager` runs a background cleanup loop every 30 seconds. + - Raw SSE viewers subscribe to buffer updates until closed. + +## Limits & Caps +- Tool timeout clamp: `default=30`, `min=5`, `max=300` in `packages/coding-agent/src/tools/tool-timeouts.ts`. +- Per-request DAP default timeout: `DEFAULT_REQUEST_TIMEOUT_MS = 30_000` in `packages/coding-agent/src/dap/client.ts`. +- Single active session: enforced by `#ensureLaunchSlot()` in `packages/coding-agent/src/dap/session.ts`. +- Idle session cleanup: `IDLE_TIMEOUT_MS = 10 * 60 * 1000`, checked every `CLEANUP_INTERVAL_MS = 30 * 1000`. +- Adapter liveness heartbeat: `HEARTBEAT_INTERVAL_MS = 5 * 1000`. +- Output capture cap: `MAX_OUTPUT_BYTES = 128 * 1024`; older text is trimmed in ~1 KiB slices and `outputTruncated` is recorded. +- Initial stop capture timeout after launch/attach: `STOP_CAPTURE_TIMEOUT_MS = 5_000`. +- Socket-mode adapter readiness timeout: `10_000` ms in `waitForCondition()` and TCP connect timeout logic in `packages/coding-agent/src/dap/client.ts`. +- Raw SSE buffer caps in `packages/coding-agent/src/debug/raw-sse-buffer.ts`: + - `MAX_RAW_SSE_EVENTS = 1_000` + - `MAX_RAW_SSE_CHARS = 512_000` + - `MAX_RAW_SSE_EVENT_CHARS = 64_000` per event, with `: omp-debug-truncated ...` marker appended on trim +- Log viewer window in `packages/coding-agent/src/debug/log-viewer.ts`: + - `INITIAL_LOG_CHUNK = 50` + - `LOAD_OLDER_CHUNK = 50` +- Report/log ingestion caps in `packages/coding-agent/src/debug/report-bundle.ts`: + - `MAX_LOG_LINES = 5000` for interactive log reading + - `MAX_LOG_BYTES = 2 * 1024 * 1024` tail-read ceiling + - report bundles include only the last `1000` log lines + - subagent session inclusion is capped at the most recent `10` JSONL files +- Interactive profiling windows in `packages/coding-agent/src/debug/index.ts`: both performance and work reports request `getWorkProfile(30)`. +- Artifact cache pruning default: `30` days in `clearArtifactCache()` and the selector confirmation text. + +## Errors +- Parameter validation in `packages/coding-agent/src/tools/debug.ts` throws `ToolError` with explicit messages such as: + - `program is required for launch` + - `attach requires pid or port` + - `set_breakpoint requires file+line or function` + - `variables requires variable_ref or scope_id` + - `memory_reference is required for read_memory` + - `count is required for read_memory` + - `data is required for write_memory` + - `command is required for custom_request` +- Adapter selection failure throws `No debugger adapter available. Installed adapters: ...`. +- Capability-gated actions throw from `requireCapability(...)`, e.g. `Active adapter does not support memory reads.` +- No-session and state errors come from `DapSessionManager`, e.g. `No active debug session. Launch or attach first.`, `No active stack frame. Run stack_trace first or supply frame_id.`, `Debugger reported no threads.` +- Launching a second live session throws `Debug session <id> is still active. Terminate it before launching another.` +- DAP transport/request failures surface as thrown errors from `DapClient`: + - `DAP request <command> timed out after <ms>ms` + - `DAP event <event> timed out after <ms>ms` + - `DAP adapter <name> is not running` + - `DAP adapter exited (code N): <stderr>` or `DAP adapter exited unexpectedly (code N)` + - adapter response `message` when a DAP request fails +- `continue` / `step_*` are intentionally non-fatal when the target stays running past the timeout: they return `details.timedOut = true` and `state: "running"` instead of throwing. +- `terminate` suppresses adapter errors while sending `terminate`/`disconnect`; it still disposes the client and returns the last summary when possible. +- Interactive selector handlers report UI errors instead of throwing: + - profiler start/stop, report bundling, log reading, system-info collection, cache clearing, and artifact opening use `ctx.showError(...)` / `ctx.showWarning(...)` + - empty logs and empty artifact caches are warnings/status messages, not failures + - copy failures in log/raw-SSE viewers become status/error text in the UI +- Report-bundle helpers are intentionally best-effort for many file reads: missing session files, missing artifact dirs, unreadable artifact files, missing log dirs, inaccessible cache dirs, and missing subagent files are skipped silently. +- `collectSystemInfo()` is best-effort for CPU probing; failure there falls back to `Unknown CPU`. + +## Notes +- `packages/coding-agent/src/prompts/tools/debug.md` tells the model only one active session is supported; that is not advisory, it is enforced in code. +- `configurationDone` is sent automatically both during launch/attach handshake and lazily before later requests if the adapter required it and the initial handshake did not complete. +- `startDebugging` reverse requests are acknowledged but not implemented; child debug sessions are not spawned. +- `output` exposes the merged `output` event stream only; the tool does not distinguish stdout, stderr, and console categories. +- Session summaries expose `needsConfigurationDone`; this is derived from adapter capabilities and whether `configurationDone` has been sent. +- Source breakpoint file paths are normalized with `path.resolve()` before caching and sending to the adapter. +- `evaluate` defaults to `repl`, so the tool can forward raw debugger commands when the adapter supports them. +- `disassemble` resolves its target from `memory_reference` first, then `instruction_reference`; it throws if neither is present. +- `RawSseDebugBuffer.recordEvent()` increments `totalEvents` before bounded retention. A snapshot can therefore show fewer retained records than total observed events. +- Raw SSE buffer listener failures are swallowed so viewer bugs do not break capture. +- `createDebugLogSource()` walks daily log files newest-first, but `loadOlderLogs()` reverses each requested slice before concatenation so older chunks prepend in chronological order. +- `clearArtifactCache()` deletes directories by directory mtime, not per-file age. +- `addDirectoryToArchive()` reads artifact files as text with `Bun.file(...).text()`. Binary artifact contents are not preserved byte-for-byte in the report bundle. +- The tool renderer truncates displayed output for the TUI preview, but the underlying text result still contains the full returned string. diff --git a/docs/tools/edit.md b/docs/tools/edit.md new file mode 100644 index 000000000..97e2aa4c0 --- /dev/null +++ b/docs/tools/edit.md @@ -0,0 +1,205 @@ +# edit + +> Applies source edits; default mode is the hashline patch language consumed from a single `input` string. + +## Source +- Entry: `packages/coding-agent/src/edit/index.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/hashline.md` +- Key collaborators: + - `packages/coding-agent/src/utils/edit-mode.ts` — selects active edit mode + - `packages/coding-agent/src/hashline/grammar.lark` — custom-tool grammar for hashline mode + - `packages/coding-agent/src/hashline/input.ts` — splits `@PATH` sections + - `packages/coding-agent/src/hashline/parser.ts` — parses ops and payload lines + - `packages/coding-agent/src/hashline/apply.ts` — validates anchors and applies edits + - `packages/coding-agent/src/hashline/anchors.ts` — stale-anchor mismatch formatting + - `packages/coding-agent/src/hashline/recovery.ts` — cache-based stale-anchor recovery + - `packages/coding-agent/src/hashline/hash.ts` — computes `LINEhh|` anchors shared with `read`/`search` + - `packages/coding-agent/src/edit/file-read-cache.ts` — per-session read snapshot cache + - `packages/coding-agent/src/tools/read.ts` — emits anchored lines and records read snapshots + - `packages/coding-agent/src/tools/search.ts` — records sparse snapshots from matches/context + - `packages/coding-agent/src/tools/fs-cache-invalidation.ts` — invalidates FS scan caches after writes + - `packages/coding-agent/src/edit/streaming.ts` — computes in-flight diff previews for the TUI + +## Inputs + +### Hashline mode (default) + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `input` | `string` | Yes | One or more edit sections. First non-blank line must be `@PATH` unless the caller supplies the legacy fallback `path` outside the model schema and the body already looks like hashline ops (`packages/coding-agent/src/hashline/input.ts`). Optional `*** Begin Patch` / `*** End Patch` envelope is ignored if present. | + +Patch language inside `input`: + +- Section header: `@PATH` +- Insert after: `+ ANCHOR` +- Insert before: `< ANCHOR` +- Delete range: `- A..B` +- Replace range: `= A..B` +- Payload line: `~TEXT` by default; separator is `HL_EDIT_SEP` and can be overridden once at process start by `PI_HL_SEP` (`packages/coding-agent/src/hashline/hash.ts`) +- Special anchors: `BOF`, `EOF` +- Anchor token: `<line><2-char-hash>`, for example `41th` + +Anchors come from `read`/`search` output. `read` formats lines as `LINEhh|TEXT` via `formatHashLine` / `formatHashLines` in `packages/coding-agent/src/hashline/hash.ts`; copy only the token left of `|` into op lines. + +Other edit modes exist (`replace`, `patch`, `vim`, `apply_patch`) and are selected outside the tool payload by `resolveEditMode()` in `packages/coding-agent/src/utils/edit-mode.ts`. Their schemas are different; this document covers the default hashline mode. + +## Outputs +- Single-shot tool result; hashline mode does not use a `resolve` preview/apply handshake. +- `content` contains one text block per call. For a successful single-file edit it is either: + - `<path>:` plus a compact diff preview from `packages/coding-agent/src/hashline/diff-preview.ts`, or + - `Updated <path>` / `Created <path>` when no compact preview text is emitted. +- Parse or recovery warnings are appended as: + +```text +Warnings: +... +``` + +- `details` is `EditToolDetails` from `packages/coding-agent/src/edit/renderer.ts`: + - `diff`: unified diff string + - `firstChangedLine`: first changed post-edit line + - `diagnostics`: LSP/format result if available + - `op`: `"create"` or `"update"` for hashline mode + - `meta`: output metadata + - `perFileResults`: present for multi-section input +- Multi-section input returns one aggregated result with combined text and per-file details. +- While the model is still typing arguments, the TUI can compute a diff preview with `packages/coding-agent/src/edit/streaming.ts`; that preview is not a deferred action and does not block execution. + +## Flow +1. `EditTool.execute()` in `packages/coding-agent/src/edit/index.ts` resolves the active mode. Default is `hashline`; `customFormat` exposes `packages/coding-agent/src/hashline/grammar.lark` with `$HFMT$` / `$HSEP$` placeholders filled from `packages/coding-agent/src/hashline/hash.ts`. +2. `executeHashlineSingle()` in `packages/coding-agent/src/hashline/execute.ts` splits the raw `input` into `@PATH` sections with `splitHashlineInputs()`. +3. If multiple sections target the same path, `mergeSamePathSections()` concatenates them before execution so every op still refers to the original file snapshot. +4. Multi-section calls run a preflight pass (`preflightHashlineSection()`): parse ops, enforce plan-mode write rules, load the current file, reject anchor-scoped edits against missing files, reject auto-generated files, apply edits in memory, and fail if the result is a no-op. This prevents partial batches. +5. `parseHashlineWithWarnings()` in `packages/coding-agent/src/hashline/parser.ts` tokenizes the diff body: + - ignores blank lines and optional `*** Begin Patch` + - stops at `*** End Patch` + - stops at `*** Abort` and emits `ABORT_WARNING` + - turns `+` / `<` payload runs into one `insert` edit per payload line + - turns `- A..B` into one `delete` edit per line in the range + - turns `= A..B` into inserts before `A`, then deletes for `A..B`; no payload means replace with a single empty line +6. `applyHashlineEdits()` in `packages/coding-agent/src/hashline/apply.ts` validates every referenced anchor before mutating anything. Each anchor hash is recomputed from current file content with `computeLineHash()`. +7. If any anchor hash differs, `applyHashlineEdits()` throws `HashlineMismatchError`. `execute.ts` catches only that class and calls `tryRecoverHashlineWithCache()`. +8. Recovery replays the edits against the most recent cached read/search snapshot for that path (`packages/coding-agent/src/edit/file-read-cache.ts`), then 3-way merges the result onto current disk content using `Diff.applyPatch(..., { fuzzFactor: 3 })` in `packages/coding-agent/src/hashline/recovery.ts`. On success the edit proceeds with a warning; on failure the original mismatch error is re-thrown. +9. Before splicing lines, `absorbReplacementBoundaryDuplicates()` normalizes some malformed-but-recoverable ranges: + - duplicate prefix/suffix lines adjacent to a replacement can be absorbed by widening the delete range + - pure inserts can auto-drop duplicated leading/trailing payload lines when `edit.hashlineAutoDropPureInsertDuplicates` is enabled + - all such fixes append warnings +10. `after_anchor` inserts are normalized to `before_anchor` of the next line, or `EOF` if the anchor was the last line. +11. Anchor-targeted edits are bucketed by target line and applied bottom-up so earlier splices do not invalidate later original line numbers. `BOF` and `EOF` inserts are applied after that. +12. The edited text is restored to the original BOM and line ending style with helpers from `packages/coding-agent/src/edit/normalize.ts` and persisted via `serializeEditFileText()` in `packages/coding-agent/src/edit/read-file.ts`. +13. The writethrough callback from `createLspWritethrough()` may format the file and fetch diagnostics. Late diagnostics are queued back into session state as a hidden deferred message by `EditTool.#injectLateDiagnostics()` in `packages/coding-agent/src/edit/index.ts`. +14. `invalidateFsScanAfterWrite()` calls `invalidateFsScanCache(path)` so filesystem-backed tools do not serve stale scan results. +15. The session file-read cache is refreshed with the post-edit file text via `recordContiguous()`, making the just-written content the new recovery base for subsequent stale-anchor merges. +16. The final response is built from a unified diff (`generateDiffString()`), a compact preview, and any accumulated warnings. + +## Modes / Variants +- `hashline` — default mode; line-anchored patch language described here (`packages/coding-agent/src/utils/edit-mode.ts`). +- `replace` — exact/fuzzy old/new text replacement (`packages/coding-agent/src/edit/modes/replace.ts`). +- `patch` — structured JSON diff-hunk mode (`packages/coding-agent/src/edit/modes/patch.ts`). +- `apply_patch` — freeform Codex-style `*** Begin Patch` envelope, internally expanded into patch-mode entries (`packages/coding-agent/src/edit/modes/apply-patch.ts`). +- `vim` — persistent modal editing buffer (`packages/coding-agent/src/tools/vim.ts`). + +Hashline op examples: + +```text +@src/a.ts ++ 4fb +~const added = true; +``` + +```text +@src/a.ts +< 4fb +~const addedBefore = true; +``` + +```text +@src/a.ts +- 4fb..6qx +``` + +```text +@src/a.ts += 4fb..5dm +~const clean = (name || DEF).trim(); +~return clean.length === 0 ? DEF : clean.toUpperCase(); +``` + +BOF/EOF examples: + +```text +@src/a.ts ++ BOF +~const HEADER = true; +``` + +```text +@src/a.ts ++ EOF +~export const done = true; +``` + +## Side Effects +- Filesystem + - Reads target files with `readEditFileText()`. + - Writes full updated file contents with `serializeEditFileText()`. + - Preserves BOM and original line-ending style. +- Subprocesses / native bindings + - `createLspWritethrough()` may trigger formatter / diagnostics work through the LSP subsystem. + - `invalidateFsScanAfterWrite()` calls native `invalidateFsScanCache()` from `@oh-my-pi/pi-natives`. +- Session state + - Reads and updates the per-session `FileReadCache` used for stale-anchor recovery. + - Stores pending deferred-diagnostics abort controllers per path inside `EditTool`. + - Queues late diagnostics back into the session transcript as a hidden custom message. +- Background work / cancellation + - A new edit to the same path aborts the prior deferred diagnostics fetch for that path (`packages/coding-agent/src/edit/index.ts`). + - The tool itself is marked `nonAbortable = true` and `concurrency = "exclusive"` in `packages/coding-agent/src/edit/index.ts`. + +## Limits & Caps +- Default mode is `hashline` (`DEFAULT_EDIT_MODE`) in `packages/coding-agent/src/utils/edit-mode.ts`. +- Anchor hashes are always 2 lowercase letters from a stable 647-entry bigram table (`HL_BIGRAMS_COUNT`) in `packages/coding-agent/src/hashline/hash.ts`. +- The visible mismatch report shows 2 lines of context on each side (`MISMATCH_CONTEXT`) in `packages/coding-agent/src/hashline/constants.ts`. +- Stale-anchor recovery uses `fuzzFactor: 3` (`HASHLINE_RECOVERY_FUZZ_FACTOR`) in `packages/coding-agent/src/hashline/recovery.ts`. +- The per-session read cache keeps at most 30 paths (`MAX_PATHS_PER_SESSION`) in `packages/coding-agent/src/edit/file-read-cache.ts`. +- Hashline streaming chunk defaults are 200 lines or 64 KiB per chunk (`packages/coding-agent/src/hashline/types.ts`, consumed by `packages/coding-agent/src/hashline/stream.ts`). +- `HL_EDIT_SEP` defaults to `~`; `HL_BODY_SEP` is always `|` (`packages/coding-agent/src/hashline/hash.ts`). + +## Errors +- Missing section header: + - `input must begin with "@PATH" on the first non-blank line; got: ... Example: "@src/foo.ts" then edit ops.` +- Empty header: + - `Input header "@" is empty; provide a file path.` +- Bad anchor token: + - `line N: expected a full anchor such as "119sr"; got "...".` +- Bad range syntax: + - `line N: explicit ranges are required for delete/replace...` + - `line N: range must include exactly two full anchors separated by "..".` + - `line N: range A..B ends before it starts.` + - `line N: range A..B uses two different hashes for the same line.` +- Missing payload for `+` / `<`: + - `line N: + and < operations require at least one ~TEXT payload line.` +- Stray payload line: + - `line N: payload line has no preceding +, <, or = operation.` +- Unknown op: + - `line N: unrecognized op. Use < ANCHOR..., + ANCHOR..., - A..B..., = A..B...` +- Missing file for anchor-scoped edits: + - `File not found: <path>` +- Out-of-range anchor: + - `Line N does not exist (file has M lines)` +- Stale anchors throw `HashlineMismatchError`. The error message contains re-read guidance and reprints nearby current file lines as `LINEhh|TEXT`; mismatched lines are marked `*`. `displayMessage` renders the same information in a code-frame style. +- No-op edit: + - `Edits to <path> resulted in no changes being made.` +- Recovery failure is silent internally: if cache-based merge cannot prove a valid result, the original mismatch error is surfaced unchanged. + +## Notes +- `read` and `search` are the authoritative source of anchors. The edit parser does not want the trailing `|TEXT`; copy only the `LINEhh` token. +- Multi-op patches are parsed against the original file snapshot. Do not renumber later anchors after earlier ops; `applyHashlineEdits()` buckets and applies them bottom-up. +- `= A..B` is not a primitive replace in the parser. It expands to inserts before `A` plus deletes for `A..B`, which is why stale-anchor checking still happens on the original range lines. +- Interior lines of a multi-line range use hash `**` (`RANGE_INTERIOR_HASH`) and are not individually verified; only the first and last anchor hashes are checked. +- `computeLineHash()` trims trailing whitespace before hashing. Anchors survive line-ending changes and trailing-space-only changes, but not substantive line edits. +- For punctuation-only lines, the hash mixes in the line number; identical `}` lines on different lines intentionally get different anchors. +- `splitHashlineInputs()` normalizes absolute `@PATH` headers back to a cwd-relative path when the file is inside the current working tree. +- Optional `*** Begin Patch` / `*** End Patch` markers are accepted in hashline mode, but the file sections are still `@PATH`-based, not Codex `*** Update File:` hunks. +- `*** Abort` terminates parsing early and returns `ABORT_WARNING`; ops parsed before the marker still apply. +- File-read cache invalidation is conflict-based, not write-through invalidation. If `read` later records content for a line that disagrees with the cached snapshot, the entire snapshot for that path is replaced with the newly observed lines (`packages/coding-agent/src/edit/file-read-cache.ts`). +- There is no resolve-style apply/discard phase for hashline edits. The only preview path is the transient TUI diff preview in `packages/coding-agent/src/edit/streaming.ts`. diff --git a/docs/tools/eval.md b/docs/tools/eval.md new file mode 100644 index 000000000..f10227954 --- /dev/null +++ b/docs/tools/eval.md @@ -0,0 +1,240 @@ +# eval + +> Execute Python or JavaScript code in persistent cell-based runtimes. + +## Source +- Entry: `packages/coding-agent/src/tools/eval.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/eval.md` +- Key collaborators: + - `packages/coding-agent/src/eval/parse.ts` — lenient cell parser + - `packages/coding-agent/src/eval/sniff.ts` — language sniffing heuristics + - `packages/coding-agent/src/eval/backend.ts` — backend execution contract + - `packages/coding-agent/src/eval/js/index.ts` — JS backend adapter + - `packages/coding-agent/src/eval/js/executor.ts` — JS execution + output sink + - `packages/coding-agent/src/eval/js/context-manager.ts` — persistent VM contexts, prelude, tool bridge + - `packages/coding-agent/src/eval/js/prelude.txt` — JS global helpers + - `packages/coding-agent/src/eval/py/index.ts` — Python backend adapter + - `packages/coding-agent/src/eval/py/executor.ts` — kernel session retention, reset, cleanup + - `packages/coding-agent/src/eval/py/kernel.ts` — Jupyter gateway/kernel protocol, display capture + - `packages/coding-agent/src/eval/py/prelude.py` — Python helper functions and status events + - `packages/coding-agent/src/session/streaming-output.ts` — truncation, artifacts, streamed chunks + - `docs/python-repl.md` — Python kernel/gateway internals + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `input` | `string` | Yes | Cell program text. Parsed by `parseEvalInput()` in `packages/coding-agent/src/eval/parse.ts`, not by JSON subfields. | + +`input` syntax accepted at runtime: + +- Cell header: `*** Begin <LANG>`; parser accepts `PY`, `PYTHON`, `IPY`, `IPYTHON`, `JS`, `JAVASCRIPT`, `TS`, `TYPESCRIPT` case-insensitively. +- Optional attributes immediately after the header, first occurrence wins: + - `*** Title: ...` + - `*** Timeout: <n>[ms|s|m]` (default 30s) + - `*** Reset` +- Cell body: every following line until `*** End ...`, the next `*** Begin ...`, or `*** Abort`. + +Leniencies in `packages/coding-agent/src/eval/parse.ts`: + +- Markers accept two or more leading `*` and flexible whitespace. +- `*** End` does not need to repeat the language token. +- Missing end markers between adjacent cells are tolerated; the next `*** Begin` closes the prior cell. +- Bare code or a single markdown fence such as ```` ```py ```` is treated as one implicit cell. +- If `*** Abort` appears, the in-progress cell is dropped and the result carries an abort warning. + +The tool also exposes a custom Lark grammar from `packages/coding-agent/src/eval/eval.lark` for constrained sampling. That grammar is stricter than the runtime parser: it only advertises `PY` / `JS` / `TS` headers and an `*** End Cell` closer. + +## Outputs + +Final result from `EvalTool.execute()` is single-shot, but `onUpdate` streams partial text and `details` while cells run. + +Returned shape: + +- `content`: one text block containing combined cell output, or `(no text output)` / `(no output)` when only rich outputs exist. +- `details` (`EvalToolDetails` from `packages/coding-agent/src/eval/types.ts`): + - `cells`: per-cell code, status (`pending`/`running`/`complete`/`error`), output, duration, exit code, status events, markdown flag + - `language`: first backend used + - `languages`: distinct backends used, in first-use order + - `jsonOutputs`: structured values emitted via `display(...)` + - `images`: image payloads emitted by Python rich display or JS `display({ type: "image", ... })` + - `statusEvents`: aggregated helper/tool status events + - `notice`: backend fallback notice + - `meta`: truncation metadata + - `isError`: set on cell failure or cancellation + +Renderer behavior in `packages/coding-agent/src/tools/eval.ts`: + +- call preview renders parsed code cells with syntax highlighting +- result view renders each cell separately, including status, duration, and output +- markdown outputs are rendered with the Markdown component instead of plain text +- `jsonOutputs` render as a tree, collapsed or expanded depending on UI state +- timeout / fallback / truncation notices render as dim metadata lines +- images are carried in `details.images`; generic tool UI image handling renders them outside the text block + +Side-channel artifacts: + +- `session.allocateOutputArtifact?.("eval")` may allocate an `artifact://...` backing store for spilled output. +- Truncated output metadata points at that artifact when available. + +## Flow + +1. `EvalTool.execute()` in `packages/coding-agent/src/tools/eval.ts` parses `params.input` with `parseEvalInput()`. +2. `parseEvalInput()` normalizes newlines, collects cells, parses attributes, and assigns each cell a language from the header, language sniffing, or the default `python`. +3. Back in `execute()`, each parsed cell is resolved to a backend with `resolveBackend()`: + - explicit `python`/`js` requests are validated against session settings and backend availability + - otherwise `sniffEvalLanguage()` in `packages/coding-agent/src/eval/sniff.ts` tries shebangs and language markers + - if no explicit language was present, later cells prefer the previous runtime language before re-sniffing + - Python is preferred when available; JS is the fallback when Python is unavailable or disabled +4. The tool allocates an `OutputSink`, a `TailBuffer`, per-cell result objects, and a `sessionAbortController`. `session.trackEvalExecution?.(...)` can wrap the whole run for external cancellation tracking. +5. Cells execute sequentially. For each cell, `execute()`: + - clamps the cell timeout through `clampTimeout("eval", ...)` + - builds a combined abort signal from the tool signal, the timeout, and the session abort controller + - marks the cell `running` and emits an update + - calls the backend’s `execute()` with `cwd`, `sessionId`, `sessionFile`, `kernelOwnerId`, `deadlineMs`, `reset`, artifact info, and chunk callback +6. JS cells dispatch through `packages/coding-agent/src/eval/js/index.ts` into `executeJs()`; Python cells dispatch through `packages/coding-agent/src/eval/py/index.ts` into `executePython()`. +7. Backend text chunks stream into the shared `OutputSink`; rich outputs are accumulated separately as JSON, images, markdown markers, and status events. +8. After each cell: + - text output is trimmed and stored on that cell result + - multi-cell runs prefix text with `[i/n]` and the optional title + - cancellations return early with `isError: true` and a cell-specific abort message + - non-zero exit codes return early with `isError: true` and a message naming the failed cell + - later cells are skipped after the first error, but earlier cell state persists in the underlying runtime +9. On success, the tool joins all cell outputs, synthesizes `(no text output)` or `(no output)` when needed, and attaches truncation metadata from `summarizeFinal()`. +10. The renderer uses `details.cells`, `details.jsonOutputs`, and `details.statusEvents` to build notebook-style output. `mergeCallAndResult = true` and `inline = true`, so call and result render together in the transcript. + +## Modes / Variants + +### Parsing modes + +- Explicit multi-cell format with `*** Begin ...` / `*** End ...` +- Implicit single-cell fallback for bare code or a single fenced block +- Abort-recovery parse path when `*** Abort` is present + +### Backend selection + +- Explicit Python backend +- Explicit JavaScript backend +- Auto-detected backend via `sniffEvalLanguage()` +- Fallback from requested/inferred Python to JS when Python is unavailable +- Fallback notice when JS markers are seen but `eval.js` is disabled and Python is used instead + +### JavaScript runtime + +Implemented in `packages/coding-agent/src/eval/js/context-manager.ts` and `packages/coding-agent/src/eval/js/prelude.txt`. + +- Persistent `vm.Context` instances keyed by `js:${sessionId}` in `vmContexts` +- `*** Reset` calls `resetVmContext(sessionKey)` before the cell executes +- Top-level `await` and bare `return` are supported by wrapping code in an async IIFE when `wrapCode()` sees `await` or `return` +- Top-level static `import ... from ...` is rewritten to `await import(...)` by `rewriteStaticImports()` +- The prelude installs globals: + - `display`, `print` + - `read`, `write`, `append`, `sort`, `uniq`, `counter`, `diff`, `tree`, `env`, `output` + - `tool.<name>(args)` proxy for arbitrary session tool calls +- JS helpers are async because they cross the VM/tool boundary +- `display(value)` behavior: + - plain objects/arrays become JSON outputs + - `{ type: "image", data, mimeType }` becomes an image output + - scalars become text +- The VM exposes a restricted `process` subset plus `Buffer`, `fetch`, `Blob`, `File`, `Headers`, `Request`, `Response`, `fs`, `require`, and browser-style globals +- Per-session VM runs are serialized with `runQueued()` + +### Python runtime + +Implemented in `packages/coding-agent/src/eval/py/executor.ts`, `packages/coding-agent/src/eval/py/kernel.ts`, and `packages/coding-agent/src/eval/py/prelude.py`. See `docs/python-repl.md` for gateway and kernel details. + +- Default mode is retained `session` kernels keyed by `python:${sessionId}` +- Optional `python.kernelMode = "per-call"` creates a fresh kernel for each cell and shuts it down afterward +- `*** Reset` disposes the retained kernel for that session before the cell runs; later Python cells in the same tool call reuse the fresh kernel +- Startup path: + - availability check + - create/connect kernel + - initialize cwd / env / `sys.path` + - execute `PYTHON_PRELUDE` +- Python cells run inside IPython/Jupyter, so top-level `await` works; the prompt warns not to use `asyncio.run(...)` +- The Python prelude defines synchronous helpers with the same surface as JS (except `tool.<name>` exists only in JS) +- `display(value)` wraps dict/list/tuple values in `IPython.display.JSON`; rich display MIME bundles are preserved +- Kernel `display_data` / `execute_result` messages map to: + - `application/x-omp-status` → status event + - `image/png` → image output + - `application/json` → JSON output + - `text/markdown` → markdown output + - `text/plain` → text output + - `text/html` → HTML converted to markdown with `htmlToBasicMarkdown()` +- Interactive stdin is rejected: `input_request` sends an empty reply, marks `stdinRequested`, and the executor returns exit code `1` + +### Multi-language call behavior + +A single tool call can mix Python and JS cells. Persistence is per language runtime: + +- resetting Python does not touch JS state +- resetting JS does not touch Python state +- each backend keeps its own retained session keyed from the same session-derived ID + +## Side Effects + +- Filesystem + - JS/Python prelude helpers can read, write, append, diff, and traverse files under the session cwd or absolute paths. + - Output may spill to an artifact file via `OutputSink`. +- Network + - Python backend talks to a Jupyter kernel gateway over HTTP and WebSocket. + - External gateway mode uses `PI_PYTHON_GATEWAY_URL` and optional `PI_PYTHON_GATEWAY_TOKEN`. + - JS runtime exposes `fetch` and `tool.<name>()`; those tools may perform additional network I/O. +- Subprocesses / native bindings + - Python availability check runs `<python> -c ...`. + - Python backend may start or connect to a kernel gateway; details are in `docs/python-repl.md`. +- Session state + - `session.assertEvalExecutionAllowed?.()` can block execution. + - `session.trackEvalExecution?.(...)` can register cancellable eval work. + - `session.getSessionFile?.()` and `session.getEvalKernelOwnerId?.()` influence kernel reuse and artifact lookup. + - JS VM contexts persist in `vmContexts` across eval calls until reset/disposal. + - Python retained kernels persist in `kernelSessions` until reset, eviction, idle cleanup, or owner cleanup. +- User-visible prompts / interactive UI + - none; stdin requests are rejected programmatically +- Background work / cancellation + - Python retained kernels have heartbeat and idle cleanup timers. + - Cancellation interrupts a running Python kernel and aborts JS promise waits. + +## Limits & Caps + +- Per-cell timeout default: 30s (`DEFAULT_TIMEOUT_MS` in `packages/coding-agent/src/eval/parse.ts`; `TOOL_TIMEOUTS.eval.default` in `packages/coding-agent/src/tools/tool-timeouts.ts`) +- Timeout clamp: 1s minimum, 600s maximum (`TOOL_TIMEOUTS.eval` in `packages/coding-agent/src/tools/tool-timeouts.ts`) +- Transcript code/output preview: 10 lines by default (`EVAL_DEFAULT_PREVIEW_LINES` in `packages/coding-agent/src/tools/eval.ts`) +- Output truncation window: 50KB default (`DEFAULT_MAX_BYTES` in `packages/coding-agent/src/session/streaming-output.ts`) +- Output line cap inside truncation helpers: 3000 lines (`DEFAULT_MAX_LINES` in `packages/coding-agent/src/session/streaming-output.ts`) +- Streaming tail buffer for live updates: `DEFAULT_MAX_BYTES * 2` = 100KB (`packages/coding-agent/src/tools/eval.ts`) +- Python retained kernel idle timeout: 5 minutes (`IDLE_TIMEOUT_MS` in `packages/coding-agent/src/eval/py/executor.ts`) +- Python retained kernel cap: 4 sessions (`MAX_KERNEL_SESSIONS` in `packages/coding-agent/src/eval/py/executor.ts`) +- Python retained kernel cleanup sweep: every 30s (`CLEANUP_INTERVAL_MS` in `packages/coding-agent/src/eval/py/executor.ts`) +- Python owner-cleanup shutdown wait: 2000ms (`OWNER_CLEANUP_KERNEL_SHUTDOWN_TIMEOUT_MS` in `packages/coding-agent/src/eval/py/executor.ts`) +- Python heartbeat interval: 5s (`ensureKernelHeartbeat()` in `packages/coding-agent/src/eval/py/executor.ts`) +- Python external gateway availability check timeout: 5s (`AbortSignal.timeout(5000)` in `packages/coding-agent/src/eval/py/kernel.ts`) +- Python auto-restart budget: one restart per retained session before hard failure (`restartCount > 1` in `packages/coding-agent/src/eval/py/executor.ts`) + +## Errors + +- Parse errors from `parseEvalInput()` throw immediately, for example invalid timeout strings. +- Missing session without proxy executor throws `ToolError("Eval tool requires a session when not using proxy executor")`. +- Disabled/unavailable backends throw `ToolError` from `resolveBackend()`: + - `eval.py = false` + - `eval.js = false` + - Python kernel unavailable + - no backend available +- JS runtime exceptions are converted into text output plus `exitCode: 1`; cancellations return `cancelled: true` and may append `Command timed out`. +- Python execution errors from the kernel become text output and `exitCode: 1`; later cells are skipped. +- Python stdin requests are treated as errors with the message `Kernel requested stdin; interactive input is not supported.` +- Cancellation is returned, not thrown, once backend execution has started. The tool formats it as a cell failure and sets `details.isError = true`. +- If parsing encountered `*** Abort`, the final text appends `ABORT_WARNING`, explicitly telling the model that earlier cells ran and state persists. +- If output truncates, the tool still succeeds; truncation is surfaced through `details.meta` and artifact-backed full output when available. + +## Notes + +- The runtime parser is intentionally more permissive than `packages/coding-agent/src/eval/eval.lark`; maintain both when changing syntax. +- Cell language in `ParsedEvalCell` is not the last word: `EvalTool.execute()` may override backend selection for cells without an explicit header by inheriting the previous runtime language. +- `tool.<name>()` exists only in JS. Python prelude helpers do not call back into the full tool registry. +- JS helper paths reject protocol URIs (`://`) in `resolvePath()`; the JS prelude is filesystem-only unless the code calls `tool.read(...)` or another tool explicitly. +- Python helper `output(...)` depends on `PI_SESSION_FILE`; it fails outside a session-backed run. +- `display()` can produce text and structured outputs from the same value; the renderer prefers markdown over `text/plain` when both exist. +- JS static imports are rewritten only at top level. Nested imports stay invalid and surface normal JS syntax/runtime errors. +- `EvalTool` is `concurrency = "exclusive"`, so eval calls do not overlap within a session. +- The tool description shown to the model is templated by backend availability (`getEvalToolDescription()`); if Python is unavailable, the prompt omits Python-specific instructions. diff --git a/docs/tools/exit_plan_mode.md b/docs/tools/exit_plan_mode.md new file mode 100644 index 000000000..769d69106 --- /dev/null +++ b/docs/tools/exit_plan_mode.md @@ -0,0 +1,68 @@ +# exit_plan_mode + +> Submits the current plan-mode plan for user approval. + +## Source +- Entry: `packages/coding-agent/src/tools/exit-plan-mode.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/exit-plan-mode.md` +- Key collaborators: + - `packages/coding-agent/src/tools/plan-mode-guard.ts` — resolves canonical plan paths during plan mode + - `packages/coding-agent/src/plan-mode/approved-plan.ts` — renames approved plan artifact after user approval + - `packages/coding-agent/src/modes/interactive-mode.ts` — approval popup, plan preview, mode exit, tool restoration + - `packages/coding-agent/src/plan-mode/state.ts` — plan-mode state shape + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `title` | `string` | Yes | Final plan title. `.md` is optional; the runtime normalizes to `local://<title>.md`. Allowed characters: letters, numbers, `_`, `-`. | + +## Outputs +- Single-shot success result with `content[0].text = "Plan ready for approval."`. +- `details` contains: + - `planFilePath` — current plan artifact path from plan-mode state, typically `local://PLAN.md` + - `planExists` — whether that file existed at call time + - `title` — normalized title without `.md` + - `finalPlanFilePath` — normalized destination, always `local://<title>.md` +- The actual rename and mode transition happen later in the interactive controller after the user chooses an approval action. + +## Flow +1. `execute()` reads `session.getPlanModeState()` and rejects the call unless `state.enabled` is true. +2. `normalizePlanTitle()` trims whitespace, rejects empty values, rejects `/`, `\\`, and `..`, appends `.md` if missing, and enforces `^[A-Za-z0-9_-]+\.md$`. +3. The tool computes `finalPlanFilePath = local://<normalized>.md` and resolves both source and destination through `resolvePlanPath(...)` to validate them against plan-mode path rules. +4. It `stat`s the current plan file path; if the plan artifact does not exist it throws a `ToolError` telling the caller to write the finalized plan first. +5. On success it returns the approval-ready payload; it does not mutate files itself. +6. `packages/coding-agent/src/modes/controllers/event-controller.ts` watches successful `exit_plan_mode` results and forwards `details` to `InteractiveMode.handleExitPlanModeTool(...)`. +7. The interactive controller aborts the agent, renders the current plan, and shows four choices: `Approve and execute`, `Approve and keep context`, `Refine plan`, `Stay in plan mode`. +8. If the user approves, `#approvePlan(...)` renames `local://PLAN.md` to `local://<title>.md`, exits plan mode, restores the previous tool set, optionally clears session context, writes the approved plan into the new local root when context is reset, and injects a synthetic system prompt instructing execution from the finalized artifact. + +## Side Effects +- Filesystem + - Tool itself only `stat`s the current plan file. + - Approval path later renames the plan artifact via `fs.rename(...)` and may rewrite the approved plan into a fresh local root with `Bun.write(...)`. +- Session state + - Requires active plan-mode state. + - Approval flow aborts the current agent loop, exits plan mode, restores previous active tools, clears or preserves context depending on the user choice, and records the approved plan reference path. +- User-visible prompts / interactive UI + - Successful calls trigger a plan preview and an approval/refinement selector in interactive mode. +- Background work / cancellation + - The controller aborts the running agent before showing the popup to prevent repeated `exit_plan_mode` calls. + +## Limits & Caps +- `title` accepts only `[A-Za-z0-9_-]` plus optional `.md` (`packages/coding-agent/src/tools/exit-plan-mode.ts`). +- Destination must be under the `local:` scheme; approval rename rejects non-`local:` source or destination paths (`packages/coding-agent/src/plan-mode/approved-plan.ts`). +- In plan mode, only the plan file may be edited; other writes are blocked by `enforcePlanModeWrite(...)` in `packages/coding-agent/src/tools/plan-mode-guard.ts`. + +## Errors +- Plan mode inactive: throws `ToolError("Plan mode is not active.")`. +- Empty title: throws `ToolError("Title is required and must not be empty.")`. +- Path traversal / separators: throws `ToolError("Title must not contain path separators or '..'.")`. +- Invalid characters: throws `ToolError("Title may only contain letters, numbers, underscores, or hyphens.")`. +- Missing plan artifact: throws `ToolError("Plan file not found at ... Write the finalized plan ... before calling exit_plan_mode.")`. +- Approval-time failures surface in the UI from `InteractiveMode.handleExitPlanModeTool(...)`, including destination already exists and rename failures from `renameApprovedPlanFile(...)`. + +## Notes +- This tool is hidden/internal: it is injected when `plan.enabled` is on and is not part of normal discoverable built-ins (`packages/coding-agent/src/tools/index.ts`, `packages/coding-agent/src/session/agent-session.ts`). +- The tool returning success does not mean plan mode has ended; it only means the request was handed off to the approval UI. +- `resolvePlanPath(...)` special-cases bare filenames matching the plan basename so `PLAN.md` maps back to the canonical session-scoped `local://PLAN.md` artifact. +- `Approve and keep context` skips the full conversation reset; `Approve and execute` clears context, then copies the approved plan into the new session-local artifact root before execution resumes. diff --git a/docs/tools/find.md b/docs/tools/find.md new file mode 100644 index 000000000..9a83e9c7d --- /dev/null +++ b/docs/tools/find.md @@ -0,0 +1,106 @@ +# find + +> Find filesystem paths by glob; use `search` when you need content matches instead of path matches. + +## Source +- Entry: `packages/coding-agent/src/tools/find.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/find.md` +- Key collaborators: + - `packages/coding-agent/src/tools/path-utils.ts` — normalize inputs; split base path vs glob. + - `packages/coding-agent/src/tools/list-limit.ts` — apply result-count caps. + - `packages/coding-agent/src/session/streaming-output.ts` — truncate text output at byte cap. + - `packages/coding-agent/src/tools/tool-result.ts` — build `content` and `details.meta`. + - `packages/coding-agent/src/tools/output-meta.ts` — encode limit / truncation metadata. + - `packages/coding-agent/src/tools/tool-errors.ts` — map user-facing tool errors. + - `packages/coding-agent/src/tools/index.ts` — register the built-in local implementation. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `paths` | `string[]` | Yes | One or more globs, files, or directories. Empty strings are rejected. Multiple entries may be merged into one brace-union search when their base paths can be resolved together. | +| `hidden` | `boolean` | No | Whether hidden files are included. Defaults to `true` (`hidden ?? true`). | +| `limit` | `number` | No | Max returned paths. Defaults to `1000`. Must be a finite positive number; non-integers are floored. | + +## Outputs +The tool returns a single text block plus structured `details`. + +- Success text: newline-delimited paths, one per line, relative to the session cwd when possible; absolute when outside cwd. Exact file inputs return that file path as one line. +- Empty result text: `No files found matching pattern`. +- Multi-path partial miss: appends `Skipped missing paths: ...` after the result block, or after the empty-result line. +- `details` may include: + - `scopePath`: display form of the searched root or merged roots. + - `fileCount`: number of paths returned after result limiting. + - `files`: returned paths as an array. + - `truncated`: whether result count or byte truncation occurred. + - `resultLimitReached`: reached result limit. + - `missingPaths`: skipped missing inputs in multi-path calls. + - `truncation` / `meta.limits`: structured truncation and limit metadata for renderers. +- Streaming: when the runtime supplies `onUpdate`, the local implementation emits incremental newline-delimited text snapshots during globbing, throttled to 200 ms. + +## Flow +1. `FindTool.execute()` normalizes each `paths` entry with `normalizePathLikeInput()` and `/\\/g -> "/"` (`packages/coding-agent/src/tools/find.ts`). Empty normalized entries fail with `` `paths` must contain non-empty globs or paths ``. +2. For multi-path local calls, `partitionExistingPaths(..., parseFindPattern)` (`packages/coding-agent/src/tools/path-utils.ts`) stats each base path. Missing entries are skipped; if all are missing, the tool throws `Path not found: ...`. Single missing paths still hard-fail. +3. The tool tries `resolveExplicitFindPatterns()` to merge multiple inputs into one search rooted at a common base path. If that does not apply, it parses one input with `parseFindPattern()`. +4. `parseFindPattern()` determines `(basePath, globPattern, hasGlob)`: + - no glob chars (`*`, `?`, `[`, `{`) => search that path with implicit `**/*`. + - glob in the first segment => search from `.` and, unless the pattern already starts with `**/`, prefix it with `**/`. + - glob later in the path => split at the first glob-bearing segment. +5. `resolveToCwd()` converts the base path to an absolute path under the session cwd. A resolved `/` is rejected with `Searching from root directory '/' is not allowed`. +6. `limit` is defaulted to `DEFAULT_LIMIT` (`1000`) and validated as a positive finite integer. `hidden` defaults to `true`. The tool also creates a 5 s timeout via `AbortSignal.timeout(GLOB_TIMEOUT_MS)`. +7. Execution then branches: + - **Custom operations branch**: if `FindToolOptions.operations.glob` exists, the tool checks existence with `operations.exists()`, short-circuits exact-file inputs via `operations.stat()` when available, then calls `operations.glob(globPattern, searchPath, { ignore: ["**/node_modules/**", "**/.git/**"], limit })`. + - **Built-in local branch**: the tool stats `searchPath`. Exact-file inputs return immediately. Directory inputs call `natives.glob()` with `fileType: File`, `hidden`, `maxResults: limit`, `sortByMtime: true`, `gitignore: true`, and the combined abort signal. +8. In the local branch, optional `onMatch` callbacks convert each match to a cwd-relative display path and emit throttled progress updates. +9. After native glob returns, JS sorts `result.matches` by `mtime` descending (`(b.mtime ?? 0) - (a.mtime ?? 0)`) before formatting paths. +10. `buildResult()` applies `applyListLimit()` to cap the array again at `limit`, joins paths with `\n`, then runs `truncateHead()` with `maxLines: Number.MAX_SAFE_INTEGER`. In practice this leaves the 50 KB byte cap in place while disabling the default 3000-line cap. +11. `toolResult()` packages text plus `details`, and records result-limit / truncation metadata for renderers. + +## Modes / Variants +- **Exact file path**: if the parsed input has no glob and the resolved path stats as a file, output is that one path. +- **Directory path**: if the parsed input has no glob and stats as a directory, the tool searches it with implicit `**/*`. +- **Single glob path**: one input parsed by `parseFindPattern()`. +- **Merged multi-path search**: multiple inputs resolved by `resolveExplicitFindPatterns()` into one brace-union glob rooted at a common base path. +- **Partial multi-path search with missing inputs**: local multi-path calls skip missing base paths and surface them as `missingPaths` / `Skipped missing paths: ...`. +- **Custom delegated search**: uses injected `FindOperations` instead of local fs + native glob. + +## Side Effects +- Filesystem + - Stats the resolved base path, and in local multi-path mode stats every candidate base path up front. + - Does not write files. +- Subprocesses / native bindings + - Built-in local mode calls the native `@oh-my-pi/pi-natives` glob implementation. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Emits structured progress updates when `onUpdate` is provided. + - Adds truncation / limit metadata to the tool result. +- Background work / cancellation + - Local globbing is cancellable through the caller abort signal plus an internal 5 s timeout. + +## Limits & Caps +- Default result limit: `1000` (`DEFAULT_LIMIT` in `packages/coding-agent/src/tools/find.ts`). +- Local glob timeout: `5000` ms (`GLOB_TIMEOUT_MS` in `packages/coding-agent/src/tools/find.ts`). +- Output byte cap: `50 * 1024` bytes (`DEFAULT_MAX_BYTES` in `packages/coding-agent/src/session/streaming-output.ts`). +- Default generic line cap in `truncateHead()` is `3000`, but `find` overrides `maxLines` to `Number.MAX_SAFE_INTEGER`, so byte size — not line count — is the practical output truncation cap. +- Streaming update throttle: `200` ms between `onUpdate` emissions. +- Sort order: most recent `mtime` first in the built-in local branch and promised in the prompt. The tool re-sorts in JS even though native glob receives `sortByMtime: true` so native code can still stop early at `maxResults`. + +## Errors +- User-facing `ToolError`s from `FindTool.execute()` include: + - `` `paths` must contain non-empty globs or paths `` + - `Path not found: ...` + - `Searching from root directory '/' is not allowed` + - `Limit must be a positive number` + - `Path is not a directory: ...` + - `find timed out after 5s` +- If the caller aborts, the local branch converts `AbortError` into `ToolAbortError`. +- Non-`ENOENT` stat failures and other unexpected errors are rethrown. +- Empty matches are not errors; they return the no-files text result. + +## Notes +- Reach for `find` for filename / path discovery. Reach for `search` when the selection criterion is file contents or regex matches; `search` takes a `pattern` and returns anchored content matches, while `find` only returns matching paths (`packages/coding-agent/src/prompts/tools/find.md`, `packages/coding-agent/src/prompts/tools/search.md`). +- Bare top-level globs are made recursive. `*.ts` is parsed as base `.` plus glob `**/*.ts`; `src/*.ts` stays rooted at `src` with a non-recursive `*.ts` segment; `src/**/*.ts` preserves explicit recursion. +- `.gitignore` is always enabled in the built-in local branch (`gitignore: true`). There is no model-facing flag to disable it. +- `hidden` defaults to `true`; hidden-file exclusion is opt-out, not opt-in. +- Multi-path missing-input tolerance only applies in the built-in local branch. The custom-operations branch hard-fails the first missing `searchPath` it checks. +- The custom `FindOperations.glob()` hook receives `ignore` and `limit`, but not the `hidden` flag or an explicit `.gitignore` toggle. A remote delegate must account for that itself if it wants parity with the local branch. +- Built-in local globbing asks the native layer for `fileType: File`, so recursive directory searches yield files, not directories. Directory outputs are only possible through exact-path passthrough or custom delegates that return them. diff --git a/docs/tools/github.md b/docs/tools/github.md new file mode 100644 index 000000000..31ebfb301 --- /dev/null +++ b/docs/tools/github.md @@ -0,0 +1,313 @@ +# github + +> Dispatch GitHub CLI operations for repositories, issues, pull requests, search, and Actions run watching. + +## Source +- Entry: `packages/coding-agent/src/tools/gh.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/github.md` +- Key collaborators: + - `packages/coding-agent/src/tools/gh-format.ts` — shorten commit SHAs for summaries. + - `packages/coding-agent/src/tools/gh-renderer.ts` — TUI rendering, especially `run_watch` live/result views. + - `packages/coding-agent/src/utils/git.ts` — `gh`/`git` process wrappers, repo locking, branch config writes. + - `packages/utils/src/dirs.ts` — base directory for dedicated PR worktrees. + - `packages/coding-agent/src/sdk.ts` — session artifact allocation hook. + - `packages/coding-agent/src/session/artifacts.ts` — artifact filename format `<id>.<toolType>.log`. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `op` | `"repo_view" \| "issue_view" \| "pr_create" \| "pr_view" \| "pr_diff" \| "pr_checkout" \| "pr_push" \| "search_issues" \| "search_prs" \| "search_code" \| "search_commits" \| "search_repos" \| "run_watch"` | Yes | Dispatch selector. `GithubTool.execute()` switches only on this field. | +| `repo` | `string` | No | `owner/repo` override. Ignored when the identifier argument is already a full GitHub URL. Required in practice when `gh` cannot infer repo context from the current checkout. | +| `branch` | `string` | No | Used by `repo_view`, `pr_push`, and `run_watch`. `run_watch` falls back to current git branch when `run` is omitted; `pr_push` falls back to current branch. | +| `issue` | `string` | No | Used only by `issue_view`. Required there; accepts an issue number or GitHub issue URL. | +| `pr` | `string \| string[]` | No | Used by `pr_view`, `pr_diff`, `pr_checkout`. Each item may be a PR number, branch name, or GitHub PR URL. Array form enables batching. Omitted means current branch PR. | +| `comments` | `boolean` | No | Used by `issue_view` and `pr_view`. Defaults to `true`. | +| `nameOnly` | `boolean` | No | Used only by `pr_diff`; adds `--name-only`. | +| `exclude` | `string[]` | No | Used only by `pr_diff`; each entry becomes `--exclude <pattern>`. Empty strings are rejected. | +| `force` | `boolean` | No | Used only by `pr_checkout`. Defaults to `false`; allows resetting an existing `pr-<number>` local branch to the PR head commit. | +| `forceWithLease` | `boolean` | No | Used only by `pr_push`; passed through to git push. | +| `title` | `string` | No | Used only by `pr_create`. Required unless `fill` is `true`. | +| `body` | `string` | No | Used only by `pr_create`. Mutually exclusive with `fill`. Empty/omitted body becomes `--body ""` to suppress the interactive editor. Non-empty body is written to a temp file and passed as `--body-file`. | +| `base` | `string` | No | Used only by `pr_create`; passed as `--base`. | +| `head` | `string` | No | Used only by `pr_create`; passed as `--head`. | +| `draft` | `boolean` | No | Used only by `pr_create`. Defaults to `false`. | +| `fill` | `boolean` | No | Used only by `pr_create`. Defaults to `false`. Mutually exclusive with `title` and `body`. | +| `reviewer` | `string[]` | No | Used only by `pr_create`; each entry becomes `--reviewer`. | +| `assignee` | `string[]` | No | Used only by `pr_create`; each entry becomes `--assignee`. | +| `label` | `string[]` | No | Used only by `pr_create`; each entry becomes `--label`. | +| `query` | `string` | No | Used by all `search_*` ops. Required there. | +| `limit` | `number` | No | Used by all `search_*` ops. Defaults to `10`, floored, clamped to `50`, and must be `> 0`. | +| `run` | `string` | No | Used only by `run_watch`. Must be a numeric run ID or full GitHub Actions run URL. | +| `tail` | `number` | No | Used only by `run_watch`. Defaults to `15`, floored, clamped to `200`, and must be `> 0`. | + +## Outputs +The tool returns a single text result built by `buildTextResult()` in `packages/coding-agent/src/tools/gh.ts`. + +- `content`: one text block. Multi-item ops join sections with blank lines and `---` separators. +- `sourceUrl`: set for single repo/issue/PR/run results when a canonical URL is known. +- `details`: optional structured metadata used by the TUI renderer. + - Common fields: `artifactId`, `repo`, `branch`, `worktreePath`, `remote`, `remoteBranch`, `headSha`, `runId`, `runIds`, `status`, `conclusion`, `failedJobs`. + - `pr_checkout` adds `checkouts: GhPrCheckoutSummary[]`. + - `run_watch` adds `watch: GhRunWatchViewDetails`, which drives the custom live/result renderer in `packages/coding-agent/src/tools/gh-renderer.ts`. +- Artifact trailer: when `artifactId` is present, the text body gets an appended line like `Full failed-job logs: artifact://<id>`. + - `run_watch` allocates artifacts with `session.allocateOutputArtifact("github")`; persistent sessions therefore save failed-log bodies as `<artifact-dir>/<id>.github.log`. + +`run_watch` is the only streaming op. It emits `onUpdate` snapshots while polling, then returns one final text result. + +## Flow +1. `GithubTool.createIf()` exposes the tool only when `git.github.available()` finds `gh` on `PATH`. +2. `GithubTool.execute()` wraps dispatch in `untilAborted()` and switches on `params.op`. +3. Each op normalizes optional strings, arrays, booleans, and numeric caps locally in `packages/coding-agent/src/tools/gh.ts`. +4. CLI execution goes through `git.github.run/json/text()` in `packages/coding-agent/src/utils/git.ts`: + - spawns `gh ...` with `Bun.spawn()`; + - trims stdout/stderr unless `trimOutput: false`; + - maps common auth/repo-context failures into tool-facing `ToolError` messages; + - `json()` rejects empty or invalid JSON. +5. Read-style ops (`repo_view`, `issue_view`, `pr_view`, `search_*`) fetch JSON and format Markdown-like text summaries. +6. `pr_view` optionally makes an extra REST call to `/repos/<repo>/pulls/<n>/comments` so inline review comments are included; this is separate from `gh pr view --json`. +7. `pr_diff` fetches raw text from `gh pr diff`; multi-PR batches are handled with `Promise.all()`. +8. `pr_checkout` resolves PR metadata first, then enters `git.withRepoLock()` before any git mutation so parallel checkout calls for the same primary repo do not race on shared `.git` state. +9. `pr_push` reads PR head metadata back from git branch config, derives a refspec, then pushes with `git.push()`. +10. `pr_create` shells out once, then best-effort re-reads the created PR with `gh pr view` for a richer summary. +11. `run_watch` chooses either run mode (`run` supplied) or commit mode (`run` omitted), polls GitHub Actions APIs every 3 seconds, emits streaming updates, and may save a full failed-log artifact before returning. +12. Final text goes through `toolResult().text(...)`; if `session.allocateOutputArtifact()` returns a slot, failed-log text is persisted with `Bun.write()`. + +## Modes / Variants + +### `repo_view` + +| Aspect | Value | +| --- | --- | +| Required fields | `op` | +| Optional fields | `repo`, `branch` | +| `gh` command | `gh repo view [<repo>] [--branch <branch>] --json <GH_REPO_FIELDS>` | +| Batching | None | +| Output | `# <owner/repo>` header, description, URL, default branch, requested branch, visibility, permission, primary language, stars, forks, archive/fork flags, updated timestamp, homepage, topics. `sourceUrl = data.url`. | + +If `repo` is omitted, `gh` repository resolution is used. + +### `issue_view` + +| Aspect | Value | +| --- | --- | +| Required fields | `op`, `issue` | +| Optional fields | `repo`, `comments` | +| `gh` command | `gh issue view <issue> [--repo <repo>] --json <GH_ISSUE_FIELDS or GH_ISSUE_FIELDS_NO_COMMENTS>` | +| Batching | None | +| Output | Single issue summary with metadata, `## Body`, and optional `## Comments`. `sourceUrl = data.url`. | + +`comments: false` switches the requested JSON field set and suppresses comment rendering. + +### `pr_create` + +| Aspect | Value | +| --- | --- | +| Required fields | `op` plus either `fill=true` or `title` | +| Optional fields | `repo`, `title`, `body`, `base`, `head`, `draft`, `fill`, `reviewer[]`, `assignee[]`, `label[]` | +| `gh` command | `gh pr create ...` with flags assembled from provided fields | +| Batching | None | +| Output | `# Created Pull Request ...` summary with URL, state, draft flag, base/head, author, created time, labels, optional body. `sourceUrl` is the created PR URL. | + +Branches: +- `fill && (title || body !== undefined)` throws. +- Non-empty `body` is written under a temp dir `gh-pr-body-*` in `os.tmpdir()`, passed as `--body-file`, then removed in `finally`. +- After creation, the tool parses the returned URL and best-effort runs `gh pr view <number> --repo <repo> --json <GH_PR_FIELDS_NO_COMMENTS>`; failures there are swallowed. + +### `pr_view` + +| Aspect | Value | +| --- | --- | +| Required fields | `op` | +| Optional fields | `repo`, `pr`, `comments` | +| `gh` command | For each requested PR: `gh pr view [<pr>] [--repo <repo>] --json <GH_PR_FIELDS or GH_PR_FIELDS_NO_COMMENTS>` | +| Batching | Yes. `pr` may be `string[]`; each entry is fetched independently with `Promise.all()`. Omitted `pr` means one fetch for current branch PR. | +| Output | Single PR: one PR summary with body, up to 50 files, optional reviews, inline review comments, and issue comments. Batched: `# <n> Pull Requests` plus per-PR sections. | + +When comments are enabled and both repo + PR number are known, the tool paginates `/repos/<repo>/pulls/<n>/comments` with `per_page=100` to supplement `gh pr view` output. + +### `pr_diff` + +| Aspect | Value | +| --- | --- | +| Required fields | `op` | +| Optional fields | `repo`, `pr`, `nameOnly`, `exclude[]` | +| `gh` command | For each requested PR: `gh pr diff [<pr>] [--repo <repo>] --color never [--name-only] [--exclude <glob> ...]` | +| Batching | Yes. Same `pr` normalization as `pr_view`; fetched with `Promise.all()`. | +| Output | Single PR: `# Pull Request Diff` or `# Pull Request Files` followed by raw CLI output. Batched: `# <n> Pull Request Diffs` / `File Lists` plus labeled sections. | + +Diff stdout is preserved without trimming. Empty output becomes `No diff output.` or `No changed files.`. + +### `pr_checkout` + +| Aspect | Value | +| --- | --- | +| Required fields | `op` | +| Optional fields | `repo`, `pr`, `force` | +| `gh` command | For each requested PR: `gh pr view [<pr>] [--repo <repo>] --json <GH_PR_CHECKOUT_FIELDS>`; cross-repo PRs may also call `gh repo view <headRepository> --json <GH_REPO_CLONE_FIELDS>`. | +| Batching | Yes. `pr` may be `string[]`; each PR is resolved in parallel, but git mutations are serialized per primary repo by `git.withRepoLock()`. | +| Output | Single PR: checkout/worktree summary plus `details.repo`, `details.branch`, `details.worktreePath`, `details.remote`, `details.remoteBranch`, `details.checkouts`. Batched: `# <n> Pull Request Worktrees (...)` plus one section per PR and aggregated `details.checkouts`. | + +Worktree and metadata behavior: +- Local branch name is always `pr-<number>`. +- Worktree path is `path.join(getWorktreesDir(), encodeRepoPathForFilesystem(primaryRepoRoot), localBranch)`, where `getWorktreesDir()` is `~/.omp/wt`; effective path is `~/.omp/wt/<encoded-primary-repo-root>/pr-<number>`. +- Existing worktree detection is by branch ref `refs/heads/pr-<number>` from `git.worktree.list()`. +- New worktree creation calls `git.worktree.add(repoRoot, finalWorktreePath, localBranch, { signal })` after verifying the path is neither already registered nor already present on disk. +- For same-repo PRs, remote is `origin`. For cross-repo PRs, the tool resolves a clone URL for the head repo, reuses an existing remote with the same URL when possible, or creates `fork-<owner>` / `fork-<owner>-<n>`. +- The branch push metadata is persisted with `git config` under the repository's shared `.git/config` as: + - `branch.pr-<number>.remote` + - `branch.pr-<number>.merge` + - `branch.pr-<number>.pushRemote` + - `branch.pr-<number>.ompPrHeadRef` + - `branch.pr-<number>.ompPrUrl` + - `branch.pr-<number>.ompPrIsCrossRepository` + - `branch.pr-<number>.ompPrMaintainerCanModify` +- If `refs/heads/pr-<number>` already exists at a different commit, checkout fails unless `force=true`, in which case `git branch --force` resets it to the fetched PR head. +- If a matching worktree already exists, the tool reuses it and reports `reused: true`. + +### `pr_push` + +| Aspect | Value | +| --- | --- | +| Required fields | `op` | +| Optional fields | `branch`, `forceWithLease` | +| `gh` command | None. This path uses git, not `gh`. | +| Batching | None | +| Output | `# Pushed Pull Request Branch` summary with local branch, remote, remote branch, remote URL, PR URL, and force-with-lease flag. `sourceUrl = prUrl` when known. | + +Push target resolution reads the `branch.<name>.ompPrHeadRef`, `pushRemote`/`remote`, `ompPrUrl`, `ompPrMaintainerCanModify`, and `ompPrIsCrossRepository` git-config keys written by `pr_checkout`. If the current checked-out branch matches the target branch, the source ref is `HEAD`; otherwise it pushes `refs/heads/<branch>`. The refspec is `HEAD:refs/heads/<headRef>` or `refs/heads/<branch>:refs/heads/<headRef>`. + +### `search_issues` + +| Aspect | Value | +| --- | --- | +| Required fields | `op`, `query` | +| Optional fields | `repo`, `limit` | +| `gh` command | `gh search issues --limit <limit> --json <GH_SEARCH_FIELDS> [--repo <repo>] -- <query>` | +| Batching | None | +| Output | `# GitHub issues search`, echoed query, optional repo, result count, then one bullet per issue with repo/state/author/labels/timestamps/URL. | + +### `search_prs` + +| Aspect | Value | +| --- | --- | +| Required fields | `op`, `query` | +| Optional fields | `repo`, `limit` | +| `gh` command | `gh search prs --limit <limit> --json <GH_SEARCH_FIELDS> [--repo <repo>] -- <query>` | +| Batching | None | +| Output | Same shape as `search_issues`, labeled as pull requests. | + +### `search_code` + +| Aspect | Value | +| --- | --- | +| Required fields | `op`, `query` | +| Optional fields | `repo`, `limit` | +| `gh` command | `gh search code --limit <limit> --json <GH_SEARCH_CODE_FIELDS> [--repo <repo>] -- <query>` | +| Batching | None | +| Output | `# GitHub code search`, result count, then one bullet per match with path, repo, short commit SHA, URL, and first normalized text-match fragment line when present. | + +### `search_commits` + +| Aspect | Value | +| --- | --- | +| Required fields | `op`, `query` | +| Optional fields | `repo`, `limit` | +| `gh` command | `gh search commits --limit <limit> --json <GH_SEARCH_COMMITS_FIELDS> [--repo <repo>] -- <query>` | +| Batching | None | +| Output | `# GitHub commits search`, result count, then one bullet per commit: short SHA + first commit-message line, repo, author, date, URL. | + +### `search_repos` + +| Aspect | Value | +| --- | --- | +| Required fields | `op`, `query` | +| Optional fields | `limit` | +| `gh` command | `gh search repos --limit <limit> --json <GH_SEARCH_REPOS_FIELDS> -- <query>` | +| Batching | None | +| Output | `# GitHub repositories search`, result count, then one bullet per repo with first description line, language, stars, forks, open issues, visibility, archive/fork flags, updated time, URL. | + +`repo` is intentionally not used for this op. + +### `run_watch` + +| Aspect | Value | +| --- | --- | +| Required fields | `op` | +| Optional fields | `repo`, `branch`, `run`, `tail` | +| `gh` command | Repo resolution: `gh repo view --json nameWithOwner -q .nameWithOwner` when `repo` and run URL repo are both absent. Single-run mode uses `gh api --method GET /repos/<repo>/actions/runs/<runId>` and `gh api --method GET /repos/<repo>/actions/runs/<runId>/jobs`. Commit mode uses `gh api --method GET /repos/<repo>/branches/<branch>`, `gh api --method GET /repos/<repo>/actions/runs`, `gh api --method GET /repos/<repo>/actions/runs/<runId>/jobs`, and `gh api /repos/<repo>/actions/jobs/<jobId>/logs` for failed jobs. | +| Batching | Implicit batching only in commit mode: all workflow runs for one commit are tracked together. | +| Output | Streaming watch snapshots via `onUpdate`, then a final text report. On failure, appends `Full failed-job logs: artifact://<id>` and sets `details.artifactId`. | + +Watch flow: +- `run` parsing accepts either a decimal run ID or a full run URL. URL repo must match explicit `repo` when both are given. +- Poll interval is fixed at 3 seconds (`RUN_WATCH_INTERVAL_DEFAULT`). +- Failure grace period is fixed at 5 seconds (`RUN_WATCH_GRACE_DEFAULT`). When any failed job appears before completion, the tool emits a note, waits once, re-fetches state, then collects logs so concurrent failures are included. +- Failed-job logs are fetched with `gh api /repos/<repo>/actions/jobs/<jobId>/logs` via `git.github.run()`, not `json()`. Non-zero exit leaves `available: false` instead of failing the whole watch. +- Inline result includes only the last `tail` lines per failed job. The saved artifact contains full logs (`mode: "full"`). +- In commit mode, success is intentionally double-checked: once all known runs are successful, the tool waits one more poll interval and succeeds only if the set of run IDs is unchanged. This avoids returning before late workflow runs appear for the same commit. +- `details.watch` drives a specialized renderer in `packages/coding-agent/src/tools/gh-renderer.ts`; non-watch results fall back to generic text rendering. + +## Side Effects +- Filesystem + - `pr_create` may create a temp dir under `os.tmpdir()` named `gh-pr-body-*`, write `body.md`, then remove the dir in `finally`. + - `pr_checkout` may create directories under `~/.omp/wt/<encoded-primary-repo-root>/` and add git worktrees there. + - `run_watch` may write a session artifact with full failed-job logs. +- Network + - Every op shells out to `gh`, which then talks to GitHub APIs except `pr_push`. + - `pr_push` uses git network transport to the configured remote. +- Subprocesses / native bindings + - All `gh` calls use `Bun.spawn(["gh", ...args])`. + - `pr_checkout` and `pr_push` also invoke git helpers from `packages/coding-agent/src/utils/git.ts`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - `run_watch` consumes `session.allocateOutputArtifact()` when failed-job logs are persisted. + - Returned `details` objects carry run/checkouts metadata for the renderer/UI. +- User-visible prompts / interactive UI + - `gh` interactive editor fallback is suppressed for `pr_create` by forcing either `--body-file` or `--body ""`. + - `gh-renderer` provides compact headers for all ops and a custom live watch view for `run_watch`. +- Background work / cancellation + - `run_watch` loops until success/failure and uses `abortableSleep()` between polls. + - `GithubTool.execute()` is wrapped in `untilAborted()`; `git.github.run()` forwards the abort signal into `Bun.spawn()`. + +## Limits & Caps +- Search result default: `10` (`SEARCH_LIMIT_DEFAULT` in `packages/coding-agent/src/tools/gh.ts`). +- Search result max: `50` (`SEARCH_LIMIT_MAX`). +- PR file preview inside `pr_view`: first `50` files only (`FILE_PREVIEW_LIMIT`). +- Run-watch poll interval: `3s` (`RUN_WATCH_INTERVAL_DEFAULT`). +- Run-watch failure grace period: `5s` (`RUN_WATCH_GRACE_DEFAULT`). +- Run-watch failed-log tail default: `15` lines (`RUN_WATCH_TAIL_DEFAULT`). +- Run-watch failed-log tail max: `200` lines (`RUN_WATCH_TAIL_MAX`). +- PR review comments page size: `100` (`REVIEW_COMMENTS_PAGE_SIZE`). +- Actions jobs page size: `100` (`RUN_JOBS_PAGE_SIZE`). +- Search and tail numeric inputs are floored with `Math.floor()`, clamped to the max, and rejected when non-finite or `<= 0`. +- `pr_view`/`pr_diff`/`pr_checkout` batch fan-out is unbounded in tool code; all requested PRs are launched with `Promise.all()`. + +## Errors +- Tool creation is skipped entirely when `gh` is not installed. +- `git.github.run()` throws `ToolError("GitHub CLI (gh) is not installed...")` if `gh` is missing at execution time. +- `git.github.text/json()` map common failures to model-facing messages: + - not authenticated → `GitHub CLI is not authenticated. Run \`gh auth login\`.` + - missing repo context without explicit `repo` → `GitHub repository context is unavailable. Pass \`repo\` explicitly or run the tool inside a GitHub checkout.` + - otherwise stderr/stdout text, or fallback `GitHub CLI command failed: gh ...` +- `json()` also throws on empty stdout or invalid JSON. +- Local validation errors throw `ToolError`, including: + - missing required per-op fields (`issue`, `query`, `title unless fill=true`) + - invalid numeric `limit` / `tail` + - invalid `run` format + - `fill` combined with `title` or `body` + - empty exclude patterns + - missing git repo / branch / HEAD context for checkout, push, or watch + - `pr_push` on a branch without `ompPrHeadRef` metadata + - conflicting existing worktree path or branch without `force` +- `run_watch` treats failed-job log fetches specially: missing log content does not fail the watch; it marks that log `available: false` and prints `Log tail unavailable.` / `Full log unavailable.`. +- `pr_create` swallows only the post-create best-effort `gh pr view` refresh; the create step itself still fails normally. + +## Notes +- `appendRepoFlag()` intentionally skips `--repo` when the identifier argument is already a full GitHub URL; that lets `gh` derive repo/number from the URL. +- `normalizePrIdentifierList()` accepts `reviewer`, `assignee`, and `label` arrays too; the helper name is broader than its callers. +- `pr_push` depends on `pr_checkout` having run first for that local branch; there is no alternate metadata source. +- `pr_checkout` stores push metadata in branch config, not in the worktree directory. Reusing the same `pr-<number>` branch reuses those config keys. +- Worktree write serialization is keyed by the primary repo root, not the current worktree path, because git worktrees share `.git/config`, `packed-refs`, commit-graph, and worktree metadata files. +- `search_repos` is the only search op that never forwards `repo`; repository scoping must be expressed in the query itself. +- `run_watch` success on commit mode means “all observed runs succeeded and no additional runs appeared one poll later”, not merely “latest poll looked green”. +- The TUI renderer collapses failed log previews unless the result view is expanded; the underlying text result still contains the same tailed lines plus any artifact reference. diff --git a/docs/tools/inspect_image.md b/docs/tools/inspect_image.md new file mode 100644 index 000000000..16d9be70f --- /dev/null +++ b/docs/tools/inspect_image.md @@ -0,0 +1,120 @@ +# inspect_image + +> Send a local image file to a vision-capable model and return text analysis. + +## Source +- Entry: `packages/coding-agent/src/tools/inspect-image.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/inspect-image.md` +- Key collaborators: + - `packages/coding-agent/src/tools/inspect-image-renderer.ts` — TUI call/result rendering. + - `packages/coding-agent/src/utils/image-loading.ts` — path resolution, type detection, size gate, optional resize. + - `packages/coding-agent/src/utils/image-resize.ts` — downscale and recompress oversized images. + - `packages/coding-agent/src/tools/path-utils.ts` — resolve input path relative to session cwd. + - `packages/utils/src/mime.ts` — detect supported image formats from file bytes. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `path` | `string` | Yes | Image path passed to `loadImageInput`; resolved relative to `session.cwd` by `resolveReadPath(...)`. | +| `question` | `string` | Yes | User prompt sent as a text content block alongside the image. | + +## Outputs +The tool returns a single `AgentToolResult`: + +- `content`: one text block, `[{ type: "text", text }]`, where `text` is the concatenated assistant text content from the model response. +- `details`: + - `model`: `<provider>/<id>` of the selected model. + - `imagePath`: resolved filesystem path returned by `loadImageInput(...)`. + - `mimeType`: MIME type actually sent to the model after optional resize/re-encode. + +Model-visible output is single-shot, not streamed by this tool. + +TUI rendering adds presentation-only truncation from `packages/coding-agent/src/tools/inspect-image-renderer.ts`: + +- call preview truncates `question` to 100 columns, +- result view shows 4 lines collapsed or 16 lines expanded, +- each rendered output line is truncated to 120 columns, +- footer metadata shows `model · mimeType` when present. + +## Flow +1. `InspectImageTool.execute(...)` rejects immediately if `images.blockImages` is enabled in session settings. +2. It reads `session.modelRegistry`; missing registry, empty registry, missing API key, or unresolved model each raise `ToolError` from `packages/coding-agent/src/tools/inspect-image.ts`. +3. Model selection tries, in order, `pi/vision`, `pi/default`, the active model string from the session, then `availableModels[0]`. `expandRoleAlias(...)` and `resolveModelFromString(...)` handle each lookup. +4. The chosen model must advertise `input.includes("image")`; otherwise execution fails before reading the file. +5. `loadImageInput(...)` in `packages/coding-agent/src/utils/image-loading.ts` resolves the path with `resolveReadPath(...)`, detects MIME type with `readImageMetadata(...)`, and rejects files larger than `MAX_IMAGE_INPUT_BYTES` (`20 * 1024 * 1024`, 20 MiB) using `ImageInputTooLargeError`. +6. `readImageMetadata(...)` in `packages/utils/src/mime.ts` inspects file headers only. Supported detected MIME types are `image/png`, `image/jpeg`, `image/gif`, and `image/webp`. +7. If `images.autoResize` is true, `loadImageInput(...)` calls `resizeImage(...)`. Resize failures are swallowed there and the original bytes are kept. +8. If MIME detection returned no supported image type, `execute(...)` throws `ToolError("inspect_image only supports PNG, JPEG, GIF, and WEBP files detected by file content.")`. +9. The tool calls `completeSimple(...)` with one user message containing two content parts in order: + - `{ type: "image", data: imageInput.data, mimeType: imageInput.mimeType }` + - `{ type: "text", text: params.question }` +10. `systemPrompt` is a one-element array rendered from `packages/coding-agent/src/prompts/tools/inspect-image-system.md`. +11. If the model response stop reason is `error` or `aborted`, the tool maps that to `ToolError`. +12. `extractResponseText(...)` concatenates only `text` content blocks from the assistant message, trims the result, and fails if nothing remains. +13. Success returns the text plus `details`; `inspectImageToolRenderer` formats the result for the TUI. + +## Modes / Variants +- **Original image path**: `images.autoResize` disabled. The original file bytes are base64-encoded and sent with the detected MIME type. +- **Auto-resized path**: `images.autoResize` enabled. `resizeImage(...)` may downscale and re-encode the image before upload. +- **Unsupported image path**: file exists but header sniffing does not identify PNG/JPEG/GIF/WEBP. The tool returns a `ToolError` before any model call. +- **Oversize image path**: file size exceeds 20 MiB before upload. The tool returns a `ToolError` before any model call. + +## Side Effects +- Filesystem + - Resolves and reads the target image from disk. + - Stats the file once with `Bun.file(...).stat()` and reads it fully with `fs.readFile(...)`. +- Network + - Sends the final base64 image payload plus question text to the selected model through `completeSimple(...)`. +- Session state + - Reads session settings, active model preferences, cwd, and model registry. +- Background work / cancellation + - Passes the caller `AbortSignal` into `completeSimple(...)`. + - Image preprocessing is local and not cancellation-aware in these helpers. + +## Limits & Caps +- Supported detected input formats: `image/png`, `image/jpeg`, `image/gif`, `image/webp` (`SUPPORTED_IMAGE_MIME_TYPES` in `packages/utils/src/mime.ts`). +- Metadata sniff cap: `DEFAULT_IMAGE_METADATA_HEADER_BYTES = 256 * 1024` bytes. Format detection only reads up to 256 KiB from the file header. +- Upload input cap: `MAX_IMAGE_INPUT_BYTES = 20 * 1024 * 1024` bytes (20 MiB) in `packages/coding-agent/src/utils/image-loading.ts`. +- Auto-resize defaults in `packages/coding-agent/src/utils/image-resize.ts`: + - `maxWidth: 1568` + - `maxHeight: 1568` + - `maxBytes: 500 * 1024` bytes (500 KiB target) + - `jpegQuality: 75` +- Resize fast path: if the original image is already within `1568x1568` and within `maxBytes / 4` (125 KiB by default), `resizeImage(...)` returns the original bytes unchanged. +- Resize quality ladder: after the first encode pass, lossy retries use qualities `[70, 60, 50, 40]`. +- Resize dimension ladder: if quality reduction still misses the byte target, retries scale dimensions by `[1.0, 0.75, 0.5, 0.35, 0.25]` and stop if either dimension would fall below `100` pixels. +- First resize pass encodes PNG, JPEG, and WebP, then keeps the smallest encoded buffer. Fallback passes encode JPEG and WebP only, again keeping the smaller output. +- Renderer caps: + - `INSPECT_QUESTION_PREVIEW_WIDTH = 100` + - `INSPECT_OUTPUT_COLLAPSED_LINES = 4` + - `INSPECT_OUTPUT_EXPANDED_LINES = 16` + - `INSPECT_OUTPUT_LINE_WIDTH = 120` + +## Errors +- Settings gate: + - `Image submission is disabled by settings (images.blockImages=true). Disable it to use inspect_image.` +- Model resolution / capability: + - `Model registry is unavailable for inspect_image.` + - `No models available for inspect_image.` + - `Unable to resolve a model for inspect_image.` + - `Resolved model <provider>/<id> does not support image input. Configure a vision-capable model for modelRoles.vision.` + - `No API key available for <provider>/<id>. Configure credentials for this provider or choose another vision-capable model.` +- Input file: + - `Image file too large: <size> exceeds <limit> limit.` from `ImageInputTooLargeError`, remapped to `ToolError`. + - `inspect_image only supports PNG, JPEG, GIF, and WEBP files detected by file content.` when header sniffing fails. +- Model call: + - `inspect_image request failed.` if the response stop reason is `error` without a provider message. + - Provider `errorMessage` is passed through when present. + - `inspect_image request aborted.` on aborted responses. + - `inspect_image model returned no text output.` when the assistant message contains no text blocks after filtering. + +Failures surface as thrown `ToolError`s from `execute(...)`; the normal success return shape is not used for error reporting. + +## Notes +- The model-facing prompt path on disk is `packages/coding-agent/src/prompts/tools/inspect-image.md`; the assignment's underscore form does not exist. +- Format support is based on file content, not filename extension. Renaming a non-image file to `.png` does not make it valid. +- `resolveReadPath(...)` tries macOS-specific path variants: shell-unescaped spaces, AM/PM narrow no-break-space filenames, NFD normalization, and curly-quote variants. +- `loadImageInput(...)` also computes `textNote`, `dimensionNote`, and final `bytes`, but `inspect_image` does not include those in tool output. +- Auto-resize can change the MIME type sent to the model. A JPEG or GIF input may be uploaded as PNG, JPEG, or WebP depending on which encoder output is smallest. +- If `resizeImage(...)` throws or cannot decode the image, `loadImageInput(...)` silently keeps the original base64 payload instead of failing. diff --git a/docs/tools/irc.md b/docs/tools/irc.md new file mode 100644 index 000000000..016ade962 --- /dev/null +++ b/docs/tools/irc.md @@ -0,0 +1,118 @@ +# irc + +> Send short prose messages to other live agents in the current process. + +## Source +- Entry: `packages/coding-agent/src/tools/irc.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/irc.md` +- Key collaborators: + - `packages/coding-agent/src/registry/agent-registry.ts` — process-global live agent directory. + - `packages/coding-agent/src/session/agent-session.ts` — side-channel reply generation and history injection. + - `packages/coding-agent/src/prompts/system/irc-incoming.md` — no-tools auto-reply prompt. + - `packages/coding-agent/src/tools/index.ts` — tool availability gating. + - `packages/coding-agent/src/config/settings-schema.ts` — `irc.enabled` default. + - `packages/coding-agent/src/modes/controllers/event-controller.ts` — renders IRC events into chat UI. + - `packages/coding-agent/src/modes/utils/ui-helpers.ts` — formats `[IRC]` transcript lines. + - `packages/coding-agent/src/task/executor.ts` — carries `irc.enabled` into subagents. + +## Inputs + +### `op: "list"` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `op` | `"list"` | Yes | Lists peers visible to the caller. | + +### `op: "send"` + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `op` | `"send"` | Yes | Sends one message to one peer or to `"all"`. | +| `to` | `string` | Yes | Peer id such as `0-Main`, or `"all"` for broadcast. Whitespace is trimmed. | +| `message` | `string` | Yes | Message body. Whitespace is trimmed; empty-after-trim is rejected. | +| `awaitReply` | `boolean` | No | Wait for prose replies. Defaults to `true` for direct messages and `false` for `to: "all"`. | + +## Outputs +- Single-shot `AgentToolResult`; no streaming updates. +- `content` is one text block. + - `list` returns either `No other live agents.` or a bullet list headed by `<n> peer(s):`. + - `send` returns delivery summary text, then optional `## Replies`, `## Failed`, and `Unknown / unavailable peers:` sections. +- `details` is structured metadata: + - `list`: `{ op, from, peers, channels }` + - `send`: `{ op, from, to, delivered, replies?, failed?, notFound? }` +- The tool does not return raw IRC frames, message ids, or a transcript object. + +## Flow +1. `IrcTool.createIf` only constructs the tool when `irc.enabled` is on and the session has both an `AgentRegistry` and `getAgentId` (`packages/coding-agent/src/tools/irc.ts`). +2. Tool discovery adds another gate in `packages/coding-agent/src/tools/index.ts`: if the caller is `0-Main` and `async.enabled` is off, `irc` is hidden because the main agent cannot talk to concurrent peers in sync mode. +3. `execute` resolves the process-global registry and sender id. Missing either returns a text error result instead of throwing. +4. `op: "list"` calls `registry.listVisibleTo(senderId)`, which exposes every other agent in flat namespace whose status is `running` or `idle` (`packages/coding-agent/src/registry/agent-registry.ts`). +5. `list` formats human-readable lines and returns `channels` as `['all', ...peerIds]`. These are logical targets only; there is no channel join state. +6. `op: "send"` trims `to` and `message`; missing values produce text errors. +7. `send` resolves targets: + - `to === "all"`: all visible peers. + - otherwise: one exact registry id, excluding self and excluding peers not in `running`/`idle`. +8. `send` chooses `awaitReply = params.awaitReply ?? !isBroadcast`. +9. Each target is dispatched in parallel via `target.session.respondAsBackground(...)`. One slow or failing peer does not block dispatch to the others. +10. `respondAsBackground` emits an `irc_message` session event, forwards a display-only relay to the main session UI, and either: + - queues just the incoming message for later history injection when `awaitReply === false`, or + - renders `packages/coding-agent/src/prompts/system/irc-incoming.md`, runs `runEphemeralTurn` with `toolChoice: "none"`, emits an auto-reply event, then queues both incoming and reply messages for history injection. +11. Deferred injection waits until the recipient is no longer streaming; `#flushPendingBackgroundExchanges` appends the custom messages through normal `message_start`/`message_end` external events so persistence and listeners see them. +12. `send` aggregates `delivered`, `replies`, `failed`, and `notFound`, then returns one text summary plus matching `details`. + +## Modes / Variants +- `list`: enumerate visible peers and logical channels. +- `send` direct message: one exact peer id, default synchronous auto-reply. +- `send` broadcast: `to: "all"`, default fire-and-forget (`awaitReply: false`) to every visible peer. +- `send` with `awaitReply: false`: recipient records the incoming message but does not generate a reply. +- `send` with `awaitReply: true`: recipient performs a no-tools ephemeral LLM turn and returns prose. + +## Side Effects +- Session state + - Reads from the process-global `AgentRegistry`. + - Emits `irc_message` session events on recipient sessions. + - Queues IRC custom messages into recipient persisted history after the current stream finishes. + - For non-main recipients, forwards display-only relay observations into the main session UI; these relays are not persisted to the main agent history. + - Subagents inherit `irc.enabled` from task executor settings. +- User-visible prompts / interactive UI + - IRC events render as `[IRC]` transcript lines in the TUI. + - Auto-replies are generated from `packages/coding-agent/src/prompts/system/irc-incoming.md` and explicitly forbid tool use. +- Background work / cancellation + - `send` starts one background `respondAsBackground` call per target. + - The caller's `AbortSignal` is forwarded into each background reply turn. +- Network + - No IRC server connection. + - When `awaitReply: true`, the recipient may make model-provider API calls through `runEphemeralTurn`. +- Filesystem + - No direct filesystem writes in the tool itself. + +## Limits & Caps +- Availability gates: + - `irc.enabled` defaults to `true` in `packages/coding-agent/src/config/settings-schema.ts`. + - Main agent tool discovery suppresses `irc` when `async.enabled` is off (`packages/coding-agent/src/tools/index.ts`). +- Visibility scope: only peers in status `running` or `idle` are addressable via `listVisibleTo`. +- Reply execution: + - No tools are available in auto-reply turns (`toolChoice: "none"` in `runEphemeralTurn`). + - No internal timeout, retry, backoff, rate limit, or reply length cap is defined in `irc.ts`; behavior relies on the underlying model stream and any upstream API limits. +- Flush scheduling: deferred history injection polls every `50` ms while the recipient is still streaming (`#scheduleBackgroundExchangeFlush` in `packages/coding-agent/src/session/agent-session.ts`). + +## Errors +- The tool returns text errors, not thrown exceptions, for: + - missing registry: `IRC is unavailable in this session.` + - missing sender id: `IRC is unavailable: caller has no agent id.` + - missing `to`: `` `to` is required for op="send". `` + - missing `message`: `` `message` is required for op="send". `` + - unknown op: `Unknown irc op.` +- Unknown, self-addressed, non-running, and non-idle direct targets are reported under `details.notFound` and in the text footer `Unknown / unavailable peers:`. +- If a target has no attached session, it is treated as not found. +- Exceptions thrown by `respondAsBackground` or `runEphemeralTurn` are caught per-target and surfaced under `details.failed` as `{ id, error }`; other recipients still complete. +- If no target succeeds, `send` still returns normally with `No recipients received the message.` and optional `failed`/`notFound` metadata. + +## Notes +- This is IRC-like naming only. There are no servers, sockets, nick registration, auth handshakes, channels beyond `all`, or commands such as join/part/topic. +- Addressing is by exact agent id from the registry; there is no fuzzy lookup or aliasing. +- `channels` in `list` is synthetic output: `all` plus visible peer ids. Nothing is persisted across calls as channel membership. +- Persistence is per recipient history, not per sender history. The sender gets the tool result; the recipient later sees injected custom messages on its next turn. +- The main UI may show IRC relays for conversations it was not part of, but those relay records are explicitly display-only. +- Because reply generation snapshots in-flight assistant text, a recipient can answer based on partially streamed context. +- Direct self-messaging is rejected by resolving the target as unavailable. \ No newline at end of file diff --git a/docs/tools/job.md b/docs/tools/job.md new file mode 100644 index 000000000..ae0e569b8 --- /dev/null +++ b/docs/tools/job.md @@ -0,0 +1,143 @@ +# job + +> Wait for or cancel background jobs managed by the session async runtime. + +## Source +- Entry: `packages/coding-agent/src/tools/job.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/job.md` +- Key collaborators: + - `packages/coding-agent/src/async/job-manager.ts` — job registry, cancellation, delivery suppression. + - `packages/coding-agent/src/async/support.ts` — feature gating for background jobs. + - `packages/coding-agent/src/internal-urls/jobs-protocol.ts` — `jobs://` listing and per-job detail. + - `packages/coding-agent/src/tools/bash.ts` — explicit async bash and auto-backgrounded bash jobs. + - `packages/coding-agent/src/task/index.ts` — async task-job scheduling. + - `packages/coding-agent/src/sdk.ts` — automatic follow-up delivery for unsuppressed completions. + - `packages/coding-agent/src/config/settings-schema.ts` — `async.pollWaitDuration` options. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `poll` | `string[]` | No | Job ids to watch. If omitted and `cancel` is also omitted, the tool watches all running jobs. If provided, missing ids are silently filtered out before waiting. | +| `cancel` | `string[]` | No | Job ids to cancel before any polling. Missing ids are reported as `not_found`; non-running ids as `already_completed`. | + +## Outputs +The tool returns one text block plus `details`. + +- `content[0].text`: markdown-like plain text sections assembled by `#buildResult(...)`: + - `## Cancelled (N)` for cancel outcomes. + - `## Completed (N)` for non-running jobs, including stored `resultText` and `errorText`. + - `## Still Running (N)` for jobs still in `running`. +- `details.jobs`: array of snapshots: + - `id: string` + - `type: "bash" | "task"` + - `status: "running" | "completed" | "failed" | "cancelled"` + - `label: string` + - `durationMs: number` + - optional `resultText`, `errorText` +- `details.cancelled` appears only when `cancel` was passed; each item is `{ id, status }` where status is `"cancelled" | "not_found" | "already_completed"`. + +Streaming behavior: +- During a polling wait, `execute(...)` emits `onUpdate(...)` every 500 ms with an empty text block and fresh `details.jobs` snapshots. +- Final return is single-shot after a completion, timeout, abort, or immediate fast path. + +Related read path: +- `read jobs://` lists all current jobs. +- `read jobs://<id>` renders one job with status, label, start time, duration, and stored result/error text. + +## Flow +1. `JobTool.createIf(...)` in `packages/coding-agent/src/tools/job.ts` only exposes the tool when `isBackgroundJobSupportEnabled(...)` returns true for either `async.enabled` or `bash.autoBackground.enabled`. +2. `execute(...)` fetches `session.asyncJobManager`. If absent, it returns `Async execution is disabled; no background jobs are available.` +3. `cancel` ids are processed first: + - `manager.getJob(id)` missing → `not_found`. + - existing job with `status !== "running"` → `already_completed`. + - running job → `manager.cancel(id)`, which sets `job.status = "cancelled"`, aborts the controller, and schedules eviction. +4. Polling mode is chosen with `const shouldPoll = requestedPollIds !== undefined || cancelIds.length === 0`: + - only `cancel` present → return immediately, no wait. + - explicit `poll`, or no args at all → proceed to watch jobs. +5. Watch set resolution: + - explicit `poll` → map ids through `manager.getJob(...)` and drop missing ones. + - no `poll` and no `cancel` → `manager.getRunningJobs()`. +6. Empty watch set returns immediately: + - if cancellations happened, return snapshots for the cancelled ids that still exist. + - else return either `No matching jobs found for IDs: ...` or `No running background jobs to wait for.` +7. If every watched job is already non-running, `#buildResult(...)` returns immediately without waiting. +8. Otherwise the tool waits on `Promise.race(...)` across: + - every watched running job's `job.promise`, + - a timeout promise for `async.pollWaitDuration`, + - the tool-call abort signal when present. +9. Before waiting, it calls `manager.watchJobs(watchedJobIds)`. This suppresses automatic completion delivery for those ids while they are being watched. +10. If `onUpdate` exists, a 500 ms interval sends progress snapshots from `#snapshotJobs(...)`; one snapshot is emitted immediately before entering the race. +11. In `finally`, the tool always calls `manager.unwatchJobs(...)`, clears the timeout, and stops the progress interval. +12. `#buildResult(...)` deduplicates jobs, snapshots current manager state, then calls `manager.acknowledgeDeliveries(...)` for every non-running job in the result. That suppresses later automatic follow-up delivery for the same completions and removes queued deliveries for those ids. +13. The final text groups jobs by non-running vs still-running state. A timeout is not an error path; it simply returns the current snapshot. + +## Modes / Variants +- Poll all running jobs: call with neither `poll` nor `cancel`. +- Poll explicit ids: call with `poll` only. +- Cancel only: call with `cancel` only; cancellations happen and the tool returns immediately. +- Cancel then poll: call with both. Cancellations are applied first, then the tool watches the remaining resolved `poll` ids. +- Read-only inspection outside the tool: `jobs://` and `jobs://<id>` expose the same manager state without waiting. + +Spawn paths that produce jobs: +- `packages/coding-agent/src/tools/bash.ts` + - `async: true` always registers a `type: "bash"` job with `AsyncJobManager.register(...)` and returns a start message. + - auto-background mode (`bash.autoBackground.enabled`) starts the same managed job path for non-PTY commands, waits up to `min(bash.autoBackground.thresholdMs, timeoutMs - 1000)`, and if the command is still running returns a background-job start result instead of inline command output. +- `packages/coding-agent/src/task/index.ts` + - when `async.enabled` is on, the chosen agent is not blocking, and `tasks.length > 0`, each task item is registered as a `type: "task"` job. + +Lifecycle and exact state names: +- Conceptual scheduling path: `pending` (only task-progress bookkeeping before work starts) → `running` → `completed` / `failed`; cancellation changes a running async job to `cancelled`. +- Exact `AsyncJob.status` values in `packages/coding-agent/src/async/job-manager.ts`: `"running" | "completed" | "failed" | "cancelled"`. +- Exact per-task progress values in `packages/coding-agent/src/task/types.ts`: `"pending" | "running" | "completed" | "failed" | "aborted"`. + +## Side Effects +- Filesystem + - None in `job.ts` itself. + - Jobs being observed may already have written artifacts/results through their own tool runtimes. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Reads and mutates `session.asyncJobManager` state. + - `watchJobs(...)` / `unwatchJobs(...)` toggle delivery suppression for the watched ids. + - `acknowledgeDeliveries(...)` marks completed ids as suppressed and removes queued deliveries for them. + - `cancel(...)` aborts running jobs through each job's `AbortController`. +- User-visible prompts / interactive UI + - Polling emits periodic `onUpdate` snapshots every 500 ms. + - Automatic job completion follow-ups are generated by `packages/coding-agent/src/sdk.ts` only for unsuppressed deliveries. +- Background work / cancellation + - Waiting uses a timeout plus optional tool-call abort signal. + - Cancelling a job does not synchronously await teardown; it flips state, aborts, and returns control to the manager/job promise. + +## Limits & Caps +- Poll wait duration comes from `async.pollWaitDuration` in `packages/coding-agent/src/config/settings-schema.ts`: + - allowed values: `5s`, `10s`, `30s`, `1m`, `5m` + - default: `30s` +- Progress update cadence while polling: `PROGRESS_INTERVAL_MS = 500` in `packages/coding-agent/src/tools/job.ts`. +- Async job retention default: `DEFAULT_RETENTION_MS = 5 * 60 * 1000` in `packages/coding-agent/src/async/job-manager.ts`. +- Manager fallback max-running limit: `DEFAULT_MAX_RUNNING_JOBS = 15` in `packages/coding-agent/src/async/job-manager.ts`. +- Session wiring clamps `async.maxJobs` to `1..100` before constructing the manager in `packages/coding-agent/src/sdk.ts`; settings default is `100` in `packages/coding-agent/src/config/settings-schema.ts`. +- Async completion delivery retry backoff in `packages/coding-agent/src/async/job-manager.ts`: + - base `500` ms + - max `30_000` ms + - jitter `< 200` ms + - exponent capped at 8 doublings + +## Errors +- Tool-disabled path is returned as normal text, not thrown: `Async execution is disabled; no background jobs are available.` +- Polling a nonexistent id is not an exception: + - with `poll` only, missing ids are dropped; if none remain the tool returns `No matching jobs found for IDs: ...`. + - with `cancel`, each missing id is reported as `not_found` in `details.cancelled` and text. +- Cancelling a non-running job is not an exception; it reports `already_completed` even if the actual status is `completed`, `failed`, or `cancelled`. +- Tool-call abort during polling stops waiting and returns a final snapshot through `#buildResult(...)`; it does not cancel watched jobs. +- Failures inside the underlying async work are stored on the job (`status: "failed"`, `errorText`) and reported in normal tool output, not rethrown by `job`. +- `read jobs://<id>` missing job returns markdown content headed `# Job Not Found` rather than throwing. + +## Notes +- `job` waits for the first watched running job to settle, not for all watched jobs. If others remain `running`, they are reported under `## Still Running`; the caller must invoke `job` again to continue waiting. +- Delivery suppression is the key difference between snapshot and automatic delivery: + - snapshots (`job`, `read jobs://`) read current manager state; + - follow-up delivery comes from `AsyncJobManager.#enqueueDelivery(...)` and `sdk.ts` `onJobComplete`; + - watched or acknowledged ids are suppressed via `isDeliverySuppressed(...)`. +- `manager.cancel(id)` sets `status = "cancelled"` before the underlying promise settles. The job function may later populate `resultText` or `errorText`; `job-manager.ts` preserves that text but does not transition the status away from `cancelled`. +- `jobs://` is implemented by `JobsProtocolHandler` with `immutable = true`, but each resolve call reads live manager state at access time. +- `jobs://<id>` shows a cancellation section only when a cancelled job has `errorText`; cancelled jobs with `resultText` are not rendered with a result section there. +- Retention eviction removes the job record, suppression flags, and watch flag together. After eviction, both `job` and `read jobs://<id>` behave as if the id never existed. diff --git a/docs/tools/lsp.md b/docs/tools/lsp.md new file mode 100644 index 000000000..7a4d273d8 --- /dev/null +++ b/docs/tools/lsp.md @@ -0,0 +1,313 @@ +# lsp + +> Query language servers for diagnostics, navigation, symbols, renames, code actions, capabilities, and raw requests. + +## Source +- Entry: `packages/coding-agent/src/lsp/index.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/lsp.md` +- Key collaborators: + - `packages/coding-agent/src/lsp/client.ts` — client process lifecycle and JSON-RPC + - `packages/coding-agent/src/lsp/config.ts` — config loading, auto-detect, server selection + - `packages/coding-agent/src/lsp/lspmux.ts` — optional `lspmux` command wrapping + - `packages/coding-agent/src/lsp/edits.ts` — apply `WorkspaceEdit` and text edits + - `packages/coding-agent/src/lsp/utils.ts` — URI conversion, symbol resolution, formatting, glob expansion + - `packages/coding-agent/src/lsp/types.ts` — tool schema and protocol types + - `packages/coding-agent/src/lsp/clients/index.ts` — custom linter client cache/factory + - `packages/coding-agent/src/lsp/clients/lsp-linter-client.ts` — LSP-backed linter adapter + - `packages/coding-agent/src/lsp/clients/biome-client.ts` — Biome CLI diagnostics/formatting adapter + - `packages/coding-agent/src/lsp/clients/swiftlint-client.ts` — SwiftLint CLI diagnostics adapter + - `packages/coding-agent/src/tools/index.ts` — tool registration and `lsp.enabled` gating + - `packages/coding-agent/src/tools/tool-timeouts.ts` — timeout defaults and clamping + - `packages/coding-agent/src/lsp/defaults.json` — built-in server definitions for auto-detect + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `action` | string enum | Yes | One of `diagnostics`, `definition`, `references`, `hover`, `symbols`, `rename`, `rename_file`, `code_actions`, `type_definition`, `implementation`, `status`, `reload`, `capabilities`, `request`. | +| `file` | string | No | File path; for `diagnostics` also a glob; for workspace forms use `"*"`; for `rename_file` this is the source path. | +| `line` | number | No | 1-indexed line number for position-based actions. Defaults to `1` on the single-file action path. | +| `symbol` | string | No | Substring used to resolve the column on `line`. Supports `name#N` occurrence selectors; `N` is 1-indexed and defaults to `1`. | +| `query` | string | No | Workspace symbol query, code-action selector/filter, or LSP method name for `action=request`. | +| `new_name` | string | No | Required for `rename` and `rename_file`. | +| `apply` | boolean | No | For `rename`/`rename_file`, apply unless explicitly `false`. For `code_actions`, list unless explicitly `true`. | +| `timeout` | number | No | Seconds, clamped by `clampTimeout("lsp", ...)` to `5..60`, default `20`. | +| `payload` | string | No | JSON string for `action=request`; overrides auto-built params. | + +## Outputs +- Single-shot `AgentToolResult`. +- `content` is always one text block: `[{ type: "text", text: string }]`. +- `details` is `LspToolDetails`: `action`, `success`, optional `serverName`, optional original `request`. +- No streaming updates. +- No artifact URIs or background jobs. +- Many validation failures are returned as ordinary text results with `details.success: false`; aborts throw `ToolAbortError` instead. + +## Flow +1. `packages/coding-agent/src/tools/index.ts` registers `lsp: LspTool.createIf`; session creation also gates it behind `session.enableLsp !== false` and `settings.get("lsp.enabled")`. +2. `LspTool.execute()` in `packages/coding-agent/src/lsp/index.ts` clamps `timeout` with `clampTimeout("lsp", ...)`, builds an `AbortSignal.timeout(...)`, and combines it with the caller signal. +3. `getConfig()` loads and caches `LspConfig` per cwd, applies idle-timeout config via `setIdleTimeout()`, and reuses the cached config on later calls. +4. Config loading in `packages/coding-agent/src/lsp/config.ts` merges `defaults.json` with JSON/YAML overrides from project, project config dirs, user config dirs, plugin roots, and home; if there are no overrides it auto-detects servers from root markers plus executable discovery. +5. Server routing uses `getServersForFile()` / `getServerForFile()` from `config.ts`: extension or basename match, then sort primary servers before linters. `index.ts` further filters custom linter clients out of navigation/refactor paths with `getLspServersForFile()` / `getLspServerForFile()`. +6. `getOrCreateClient()` in `client.ts` creates one process per `command:cwd`, optionally wraps supported commands with `lspmux`, spawns the server, starts the background message reader, sends `initialize`, stores server capabilities, then sends `initialized`. +7. The message reader in `client.ts` parses LSP frames, resolves pending requests, caches `publishDiagnostics`, tracks `$/progress` tokens for project-load completion, answers `workspace/configuration`, and applies `workspace/applyEdit` requests through `applyWorkspaceEdit()`. +8. File-scoped actions call `ensureFileOpen()` before requests. Column resolution uses `resolveSymbolColumn()` from `utils.ts`: read the target file, pick first non-whitespace when `symbol` is omitted, otherwise find the exact or case-insensitive match on the target line and honor `#N` occurrence selectors. +9. Actions dispatch in `LspTool.execute()` through dedicated branches: workspace-only branches (`status`, some `diagnostics`, workspace `symbols`, workspace `reload`, `capabilities`, `request`) run before the single-file switch; all other single-file actions share one client lookup and `switch(action)`. +10. Requests go through `sendRequest()` in `client.ts`, which allocates an incrementing JSON-RPC id, installs abort and timeout handling, sends `$/cancelRequest` on abort, and rejects on timeout or process exit. +11. Actions that return edits either preview with `formatWorkspaceEdit()` or apply with `applyWorkspaceEdit()` from `edits.ts`; `rename_file` also performs the filesystem rename and then sends `workspace/didRenameFiles`. +12. Non-abort failures inside the single-file action block are converted to `LSP error: ...`; many precondition failures return explicit text without throwing. + +## Modes / Variants +### Routing and workspace scope +- `file: "*"` is only special for `diagnostics`, `symbols`, and `reload`. +- `status` ignores `file`. +- `capabilities` with omitted `file` or `"*"` inspects all non-custom LSP servers; with a concrete file it scopes to matching non-custom servers. +- `request` with omitted `file` or `"*"` chooses the first available non-custom LSP server; with a concrete file it chooses that file's primary non-linter server. +- `rename_file` sends `workspace/willRenameFiles` and `workspace/didRenameFiles` to every non-custom LSP server from `getLspServers(config)`, not just one file-scoped server. +- Diagnostics are the only tool action that queries both normal LSP servers and custom linter clients (`BiomeClient`, `SwiftLintClient`, or `LspLinterClient`). + +### `diagnostics` +**Inputs** +- Required: `file`, unless using workspace mode with `file: "*"`. +- Optional: `timeout`. + +**Execution** +- `file: "*"`: `runWorkspaceDiagnostics()` detects project type from root markers and runs one subprocess command: Rust `cargo check --message-format=short`, TypeScript `npx tsc --noEmit`, Go `go build ./...`, Python `pyright`. +- Concrete file or glob: `resolveDiagnosticTargets()` treats non-globs as one target, otherwise expands a `Bun.Glob` up to `MAX_GLOB_DIAGNOSTIC_TARGETS`. +- Per file, every matching server runs: custom clients call `lint(file)`; real LSP servers optionally wait for project load, capture `diagnosticsVersion`, `refreshFile()`, then `waitForDiagnostics()` for fresh `publishDiagnostics`. +- Results are deduplicated by range+message and severity-sorted. + +**Output text** +- Single target with no issues: `OK`. +- Single target with issues: `<summary>:\n<grouped diagnostics>`. +- Batch/glob target: one section per file, plus an initial truncation warning when the glob exceeds the file cap. +- Workspace mode: `Workspace diagnostics (<detected description>):\n<command output>`. + +### `definition` +**Inputs** +- Required: `file`. +- Optional: `line`, `symbol`, `timeout`. + +**Execution** +- Sends `textDocument/definition` with `{ textDocument, position }`. +- Accepts `Location`, `Location[]`, `LocationLink`, or `LocationLink[]`; `normalizeLocationResult()` converts `LocationLink` to `targetSelectionRange ?? targetRange`. +- Waits for project load before the request. + +**Output text** +- `No definition found` or `Found N definition(s):` followed by `file:line:col` and one context line above/below each location. + +### `type_definition` +Same as `definition`, but sends `textDocument/typeDefinition` and reports `type definition(s)`. + +### `implementation` +Same as `definition`, but sends `textDocument/implementation` and reports `implementation(s)`. + +### `references` +**Inputs** +- Required: `file`. +- Optional: `line`, `symbol`, `timeout`. + +**Execution** +- Sends `textDocument/references` with `includeDeclaration: true`. +- For project-aware servers, retries up to `REFERENCES_RETRY_COUNT` times when the only hit is the queried declaration; between retries it waits for project load and sleeps `REFERENCES_RETRY_DELAY_MS`. +- First `REFERENCE_CONTEXT_LIMIT` references include surrounding context; the rest are location-only. + +**Output text** +- `No references found` or `Found N reference(s):` with contextual entries first, then `... M additional reference(s) shown without context` when truncated. + +### `hover` +**Inputs** +- Required: `file`. +- Optional: `line`, `symbol`, `timeout`. + +**Execution** +- Sends `textDocument/hover`. +- `extractHoverText()` flattens strings, markup content, marked-string objects, or arrays into plain text. + +**Output text** +- `No hover information` or the extracted hover text. + +### `symbols` +**Inputs** +- Workspace mode: `file: "*"` or omitted file on the early workspace branch, plus required `query`. +- Document mode: required `file`. +- Optional: `timeout`. + +**Execution** +- Workspace mode sends `workspace/symbol` to every non-custom LSP server, post-filters matches with `filterWorkspaceSymbols()`, deduplicates with `dedupeWorkspaceSymbols()`, then truncates to `WORKSPACE_SYMBOL_LIMIT`. +- Document mode sends `textDocument/documentSymbol` to the primary server. If the first item has `selectionRange`, it formats hierarchical `DocumentSymbol`s; otherwise it formats flat `SymbolInformation`s. + +**Output text** +- Workspace mode: `Found N symbol(s) matching "query":` plus formatted `name @ file:line:col`, with an omission line when over the limit. +- Document mode: `Symbols in <file>:` plus hierarchical or flat symbol lines. + +### `rename` +**Inputs** +- Required: `file`, `new_name`. +- Optional: `line`, `symbol`, `apply`, `timeout`. + +**Execution** +- Waits for project load, sends `textDocument/rename`, receives a `WorkspaceEdit`. +- `apply !== false` applies edits immediately with `applyWorkspaceEdit()`. +- `apply === false` renders a preview with `formatWorkspaceEdit()`. + +**Output text** +- `Rename returned no edits`, `Applied rename:` plus applied change lines, or `Rename preview:` plus summarized edits. + +### `rename_file` +**Inputs** +- Required: `file` source path, `new_name` destination path. +- Optional: `apply`, `timeout`. + +**Execution** +- Resolves absolute source and destination, rejects identical paths, missing source, existing destination, empty rename set, or directories with more than `MAX_RENAME_PAIRS` files. +- `enumerateRenamePairs()` returns one `{oldUri,newUri}` pair for a file or walks every regular file in a directory tree. +- Sends `workspace/willRenameFiles` with `{ files: pairs }` to every non-custom LSP server; collects returned `WorkspaceEdit`s and server notes. +- Preview mode (`apply === false`) only formats those edits. +- Apply mode runs each returned `WorkspaceEdit`, renames the source path on disk, sends `textDocument/didClose` for every renamed open file, deletes those `openFiles` entries, then sends `workspace/didRenameFiles`. + +**Output text** +- Preview: `Rename preview: <file-count label> → <dest>` plus per-server edit summaries and optional server notes. +- Apply: `Renamed <file-count label> → <dest>` plus applied edit summaries, filesystem rename line, and optional server notes. + +### `code_actions` +**Inputs** +- Required: `file`. +- Optional: `line`, `symbol`, `query`, `apply`, `timeout`. + +**Execution** +- Reads cached diagnostics for the open URI from `client.diagnostics` and sends `textDocument/codeAction` for a zero-width range at the resolved position. +- When `apply !== true`, `query` is passed as `context.only: [query]`; this is a server-side kind filter. +- When `apply === true`, `query` becomes a required client-side selector: either a zero-based numeric index or a case-insensitive substring of the action title. +- Applying a `CodeAction` uses `applyCodeAction()`: optionally `codeAction/resolve`, then `applyWorkspaceEdit(edit)`, then optional `workspace/executeCommand`. +- Applying a bare `Command` only runs `workspace/executeCommand`. + +**Output text** +- List mode: `N code action(s):` plus `index: [kind] title` lines. +- Apply mode success: `Applied "title":` plus `Workspace edit:` and/or `Executed command(s):` sections. +- Apply mode miss: `No code action matches "query". Available actions:`. +- Apply mode with no edit/command: `Action "title" has no workspace edit or command to apply`. + +### `status` +**Inputs** +- None. + +**Execution** +- Reads configured servers from cached `LspConfig`, not `getActiveClients()`. +- Calls `detectLspmux()` and appends status text when `lspmux` is installed. + +**Output text** +- `Active language servers: ...` or `No language servers configured for this project`, optionally followed by `lspmux: active (multiplexing enabled)` or `lspmux: installed but server not running`. + +### `reload` +**Inputs** +- Workspace mode: `file: "*"` or omitted `file`. +- Single-file mode: required `file`. +- Optional: `timeout`. + +**Execution** +- Workspace mode reloads every non-custom LSP server. +- Single-file mode reloads the primary server for that file. +- `reloadServer()` tries `rust-analyzer/reloadWorkspace`, then `workspace/didChangeConfiguration` with `{ settings: {} }`; if neither works it kills the process so the next request cold-starts a new client. + +**Output text** +- One line per server: `Reloaded <server>`, `Restarted <server>`, or `Failed to reload <server>: ...`. + +### `capabilities` +**Inputs** +- Optional: `file`, `timeout`. + +**Execution** +- With a concrete `file`, inspects matching non-custom servers for that file. +- With omitted `file` or `"*"`, inspects every non-custom configured server. +- Starts servers as needed and dumps `client.serverCapabilities ?? {}` as pretty JSON. + +**Output text** +- Per server: `<server>:` followed by indented `capabilities: { ... }`, or `<server>: failed to start (...)`. + +### `request` +**Inputs** +- Required: `query` method name. +- Optional: `file`, `line`, `symbol`, `payload`, `timeout`. + +**Execution** +- Chooses one non-custom server: file-scoped primary server, otherwise the first configured non-custom server. +- Param building precedence: + 1. If `payload` is present, parse JSON and use it verbatim. + 2. Else if `file` is concrete and `line` is present, build `{ textDocument: { uri }, position: { line: line - 1, character } }` using `resolveSymbolColumn()`. + 3. Else if `file` is concrete, build `{ textDocument: { uri } }`. + 4. Else use `{}`. +- Opens the file first when `file` is concrete. + +**Output text** +- Success: `<server> ← <method>:\n<formatted result>`, where non-string results are `JSON.stringify(..., null, 2)` and nullish values become `null`. +- Failure: `LSP error from <server> on <method>: ...`. + +## Side Effects +- Filesystem + - Reads config files, target files, and root markers. + - `rename` and `code_actions` may edit/create/delete/rename files via `applyWorkspaceEdit()`. + - `rename_file` always renames the source path on disk in apply mode. + - Server-initiated `workspace/applyEdit` requests also mutate files through `applyWorkspaceEdit()`. +- Network + - None directly; communication is local stdio JSON-RPC to subprocesses. +- Subprocesses / native bindings + - Spawns language servers with `ptree.spawn()`. + - Workspace diagnostics spawns `cargo`, `npx`, `go`, or `pyright`. + - `BiomeClient` and `SwiftLintClient` spawn CLI tools. + - Optional `lspmux` detection spawns `lspmux status`; supported servers may be wrapped through `lspmux client`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Caches config per cwd in `configCache`. + - Caches LSP clients per `command:cwd`, with `pendingRequests`, `diagnostics`, `openFiles`, `serverCapabilities`, and project-load state. + - Caches custom linter clients by `serverName:cwd`. + - Updates client `lastActivity`; optional idle-timeout cleanup is driven by `setIdleTimeout()`. +- Background work / cancellation + - Every request has an abortable timeout signal. + - Aborting an in-flight LSP request sends `$/cancelRequest`. + - Background message readers persist for each live client until process exit/shutdown. + +## Limits & Caps +- Tool timeout clamp: default `20`, min `5`, max `60` seconds — `TOOL_TIMEOUTS.lsp` in `packages/coding-agent/src/tools/tool-timeouts.ts`. +- LSP request default timeout inside `sendRequest()`: `30_000ms` — `DEFAULT_REQUEST_TIMEOUT_MS` in `packages/coding-agent/src/lsp/client.ts`. +- Warmup initialize timeout default: `5_000ms` — `WARMUP_TIMEOUT_MS` in `packages/coding-agent/src/lsp/client.ts`. +- Project-load wait fallback: `15_000ms` — `PROJECT_LOAD_TIMEOUT_MS` in `packages/coding-agent/src/lsp/client.ts`. +- Idle-client sweep interval when enabled: `60_000ms` — `IDLE_CHECK_INTERVAL_MS` in `packages/coding-agent/src/lsp/client.ts`. +- Diagnostic message output cap: first `50` messages — `DIAGNOSTIC_MESSAGE_LIMIT` in `packages/coding-agent/src/lsp/index.ts`. +- Single-file diagnostics wait: `3_000ms` — `SINGLE_DIAGNOSTICS_WAIT_TIMEOUT_MS`. +- Batch/glob diagnostics wait per file: `400ms` — `BATCH_DIAGNOSTICS_WAIT_TIMEOUT_MS`. +- Glob diagnostic target cap: first `20` matches — `MAX_GLOB_DIAGNOSTIC_TARGETS`. +- Workspace symbol cap: first `200` entries — `WORKSPACE_SYMBOL_LIMIT`. +- Reference context cap: first `50` references include source context — `REFERENCE_CONTEXT_LIMIT`. +- References retry count: `2` retries, `250ms` backoff — `REFERENCES_RETRY_COUNT`, `REFERENCES_RETRY_DELAY_MS`. +- Directory rename cap: `1_000` file pairs — `MAX_RENAME_PAIRS`. +- `detectLspmux()` state cache TTL: `5 * 60 * 1000ms`; liveness check timeout: `1_000ms` — `STATE_CACHE_TTL_MS`, `LIVENESS_TIMEOUT_MS` in `packages/coding-agent/src/lsp/lspmux.ts`. +- Workspace diagnostics output cap: first `50` lines from the subprocess. + +## Errors +- Missing or invalid inputs are usually returned as text with `details.success: false`, not thrown: + - missing `file`/`query`/`new_name` + - invalid JSON in `payload` + - no matching server + - invalid `rename_file` source/destination conditions +- `resolveSymbolColumn()` throws explicit errors for missing files, missing symbols, and out-of-bounds `#N` selectors; these surface as `LSP error: ...` or request-specific error text. +- `sendRequest()` rejects on timeout with `LSP request <method> timed out after <ms>ms`. +- Client process exit rejects all pending requests with an exit-code/stderr error assembled in `getOrCreateClient()`. +- Single-file action failures inside the main `try` become `LSP error: <message>`. +- `request` has its own error envelope: `LSP error from <server> on <method>: <message>`. +- Some server failures are intentionally softened: + - diagnostics continue when one server fails + - `rename_file` suppresses `workspace/willRenameFiles` “method not found” errors and records other server errors as notes + - `code_actions` ignores `codeAction/resolve` failures and applies unresolved actions when possible +- Aborts are not converted to text: `ToolAbortError` is rethrown. + +## Notes +- `status` reports configured/available servers from `LspConfig`, not currently active client processes from `getActiveClients()`. +- `getLspServerForFile()` excludes `createClient` adapters and linter-only servers; navigation/refactor actions never target Biome/SwiftLint custom clients. +- `getServersForFile()` matches both file extensions and exact basenames from `fileTypes`; config can target names like `Dockerfile` if present. +- `symbol` matching is exact first, then case-insensitive, and falls back to the Nth occurrence on the specified line only; it never scans other lines. +- `code_actions` uses `query` in two different ways: server-side `context.only` filter in list mode, client-side title/index selector in apply mode. +- `rename` and `rename_file` default to apply. Preview requires `apply: false`. +- `request` with `file: "*"` is treated the same as omitted `file`: it does not build workspace-specific params. +- `reload` does not recreate a client immediately after killing it; the next request triggers reinitialization. +- `workspace/applyEdit` can apply edits initiated by the server outside the direct tool action result path. +- `detectLspmux()` can be disabled with `PI_DISABLE_LSPMUX=1`; only `rust-analyzer` is in `DEFAULT_SUPPORTED_SERVERS`. +- `configCache` is per-process and never auto-invalidated; config changes require a fresh process to be observed by `getConfig()` callers. \ No newline at end of file diff --git a/docs/tools/read.md b/docs/tools/read.md new file mode 100644 index 000000000..5483d727f --- /dev/null +++ b/docs/tools/read.md @@ -0,0 +1,300 @@ +# read + +> Read files, directories, archives, SQLite databases, internal resources, images, documents, and URLs through one `path` string. + +## Source +- Entry: `packages/coding-agent/src/tools/read.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/read.md` +- Key collaborators: + - `packages/coding-agent/src/tools/path-utils.ts` — split `path` from trailing selectors; normalize local paths. + - `packages/coding-agent/src/tools/archive-reader.ts` — detect `archive.ext:inner/path`, index archives, list/read entries. + - `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite targets, parse selectors, render tables. + - `packages/coding-agent/src/tools/fetch.ts` — URL parsing, fetch/render pipeline, URL cache/artifacts. + - `packages/coding-agent/src/internal-urls/router.ts` — resolve `agent://`, `artifact://`, `jobs://`, `local://`, `mcp://`, `memory://`, `pi://`, `rule://`, `skill://`. + - `packages/coding-agent/src/edit/notebook.ts` — convert `.ipynb` to editable `# %% [...] cell:N` text. + - `packages/coding-agent/src/utils/file-display-mode.ts` — decide hashline vs line-number vs raw display. + - `packages/coding-agent/src/workspace-tree.ts` — render directory trees. + - `packages/coding-agent/src/edit/file-read-cache.ts` — cache read lines for later hashline edit recovery. + - `packages/coding-agent/src/tools/index.ts` — registers `read: s => new ReadTool(s)`. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `path` | `string` | Yes | Filesystem path, internal URL, or web URL. May end with a trailing selector such as `:50-100` or `:raw`. | + +### Selector grammar + +For normal file-like reads, `splitPathAndSel()` in `packages/coding-agent/src/tools/path-utils.ts` recognizes the final suffix only when it matches one of these forms: + +| Suffix | Meaning | +| --- | --- | +| `:raw` | Raw/verbatim mode. Disables structural summaries and line prefixes. | +| `:N` / `:LN` | Start at 1-indexed line `N`, open-ended. | +| `:A-B` / `:LA-LB` | Inclusive 1-indexed line range. | +| `:A+C` / `:LA+LC` | `C` lines starting at `A`; tool converts this to end line `A + C - 1`. | +| `:range:raw` or `:raw:range` | Same line selection, but raw output. | + +Validation in `parseLineRangeChunk()`: +- line numbers are 1-indexed; `:0` throws. +- `+` counts must be `>= 1`. +- `-` end must be `>= start`. + +Selector parsing intentionally falls through for unrecognized trailing `:...`; archive and SQLite paths consume their own colon syntax. + +URL selectors are parsed separately in `packages/coding-agent/src/tools/fetch.ts` and support only `:raw`, `:N`, `:A-B`, and `:A+C` — no optional `L` prefix there. + +## Outputs +- Single-shot `AgentToolResult` built through `toolResult()` in `packages/coding-agent/src/tools/tool-result.ts`. +- `content` is usually one text block. Image reads may return `[text, image]`. +- `details` is path-dependent. `ReadToolDetails` may include: + - `kind: "file" | "url"` (URL path uses `kind: "url"`; file reads usually omit `kind`) + - `isDirectory` + - `resolvedPath` + - `suffixResolution` + - URL fields: `url`, `finalUrl`, `contentType`, `method`, `notes` + - `truncation` + - `displayContent` (unprefixed text + starting line for TUI rendering) + - `summary` (`lines`, `elidedSpans`) for structural summaries + - `meta` from `packages/coding-agent/src/tools/output-meta.ts` +- `details.meta.source` is set to the backing path, URL, or internal URL. +- `details.meta.truncation` carries shown range, total lines/bytes, next offset, and optional `artifactId` for cached URL output. +- Directory/archive listings and SQLite table lists also set `details.meta.limits` when list limits trigger. + +## Flow +1. `ReadTool.execute()` accepts `{ path }`. `file://...` inputs are expanded first with `expandPath()`. +2. It tries URL handling first via `parseReadUrlTarget()` from `packages/coding-agent/src/tools/fetch.ts`. + - Plain URL reads call `executeReadUrl()`. + - URL reads with line selectors load or refresh the URL cache with `loadReadUrlCacheEntry()` and paginate the cached text locally with `#buildInMemoryTextResult()`. +3. If not a web URL, it checks `session.internalRouter.canHandle(...)`. + - Internal URLs are resolved with `internalRouter.resolve()`. + - `agent://` query extraction (`/path` or `?q=`) bypasses pagination and returns the extracted content directly. + - Other internal resources are paginated in-memory by `#buildInMemoryTextResult()`. +4. It tries archive resolution next with `#resolveArchiveReadPath()`. + - `parseArchivePathCandidates()` scans for `.tar`, `.tar.gz`, `.tgz`, or `.zip` anywhere before `:sub/path`. + - On success, `#readArchive()` either lists a directory or decodes an entry as UTF-8 text. +5. It tries SQLite resolution with `#resolveSqliteReadPath()`. + - `parseSqlitePathCandidates()` scans for `.sqlite`, `.sqlite3`, `.db`, `.db3` before any `:table`, `:key`, or `?query` suffix. + - `#readSqlite()` dispatches on `parseSqliteSelector()`. +6. Otherwise it treats the input as a local filesystem path. + - `resolveReadPath()` expands `~`, resolves relative to session cwd, treats bare `/` as session cwd, and retries macOS screenshot/NFD/curly-quote variants. + - If the path does not exist, `findUniqueSuffixMatch()` does a workspace glob-based unique suffix lookup (skipped for remote mounts). +7. Directories go through `#readDirectory()`. +8. Non-directories branch by content type: + - image metadata / inline image + - editable notebook text + - markit-converted document + - structural summary for parseable code/prose + - streamed text/line-range read +9. Local text reads are streamed by `streamLinesFromFile()` rather than loading the whole file. The tool adds up to 3 lines of context before/after explicit bounded ranges. +10. Non-empty contiguous local reads are recorded into `getFileReadCache(session)` for later hashline edit recovery. +11. If suffix resolution happened, the first text block is prefixed with `[Path '...' not found; resolved to '...' via suffix match]`. + +## Modes / Variants + +### Local text files +- No selector: if summarization is enabled and the file is small enough, `#trySummarize()` calls `summarizeCode()`. + - Guards: file size `<= 2 MiB` (`MAX_SUMMARY_BYTES`), line count `<= 20_000` (`MAX_SUMMARY_LINES`). + - Summary output keeps selected declarations and replaces elided spans with `...`. + - When an elided block sits between matching brace lines, `#renderSummary()` may merge them into one anchored line rather than emitting separate opener/closer lines. +- Explicit selector or summarization miss: streamed text read. + - Default open-ended limit is `min(session setting read.defaultLimit, DEFAULT_MAX_LINES)`. + - Explicit ranges expand by `RANGE_CONTEXT_LINES = 3` on the constrained sides only. + - Non-raw output uses `resolveFileDisplayMode()`: + - hashline anchors when edit mode is hashline, read is not raw, source is mutable, edit tool exists, and `readHashLines !== false` + - otherwise optional line numbers when `readLineNumbers === true` + - raw mode suppresses both +- Prefix format in hashline mode is `lineNumber + 2-char line hash + "|"`, e.g. `41th|def alpha():`, from `formatHashLine()` in `packages/coding-agent/src/hashline/hash.ts`. +- Those anchors are what the `edit`/hashline path consumes later; immutable sources and `:raw` intentionally suppress them. + +### Directory listings +- `#readDirectory()` calls `buildDirectoryTree()` with: + - `maxDepth = 2` + - `perDirLimit = 12` + - `rootLimit = null` + - `lineCap = limit` when a line selector was present, else unlimited at this layer +- `buildDirectoryTree()` sorts siblings by recency, shows file sizes and relative ages, and may mark `limits.resultLimit` when the tree truncates. +- Empty directories render as `(empty directory)`. + +### Archives +- Supported archive containers: `.tar`, `.tar.gz`, `.tgz`, `.zip`. +- Syntax: `archive.ext`, `archive.ext:path/inside`, `archive.ext:path/inside:50-60`. +- `openArchive()` reads the whole archive into memory, then: + - tar/tgz uses `new Bun.Archive(bytes)` + - zip uses `fflate.unzipSync()` +- Archive paths normalize `/`, drop `.` segments, and reject `..`. +- Directory reads list immediate children; files show `name` plus ` (size)` when size > 0. +- Directory listing default limit is `500` entries in `#readArchiveDirectory()`. +- File entries are UTF-8 decoded. Non-UTF-8 entries return `[Cannot read binary archive entry '...' (...)]` instead of bytes. +- Text archive entries reuse the normal in-memory pagination/anchoring path. + +### SQLite databases +- Database detection requires both a matching extension and a valid SQLite file header (`isSqliteFile()`). +- Selector forms from `parseSqliteSelector()`: + +#### `db.sqlite` +- `kind: "list"` +- Lists non-`sqlite_%` tables with row counts. +- `#readSqlite()` caps the rendered list to `500` tables via `applyListLimit()`. + +#### `db.sqlite:table` +- `kind: "schema"` +- Returns `sqlite_master.sql` plus sample rows. +- Sample size is `DEFAULT_SCHEMA_SAMPLE_LIMIT = 5`. + +#### `db.sqlite:table:key` +- `kind: "row"` +- Resolves by primary key when the table has exactly one PK column; otherwise falls back to `rowid` lookup. +- No query parameters allowed on row lookups. + +#### `db.sqlite:table?limit=...&offset=...&order=...&where=...` +- `kind: "query"` +- Defaults: `limit = 20`, `offset = 0`. +- `limit` is capped at `500`. +- `order` accepts `column` or `column:asc|desc` and must name an existing column. +- `where` is accepted only after `validateWhereClause()` rejects comments, semicolons, and control keywords like `LIMIT`, `OFFSET`, `UNION`, `ATTACH`, `PRAGMA`. +- Unknown query parameters throw. + +#### `db.sqlite?q=SELECT ...` +- `kind: "raw"` +- Cannot be combined with table selectors or any other query param. +- Empty `q` throws. +- `executeReadQuery()` runs `db.prepare(sql).all()` and rejects bound parameters; it does not verify that the SQL starts with `SELECT`. + +- Rendering caps in `packages/coding-agent/src/tools/sqlite-reader.ts`: + - ASCII table width `120` (`MAX_RENDER_WIDTH`) + - per-column width `40` (`MAX_COLUMN_WIDTH`) +- `#readSqlite()` opens Bun SQLite in `{ readonly: true, strict: true }` and sets `PRAGMA busy_timeout = 3000`. + +### Documents +- `CONVERTIBLE_EXTENSIONS` in `packages/coding-agent/src/tools/read.ts` covers `.pdf`, `.doc`, `.docx`, `.ppt`, `.pptx`, `.xls`, `.xlsx`, `.rtf`, `.epub`. +- `convertFileWithMarkit()` converts the file to text/markdown. +- Converted output is then head-truncated with normal shared limits; there is no line selector support inside the source document before conversion. +- Conversion failures return a text block like `[Cannot read .pdf file: ...]`. + +### Jupyter notebooks +- `.ipynb` goes through `readEditableNotebookText()` unless `:raw` was requested. +- Output is editable plain text with markers like: + +```text +# %% [code] cell:0 +... +``` + +- Raw mode bypasses that conversion and falls back to file-text reading. + +### Images +- Image detection is metadata-based (`readImageMetadata()`). +- Max accepted image size is `20 MiB` (`MAX_IMAGE_INPUT_BYTES`, re-exported as `MAX_IMAGE_SIZE`). Larger files throw. +- If `inspect_image.enabled` is true, `read` returns metadata only (MIME, bytes, dimensions, channels, alpha) plus a suggestion to call `inspect_image`. +- Otherwise it calls `loadImageInput()` and returns: + - a text note from the image loader + - an inline image block +- Unsupported/undecodable image formats throw a `ToolError`. + +### Internal URLs +- `read` does not resolve these itself; it delegates to `session.internalRouter.resolve()`. +- Registered protocols are outside this file, but the router in `packages/coding-agent/src/internal-urls/router.ts` is built for `agent://`, `artifact://`, `jobs://`, `local://`, `mcp://`, `memory://`, `pi://`, `rule://`, and `skill://`. +- `#handleInternalUrl()` behavior: + - parses the URL with `parseInternalUrl()` so colons inside the host segment are legal + - for `agent://`, treats non-root path extraction or `?q=` extraction as a special no-pagination mode + - otherwise paginates the resolved text in memory + - passes `immutable` through to `resolveFileDisplayMode()` so anchors are suppressed for immutable resources such as artifacts, skills, memory, and agent outputs + - sets `ignoreResultLimits: true` for `skill://` so the full skill text is paginated only by explicit selectors, not by the normal default line limit + +### Web URLs +- `parseReadUrlTarget()` accepts `http://`, `https://`, or `www.` targets. +- Plain URL reads call `executeReadUrl()` in `packages/coding-agent/src/tools/fetch.ts`. +- `:raw` means raw HTML/body fallback path; plain URL reads prefer rendered/reader-friendly output. +- `:N`, `:A-B`, `:A+C` do not refetch. They page over cached output from the prior or current URL render. +- URL render pipeline in `renderUrl()`: + 1. normalize scheme (`https://` added for bare `www.`) + 2. try special handlers for known sites unless raw + 3. fetch with `loadPage()` + 4. if content is image/PDF/DOCX/etc., try binary fetch + markit/image handling + 5. handle JSON directly, feeds via feed parser, plain text directly + 6. for HTML and non-raw mode, try markdown alternates, `URL.md`, content negotiation, feed alternates, HTML-to-text renderers, extracted linked documents, then `llms.txt` + 7. fall back to raw body text/html +- URL output is wrapped with a small header: + +```text +URL: ... +Content-Type: ... +Method: ... +Notes: ... + +--- +``` + +- `method` records the winning path (`json`, `feed`, `text`, `alternate-markdown`, `md-suffix`, `content-negotiation`, `image`, `markit`, `llms.txt`, `raw`, `raw-html`, etc.). +- URL reads may return an inline image block when the fetched resource is a supported image and survives resizing. + +## Side Effects +- Filesystem + - Opens and streams local files. + - Reads entire archives into memory before indexing. + - May read URL-cache artifact files from the session artifacts directory. + - Writes URL output artifacts when URL output is truncated or when line-range pagination needs a persisted cache body. +- Network + - URL mode performs HTTP fetches, binary refetches, and alternate-endpoint probes. +- Subprocesses / native bindings + - Uses Bun SQLite for `.db`/`.sqlite*`. + - Uses `Bun.Archive` for tar/tgz and `fflate` for zip. + - URL HTML rendering can delegate into site handlers and HTML-to-text backends from `packages/coding-agent/src/tools/fetch.ts`. +- Session state + - Records local text lines into `session.fileReadCache` for later stale-anchor recovery. + - Uses `session.internalRouter` for internal URLs. + - Uses `session.allocateOutputArtifact()` for cached/truncated URL output. +- Background work / cancellation + - Most branches honor `AbortSignal`; the tool itself is marked `nonAbortable = true`, but helper paths still call `throwIfAborted(signal)`. + +## Limits & Caps +- Shared text truncation defaults from `packages/coding-agent/src/session/streaming-output.ts`: + - `DEFAULT_MAX_LINES = 3000` + - `DEFAULT_MAX_BYTES = 50 * 1024` +- Local text open-ended default line limit: `read.defaultLimit`, clamped to `[1, DEFAULT_MAX_LINES]`. +- Explicit line ranges add `3` context lines on each constrained side (`RANGE_CONTEXT_LINES`). +- File streaming chunk size: `8 * 1024` bytes (`READ_CHUNK_SIZE`). +- Local streamed byte budget for line reads: `max(DEFAULT_MAX_BYTES, maxLinesToCollect * 512)`. +- Structural summaries only run when file size `<= 2 MiB` and line count `<= 20_000`. +- Image input max: `20 MiB`. +- Directory tree caps for local directories: depth `2`, per-directory children `12`. +- Archive directory default list cap: `500` entries. +- SQLite: + - default row query limit `20` + - schema sample limit `5` + - max query limit `500` + - table list cap `500` + - render width `120`, column width `40` + - busy timeout `3000` ms +- URL read result shown to the model is truncated to `300` lines and `50 KiB` in `executeReadUrl()`; full cached output can be attached as an artifact. +- Inline fetched URL images: + - source bytes cap `20 MiB` + - post-resize inline output cap `300 KiB` +- Unique suffix auto-resolution glob timeout: `5000` ms. +- File-read cache holds `30` paths per session. + +## Errors +- Validation and operational failures surface as `ToolError`. +- Selector errors include: + - `Line selector 0 is invalid; lines are 1-indexed. Use :1.` + - invalid `A+B` / `A-B` shapes + - `Cannot combine query extraction with offset/limit` for `agent://.../path:50` +- Missing local/archive/sqlite paths first attempt unique suffix resolution; if no unique match exists they error. +- Out-of-bounds line reads do not throw. They return explanatory text with a suggestion such as `Use :1 ...` or `Use :<last line> ...`. +- Binary archive entries do not throw; they return a text notice. +- Document conversion failure returns a text notice. +- Image oversize/unsupported/invalid cases throw. +- SQLite parser rejects unsupported parameter combinations early; DB/runtime errors are caught and rethrown as `ToolError(message)`. +- URL fetch failure does not throw when HTTP fetch succeeds but `response.ok === false`; it returns a failed URL read with `method: "failed"` and explanatory notes. + +## Notes +- `readSchema` examples include `https://example.com:L1-L40`, but URL selector parsing in `packages/coding-agent/src/tools/fetch.ts` does not accept `L` prefixes. +- Hashline anchors are suppressed for raw reads and immutable internal resources because there is no editable backing target for later `edit` consumption. +- `splitPathAndSel()` intentionally treats unknown trailing `:...` as part of the path so `archive.zip:inner/file` and `db.sqlite:table:key` still work. +- `resolveReadPath()` contains macOS-specific filename fallbacks for screenshot timestamps, NFD Unicode normalization, and curly apostrophes. +- A bare `/` resolves to the session cwd, not the filesystem root. +- URL cache keys are session-scoped and normalized by requested URL + raw/rendered mode; both requested URL and final redirected URL are cached. +- URL line-range reads request `ensureArtifact: true, preferCached: true` so a later paginated read can reopen the same rendered body from artifact storage. +- Raw SQLite `q=` execution is not keyword-restricted beyond “no bound parameters”; the read tool relies on the surrounding contract to keep it read-only. +- The file-read cache is not a read acceleration cache. It exists to recover hashline edits when the file changed after the read. \ No newline at end of file diff --git a/docs/tools/recall.md b/docs/tools/recall.md new file mode 100644 index 000000000..e86f2fbef --- /dev/null +++ b/docs/tools/recall.md @@ -0,0 +1,79 @@ +# recall + +> Search the active Hindsight bank and return raw matching memories. + +## Source +- Entry: `packages/coding-agent/src/tools/hindsight-recall.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/recall.md` +- Key collaborators: + - `packages/coding-agent/src/hindsight/state.ts` — session state, recall query defaults, prompt-side auto-recall. + - `packages/coding-agent/src/hindsight/content.ts` — result formatting and UTC timestamp formatting. + - `packages/coding-agent/src/hindsight/client.ts` — HTTP `recall` call and error mapping. + - `packages/coding-agent/src/hindsight/bank.ts` — bank id and tag-filter scoping. + - `docs/tools/retain.md` — shared backend, storage, seeding, and mental-model bootstrap. + +## Inputs + +| Field | Type | Required | Description | +|---|---|---:|---| +| `query` | `string` | Yes | Natural-language search query. The tool passes it through unchanged. | + +## Outputs +Returns a single-shot tool result. + +When matches exist: +- `content[0].type = "text"` +- `content[0].text = "Found <n> relevant memories (as of YYYY-MM-DD HH:MM UTC):\n\n<bullet list>"` +- each bullet is `- <text> [<type>] (<mentioned_at>)`; the type and timestamp suffixes appear only when those fields are present +- `details = {}` + +When no matches exist: +- `content[0].text = "No relevant memories found."` +- `details = {}` + +## Flow +1. `HindsightRecallTool.createIf(...)` only exposes the tool when `memory.backend == "hindsight"`. +2. `execute(...)` wraps the whole operation in `untilAborted(...)` from `@oh-my-pi/pi-utils`. +3. It reads the active `HindsightSessionState`; missing state throws `Hindsight backend is not initialised for this session.` +4. It calls `state.client.recall(...)` with: + - `bankId` from session bootstrap, + - the model-supplied `query`, + - `budget`, `maxTokens`, and `types` from `HindsightConfig`, + - tag filters from the bank scope (`recallTags`, `recallTagsMatch`). +5. `HindsightApi.recall(...)` POSTs `/v1/default/banks/{bank_id}/memories/recall`. +6. Results are formatted into a plain-text list with `formatMemories(...)`; empty results map to the fixed no-match string. +7. Failures are logged with `logger.warn("recall failed", ...)` and rethrown. + +## Modes / Variants +- Tool path: explicit query-only recall. The tool does not compose context from recent turns; that richer path is reserved for backend auto-recall in `HindsightSessionState.beforeAgentStartPrompt(...)` / `maybeRecallOnAgentStart(...)`. +- Bank scoping is inherited from the active `HindsightSessionState`: + - `global` — no tag filter. + - `per-project` — separate bank id per cwd basename. + - `per-project-tagged` — shared bank id plus `project:<cwd basename>` filter with `tagsMatch = "any"`, so project-tagged and untagged global memories can both surface. +- Session scope: reads cross-session server-side memories, but uses per-session cached config and scope. + +## Side Effects +- Network + - `POST /v1/default/banks/{bank_id}/memories/recall` via `packages/coding-agent/src/hindsight/client.ts`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - None on success. Unlike backend auto-recall, this tool does not update `lastRecallSnippet` or refresh the system prompt. +- Background work / cancellation + - Aborts through `untilAborted(...)` if the tool call signal is cancelled. + +## Limits & Caps +- Client default budget for raw `HindsightApi.recall(...)` is `"mid"`; this tool overrides from config in `packages/coding-agent/src/hindsight/state.ts`. +- Default recall settings from `packages/coding-agent/src/config/settings-schema.ts`: + - `hindsight.recallBudget = "mid"` + - `hindsight.recallMaxTokens = 1024` + - `hindsight.recallTypes = ["world", "experience"]` +- The explicit tool path does not apply `hindsight.recallContextTurns` or `hindsight.recallMaxQueryChars`; those caps only affect backend auto-recall query composition. + +## Errors +- Throws `Hindsight backend is not initialised for this session.` when no state exists. +- HTTP and fetch failures become `HindsightError` from `packages/coding-agent/src/hindsight/client.ts` with `statusCode` and parsed `details` when available. +- Non-`Error` failures are normalized to `new Error(String(err))` before rethrow. + +## Notes +- Shared backend details are in `docs/tools/retain.md`: server-side storage, subagent aliasing, bank scoping, mission setup, and mental-model bootstrap. +- Mental models are not fetched by this tool. They may still already be present in the agent's developer instructions because the backend caches a `<mental_models>` block separately from recall results. +- The tool returns raw memory hits; it does not synthesize across them. Use `reflect` for that path. diff --git a/docs/tools/recipe.md b/docs/tools/recipe.md new file mode 100644 index 000000000..9589c670e --- /dev/null +++ b/docs/tools/recipe.md @@ -0,0 +1,155 @@ +# recipe + +> Run a task exposed by a detected project task runner. + +## Source +- Entry: `packages/coding-agent/src/tools/recipe/index.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/recipe.md` +- Key collaborators: + - `packages/coding-agent/src/tools/recipe/runner.ts` — op parsing, task resolution, prompt model. + - `packages/coding-agent/src/tools/recipe/render.ts` — shell-style call/result rendering. + - `packages/coding-agent/src/tools/recipe/runners/index.ts` — runner registration order. + - `packages/coding-agent/src/tools/recipe/runners/just.ts` — detect `just` recipes from justfiles. + - `packages/coding-agent/src/tools/recipe/runners/pkg.ts` — detect `package.json` scripts and workspaces. + - `packages/coding-agent/src/tools/recipe/runners/cargo.ts` — detect Cargo run/test targets. + - `packages/coding-agent/src/tools/recipe/runners/make.ts` — parse make targets from makefiles. + - `packages/coding-agent/src/tools/recipe/runners/task.ts` — detect Taskfile tasks via `task --list-all`. + - `packages/coding-agent/src/tools/bash.ts` — actual command execution, truncation, cwd/env handling. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `op` | `string` | Yes | Single string containing the task selector plus trailing arguments. The first whitespace-delimited token selects the task; the remainder is appended verbatim to the resolved runner command. Examples from schema/prompt: `test`, `build --release`, `pkg-a/test`, `crate/bin/server`, `pkg:test --watch`. | + +### `op` grammar + +```text +op := S* head (S+ tail)? +head := explicit-runner / implicit-task +explicit-runner := runner-id ":" task-token +implicit-task := task-token +runner-id := detected runner id (`just` | `pkg` | `cargo` | `make` | `task`) +task-token := first non-whitespace token; may contain `/` +tail := remaining characters after the first whitespace run +``` + +Resolution rules from `resolveRunnerAndTask()`: +- Leading whitespace is ignored; an empty `op` throws `ToolError` with the available task list. +- Only the first token is parsed structurally. Everything after the first whitespace run becomes `tail` and is appended to the command unchanged. +- If `head` contains `:` and the prefix matches a detected runner id, the suffix must exactly match a task in that runner. +- Otherwise `head` is treated as a task name and matched across all detected runners. +- If exactly one runner has that task, it is used. +- If multiple runners have that task, the call is rejected and the error tells the model to use `<runner-id>:<task>`. +- Namespaced task names generated by runners use `/`, not `:`. `/` is part of the task name, not a parser separator. + +## Outputs +- Delegates directly to `BashTool.execute()` and returns the same `AgentToolResult<BashToolDetails>` shape. +- Success path: one text content block containing merged command output (`result.output` from bash execution, or `(no output)`), plus any timeout clamp notice appended after a blank line. +- Recipe does not return separate `stdout`, `stderr`, or `exitCode` fields. `stdout`/`stderr` are already merged into the text block by bash execution; `exitCode` is only observed indirectly (success requires `0`, non-zero becomes an error). +- Error path: throws `ToolError`; for non-zero exits the message is the merged output followed by `Command exited with code <n>`. +- `details` may include: + - `timeoutSeconds`: effective timeout used by bash. + - `requestedTimeoutSeconds`: only when bash clamped a requested timeout; recipe never sets one itself. + - `meta`: output truncation metadata from bash execution. + - `async`: defined by bash background execution paths, but recipe does not expose an `async` input. +- When bash output is truncated, the full text is stored in an artifact and referenced via bash truncation metadata. +- Call/result rendering in the TUI uses bash shell rendering with a resolved title, command preview, and optional task cwd. + +## Flow +1. `RecipeTool.createIf()` in `packages/coding-agent/src/tools/recipe/index.ts` checks `session.settings.get("recipe.enabled")`; disabled returns `null`. +2. It probes every runner in `RUNNERS` from `packages/coding-agent/src/tools/recipe/runners/index.ts` with `Promise.all(...)` in this order: `just`, `pkg`, `cargo`, `make`, `task`. +3. Each runner returns either `null` or a `DetectedRunner { id, label, commandPrefix, tasks }`; runners with zero tasks are discarded. +4. If no runners remain, the tool is not registered. +5. Constructor stores detected runners, instantiates `BashTool`, renders the model-facing description by passing `buildPromptModel(runners)` into `packages/coding-agent/src/prompts/tools/recipe.md`, and builds shell renderers from `createRecipeToolRenderer()`. +6. On execution, `RecipeTool.execute()` calls `resolveCommand(op, this.#runners)`. +7. `resolveCommand()` in `packages/coding-agent/src/tools/recipe/runner.ts`: + 1. `parseOp()` trims only leading whitespace, extracts the first non-whitespace token as `head`, and keeps the remainder as `tail`. + 2. `resolveRunnerAndTask()` resolves `head` either as `runnerId:taskName` or as an unqualified task name. + 3. It throws `ToolError` for empty ops, missing explicit tasks, ambiguous task names, or unknown tasks; all error variants include the available task list. + 4. It builds the final shell command with `buildCommand(commandPrefix, commandName, tail)`, joining non-empty parts with spaces. + 5. If the task defines `cwd`, that relative path is returned alongside the command. +8. `RecipeTool.execute()` forwards `{ command, cwd }` into `BashTool.execute()`; recipe does not pass timeout, env, async, or pty options. +9. `BashTool.execute()` resolves internal URLs, validates/normalizes cwd against `session.cwd`, clamps timeout, applies bash interception rules, runs the command, and formats the final result. + +## Modes / Variants +- Tool enablement: + - Disabled by `recipe.enabled` setting: tool is absent. + - Enabled but no detected tasks: tool is absent. +- Task selection: + - Unqualified task name: succeeds only when exactly one detected runner owns that task. + - Explicit runner-qualified task: `<runner-id>:<task>`. +- Runner detection paths: + - `just`: requires `just` on `PATH`, a justfile, and successful `just --dump --dump-format=json`. + - `pkg`: requires a readable root `package.json`; picks a package manager command from lockfiles or `bun` availability; discovers root scripts and workspace package scripts. + - `cargo`: requires `cargo` on `PATH`, `Cargo.toml`, and successful `cargo metadata --no-deps --format-version=1`. + - `make`: requires `make` on `PATH` and a makefile; parses targets statically. + - `task`: requires `task` on `PATH`, a Taskfile, and successful `task --list-all --json`. +- Execution path: + - Always the synchronous `bash` call surface from recipe inputs. + - Bash may still auto-background long-running work if `bash.autoBackground.enabled` and session async job support are enabled. + +## Side Effects +- Filesystem + - Reads manifests from the session cwd during detection: justfiles, `package.json`, workspace `package.json` files, `Cargo.toml`, makefiles, `Taskfile.yml` / `Taskfile.yaml`. + - Command execution runs in `session.cwd` or a task-specific relative cwd resolved under it. + - Bash may allocate output artifacts for truncated command output. +- Subprocesses / native bindings + - Detection may spawn `just --dump --dump-format=json`, `cargo metadata --no-deps --format-version=1`, and `task --list-all --json`. + - Execution spawns the resolved shell command through `BashTool` / `executeBash()`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Tool availability depends on session settings. + - Constructor prompt text is specialized to detected runners/tasks. + - Bash execution may create async job records and output artifacts if bash auto-background triggers. +- User-visible prompts / interactive UI + - The model-facing tool description lists detected runners and up to 20 tasks per runner. + - TUI rendering shows a shell-style preview using the resolved title/command/cwd. +- Background work / cancellation + - Detection is parallelized across runners. + - Runtime command execution honors the passed abort signal through `BashTool`. + +## Limits & Caps +- Prompt task listing is capped at `PROMPT_TASK_LIMIT = 20` per runner in `packages/coding-agent/src/tools/recipe/runner.ts`; this affects the rendered tool description, not execution. +- Recipe itself defines no timeout input; delegated bash execution therefore uses bash's default `timeout = 300` seconds from `packages/coding-agent/src/tools/bash.ts`. +- Bash clamps timeouts to the configured bash range (`clampTimeout("bash", ...)` in `packages/coding-agent/src/tools/bash.ts`), but recipe cannot request a custom value. +- `pkg` workspace discovery normalizes workspace globs to `.../package.json` and sorts matched package files lexicographically before task generation. +- `cargo` deduplicates generated task names with a `Set`, so duplicate targets collapse to one recipe task. + +## Errors +- Detection failures in runner modules are mostly soft-failed: + - Missing binaries, missing manifests, parse failures, or non-zero probe exits usually return `null` and log with `logger.debug(...)`. + - Result: the affected runner disappears instead of surfacing an error to the model. +- Invocation failures are hard errors from `resolveRunnerAndTask()`: + - Empty `op`. + - Explicit runner prefix with missing/empty task. + - Ambiguous unqualified task name across runners. + - Unknown task name. +- Execution failures come from `BashTool.execute()`: + - Invalid cwd. + - Bash interceptor blocks. + - Aborts/timeouts. + - Non-zero exit codes. + - Missing exit status. +- All `resolveRunnerAndTask()` errors include the current available task list to help the model retry. + +## Notes +- `RecipeTool` sets `concurrency = "exclusive"`; calls do not run concurrently with other exclusive tools. +- Tool registration is all-or-nothing per runner: a detected runner with zero tasks is dropped. +- Runner ids are fixed string literals from the runner modules: `just`, `pkg`, `cargo`, `make`, `task`. +- `buildPromptModel()` includes each task's rendered command (`commandPrefix` + `commandName`) and relative cwd when present; the prompt therefore exposes the exact shell form recipe will run. +- `pkg` task names: + - Root `package.json` scripts keep bare names like `test`. + - Workspace scripts are always namespaced as `<package-name-or-dir>/<script>` and set `cwd` to that package directory. + - Script names are shell-quoted into `commandName`, so a task like `build` becomes `bun run 'build'` / `npm run 'build'` / similar. +- `pkg` command prefix selection prefers lockfiles in this order: `bun.lock`/`bun.lockb`, `pnpm-lock.yaml`, `yarn.lock`, `package-lock.json`/`npm-shrinkwrap.json`; otherwise it falls back to `bun run` if `bun` exists, else `npm run`. +- `cargo` task names are generated from metadata targets: + - Single-package manifests: `bin/<name>`, `example/<name>`, `test/<name>`. + - Multi-package workspaces: `<package>/bin/<name>`, `<package>/example/<name>`, `<package>/test/<name>`. + - Each task overrides `commandPrefix` to the full `cargo run ... --bin|--example` or `cargo test ... --test` prefix, and `commandName` to the quoted target name. +- `make` target parsing is static text parsing, not `make -qp` output: + - Recognizes makefiles named `Makefile`, `makefile`, `GNUmakefile`. + - Uses `.PHONY` lines to decide whether to include undocumented file targets; without any `.PHONY`, all parsed targets are exposed. + - If `.PHONY` exists, documented non-phony targets are kept with ` (file target)` appended to `doc`. +- `just` detection ignores private recipes and preserves declared parameter names only for prompt display; execution still accepts arbitrary `tail` text. +- `task` detection uses `desc` first, then `summary`, for task documentation. +- Recipe has no env input of its own. Commands inherit whatever environment `BashTool` supplies for normal bash execution in the session. \ No newline at end of file diff --git a/docs/tools/reflect.md b/docs/tools/reflect.md new file mode 100644 index 000000000..18286d418 --- /dev/null +++ b/docs/tools/reflect.md @@ -0,0 +1,73 @@ +# reflect + +> Ask the Hindsight server to synthesize an answer over the active memory bank. + +## Source +- Entry: `packages/coding-agent/src/tools/hindsight-reflect.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/reflect.md` +- Key collaborators: + - `packages/coding-agent/src/hindsight/bank.ts` — best-effort bank mission initialization. + - `packages/coding-agent/src/hindsight/state.ts` — session state, shared bank scope, recall/reflect config. + - `packages/coding-agent/src/hindsight/client.ts` — HTTP `reflect` call and error mapping. + - `docs/tools/retain.md` — shared backend, storage, seeding, and mental-model bootstrap. + +## Inputs + +| Field | Type | Required | Description | +|---|---|---:|---| +| `query` | `string` | Yes | Question to answer from long-term memory. | +| `context` | `string` | No | Extra guidance sent to the Hindsight reflect endpoint. | + +## Outputs +Returns a single-shot tool result: + +- `content[0].type = "text"` +- `content[0].text = response.text?.trim() || "No relevant information found to reflect on."` +- `details = {}` + +The tool returns the Hindsight server's synthesized text directly; it does not expose raw recall hits. + +## Flow +1. `HindsightReflectTool.createIf(...)` only exposes the tool when `memory.backend == "hindsight"`. +2. `execute(...)` runs under `untilAborted(...)`. +3. It reads the active `HindsightSessionState`; missing state throws `Hindsight backend is not initialised for this session.` +4. Before reflecting, it calls `ensureBankMission(...)` with the current `bankId`, config, and process-local `missionsSet`. +5. `ensureBankMission(...)` best-effort `PUT`s `/v1/default/banks/{bank_id}` with `reflect_mission` and optional `retain_mission` exactly once per bank/process; failures are swallowed. +6. It calls `state.client.reflect(...)` with the model `query`, optional `context`, configured recall budget, and bank-scope tag filters. +7. `HindsightApi.reflect(...)` POSTs `/v1/default/banks/{bank_id}/reflect` and defaults its own budget to `"low"` when callers omit one; this tool always passes the configured budget. +8. Blank or whitespace-only responses are replaced with `No relevant information found to reflect on.` +9. Failures are logged with `logger.warn("reflect failed", ...)` and rethrown. + +## Modes / Variants +- Tool path: one reflect request, optionally focused by `context`. +- Bank scoping is inherited from the active `HindsightSessionState`: + - `global` — no tag filter. + - `per-project` — separate bank id per cwd basename. + - `per-project-tagged` — shared bank id plus `project:<cwd basename>` filter with `tagsMatch = "any"`. +- Session scope: reads cross-session server-side memories, but does not persist local output. + +## Side Effects +- Network + - Optional `PUT /v1/default/banks/{bank_id}` from `ensureBankMission(...)`. + - `POST /v1/default/banks/{bank_id}/reflect` via `packages/coding-agent/src/hindsight/client.ts`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Reads session-held bank scope and config only. Does not update `lastRecallSnippet`, the mental-model cache, or the retain queue. +- Background work / cancellation + - Aborts through `untilAborted(...)` if the tool call signal is cancelled. + +## Limits & Caps +- Tool-level params: only `query` is required; `context` is optional. +- Default budget setting comes from `hindsight.recallBudget` in `packages/coding-agent/src/config/settings-schema.ts`; default `"mid"`. +- `reflect` itself has no client-side token cap parameter here; unlike `recall`, the tool does not pass `maxTokens`. +- Mission initialization tracks up to `MISSION_SET_CAP = 10_000` bank ids in `packages/coding-agent/src/hindsight/bank.ts`, then drops the oldest half of the sorted set. + +## Errors +- Throws `Hindsight backend is not initialised for this session.` when no state exists. +- HTTP and fetch failures become `HindsightError` from `packages/coding-agent/src/hindsight/client.ts` with `statusCode` and parsed `details` when available. +- `ensureBankMission(...)` failures are silent to the tool caller; only the later reflect request can fail visibly. +- Non-`Error` failures are normalized to `new Error(String(err))` before rethrow. + +## Notes +- Shared backend details are in `docs/tools/retain.md`: server-side storage, subagent aliasing, bank scoping, seed mental models from `packages/coding-agent/src/hindsight/seeds.json`, and mental-model prompt injection. +- `reflect` does not read the cached `<mental_models>` block directly. It queries the Hindsight server over the bank contents. The same session may also have separate mental-model context injected into its developer instructions. +- Reflect mission and retain mission are bank-level server settings, not per-request payload. The tool just ensures they are present best-effort before reflecting. diff --git a/docs/tools/render_mermaid.md b/docs/tools/render_mermaid.md new file mode 100644 index 000000000..27524e108 --- /dev/null +++ b/docs/tools/render_mermaid.md @@ -0,0 +1,82 @@ +# render_mermaid + +> Convert Mermaid source into terminal-friendly ASCII/Unicode text. + +## Source +- Entry: `packages/coding-agent/src/tools/render-mermaid.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/render-mermaid.md` +- Key collaborators: + - `packages/utils/src/mermaid-ascii.ts` — thin wrapper over renderer package. + - `packages/coding-agent/src/tools/index.ts` — tool registration and enablement gate. + - `packages/coding-agent/src/sdk.ts` — session-facing artifact allocation hook. + - `packages/coding-agent/src/session/session-manager.ts` — persistent-session artifact path allocation. + - `packages/coding-agent/src/session/artifacts.ts` — artifact filename generation and writes. +- Related user/runtime doc: `docs/render-mermaid.md` + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `mermaid` | `string` | Yes | Mermaid source text. Schema example: `graph TD; A-->B`. | +| `config` | `object` | No | Optional renderer options. Sanitized before rendering; numeric fields are floored and clamped to `>= 0`. | + +`config` fields: + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `useAscii` | `boolean` | No | `true` for plain ASCII, `false`/omitted for Unicode box-drawing output. Passed through unchanged. | +| `paddingX` | `number` | No | Horizontal spacing. `Math.floor`, then `Math.max(0, value)`. | +| `paddingY` | `number` | No | Vertical spacing. `Math.floor`, then `Math.max(0, value)`. | +| `boxBorderPadding` | `number` | No | Inner box padding. `Math.floor`, then `Math.max(0, value)`. | + +## Outputs +The tool returns a single text content block: + +- inline body: rendered diagram text +- optional trailer: `Saved artifact: artifact://<id>` when artifact storage is available + +`details` may include: + +- `artifactId?: string` + +No image path, SVG, PNG, or binary payload is returned. Stored artifacts are plain text `.log` files; artifact filenames are allocated as `<id>.render_mermaid.log` by `packages/coding-agent/src/session/artifacts.ts`. + +## Flow +1. `RenderMermaidTool.execute()` in `packages/coding-agent/src/tools/render-mermaid.ts` receives `mermaid` and optional `config`. +2. `sanitizeRenderConfig()` normalizes `paddingX`, `paddingY`, and `boxBorderPadding` to non-negative integers; `useAscii` is passed through. +3. The tool calls `renderMermaidAscii()` from `@oh-my-pi/pi-utils`. +4. `packages/utils/src/mermaid-ascii.ts` forwards directly to `renderMermaidASCII()` from the `beautiful-mermaid` package. +5. The tool optionally asks the session for an artifact slot with `allocateOutputArtifact("render_mermaid")`. +6. If a path is returned, `Bun.write()` persists the full rendered text to that file. +7. The tool returns the rendered text, plus an `artifact://` line and `details.artifactId` when persistence succeeded. + +## Modes / Variants +- Default render: Unicode box-drawing output when `config.useAscii` is omitted or false. +- ASCII render: plain ASCII output when `config.useAscii` is true. +- Persistent-session path: artifact text is written when `allocateOutputArtifact()` returns a path. +- Ephemeral-session path: no artifact is written; the inline text result is still returned. + +## Side Effects +- Filesystem + - May write one session artifact via `Bun.write()`. + - Artifact filename format is `<id>.render_mermaid.log`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Consumes the session artifact allocator hook. + - Returns `details.artifactId` for the tool result. + +## Limits & Caps +- No tool-local timeout, retry, truncation, or streaming path. +- Numeric config fields are quantized to integers with `Math.floor()` and clamped to `0` minimum in `sanitizeRenderConfig()`. +- Renderer engine is `beautiful-mermaid@1.1.3` per root `package.json` / `bun.lock`. +- The tool is registered as discoverable and gated by `renderMermaid.enabled` in `packages/coding-agent/src/tools/index.ts`. + +## Errors +- `renderMermaidAscii()` is not wrapped in a local `try/catch`; renderer exceptions propagate out of `execute()`. +- Invalid Mermaid syntax therefore fails the tool call rather than returning partial output. +- Artifact allocation failures inside the SDK hook are swallowed there and converted to `{}` in `packages/coding-agent/src/sdk.ts`; rendering still succeeds, just without a saved artifact. +- Artifact write failures from `Bun.write()` are not caught in the tool and will fail the call. + +## Notes +- The tool summary string says `Render a Mermaid diagram to an image`, but the implementation and prompt both produce text, not images. +- Despite the name, this tool does not use Puppeteer, browser rendering, Mermaid CLI, or native bindings; rendering stays in-process through the JS package wrapper. +- `docs/render-mermaid.md` covers operator-facing behavior and enablement; keep this file focused on the tool contract and runtime path. diff --git a/docs/tools/resolve.md b/docs/tools/resolve.md new file mode 100644 index 000000000..6ec23dcdb --- /dev/null +++ b/docs/tools/resolve.md @@ -0,0 +1,72 @@ +# resolve + +> Finalizes a queued preview action by applying or discarding it. + +## Source +- Entry: `packages/coding-agent/src/tools/resolve.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/resolve.md` +- Key collaborators: + - `docs/resolve-tool-runtime.md` — preview/apply runtime reference + - `packages/coding-agent/src/extensibility/custom-tools/loader.ts` — forwards custom pending actions into the queue + - `packages/coding-agent/src/tools/ast-edit.ts` — built-in preview producer example + - `packages/coding-agent/src/session/agent-session.ts` — tool-choice queue and invoker access + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `action` | `"apply" | "discard"` | Yes | Whether to commit or reject the queued preview. | +| `reason` | `string` | Yes | Required explanation passed through to the queued callback. | + +## Outputs +- Single-shot result. +- `execute()` returns whatever the queued invoker returns, with `details` wrapped/augmented to include: + - `action` + - `reason` + - `sourceToolName?` + - `label?` + - `sourceResultDetails?` — original `result.details` from the apply/reject callback when present +- If `discard` has no custom reject callback, the default success payload is `Discarded: <label>. Reason: <reason>`. +- The TUI renderer is inline and merges call+result into one block. + +## Flow +1. Preview-producing code calls `queueResolveHandler(...)` with a label, source tool name, and `apply(reason)` callback, plus optional `reject(reason)`. +2. `queueResolveHandler(...)` asks the session for a forced `resolve` tool choice and pushes it into the tool-choice queue with `pushOnce(...)`. +3. The queued entry is marked `now: true`; if the model rejects that forced tool choice, `onRejected` returns `requeue`, so the reminder comes back. +4. `queueResolveHandler(...)` also injects a `resolve-reminder` steering message: `This is a preview. Call the resolve tool to apply or discard these changes.` +5. When `resolve.execute()` runs, it wraps the call in `untilAborted(...)` and fetches the current queue invoker with `session.peekQueueInvoker()`. +6. If no invoker exists, it throws `ToolError("No pending action to resolve. Nothing to apply or discard.")`. +7. Otherwise it invokes the queued callback with `{ action, reason }`. +8. For `apply`, it always executes the producer's `apply(reason)` callback. +9. For `discard`, it executes `reject(reason)` when provided; if that callback is absent or returns `undefined`, `resolve` fabricates the default discard message. +10. Before returning, it merges resolve metadata into `result.details` so renderer/UI code can show the action, label, and originating tool. + +## Modes / Variants +- `apply`: runs the queued `apply(reason)` callback and returns its content. +- `discard` with reject callback: runs `reject(reason)` and returns that callback's content. +- `discard` without reject callback: returns the built-in `Discarded: ...` text payload. + +## Side Effects +- Session state + - Consumes the current pending preview through the session tool-choice queue; there is no separate pending-action stack. + - Adds a `resolve-reminder` steering message when a preview is queued. +- User-visible prompts / interactive UI + - No direct prompt. The visible effect depends on the preview-producing tool and the resolve renderer. +- Background work / cancellation + - `untilAborted(...)` lets abort signals interrupt resolution before invoking the callback completes. + +## Limits & Caps +- Hidden tool: not discoverable in the normal tool index (`packages/coding-agent/src/tools/resolve.ts`, `packages/coding-agent/src/session/agent-session.ts`). +- Exactly one active queue invoker is consulted per call via `session.peekQueueInvoker()`. +- There is no independent queue depth cap in this tool; ordering follows the shared tool-choice queue (`docs/resolve-tool-runtime.md`). + +## Errors +- No pending preview: throws `ToolError("No pending action to resolve. Nothing to apply or discard.")`. +- Any exception from the queued `apply` / `reject` callback propagates through `resolve`. +- Aborts during `untilAborted(...)` surface as the underlying abort error from the utility. + +## Notes +- `reason` is informational; `resolve` passes it through but does not interpret it. +- `queueResolveHandler(...)` is the canonical built-in integration point; custom tools use `pushPendingAction(...)`, which the loader forwards into the same mechanism. +- The tool only works because another tool already staged a preview and forced a one-shot `resolve` choice. +- `sourceResultDetails` is added only when the apply/reject callback returned a non-null `details` field; custom pending-action `details` are not forwarded automatically by the loader. diff --git a/docs/tools/retain.md b/docs/tools/retain.md new file mode 100644 index 000000000..3898d6eb6 --- /dev/null +++ b/docs/tools/retain.md @@ -0,0 +1,98 @@ +# retain + +> Queue durable facts for asynchronous write into the active Hindsight bank. + +## Source +- Entry: `packages/coding-agent/src/tools/hindsight-retain.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/retain.md` +- Key collaborators: + - `packages/coding-agent/src/hindsight/state.ts` — per-session queue, flush, auto-retain. + - `packages/coding-agent/src/hindsight/backend.ts` — session bootstrap, prompt injection, subagent aliasing. + - `packages/coding-agent/src/hindsight/bank.ts` — bank id derivation, tag scoping, mission setup. + - `packages/coding-agent/src/hindsight/client.ts` — HTTP `retain` / `retainBatch` calls. + - `packages/coding-agent/src/hindsight/content.ts` — retention transcript shaping, memory-tag stripping. + - `packages/coding-agent/src/hindsight/mental-models.ts` — bank-scoped mental-model seeding and cache rendering. + - `packages/coding-agent/src/hindsight/seeds.json` — built-in mental-model seed definitions. + - `packages/coding-agent/src/hindsight/transcript.ts` — extracts user/assistant turns for auto-retain. + +## Inputs + +| Field | Type | Required | Description | +|---|---|---:|---| +| `items` | `Array<{ content: string; context?: string }>` | Yes | One or more memories to queue. `minItems: 1`. Each item must be self-contained; `context` is optional per-item provenance. | + +## Outputs +Returns a single-shot tool result: + +- `content[0].type = "text"` +- `content[0].text = "<count> memory queued."` or `"<count> memories queued."` +- `details = { count: number }` + +The write is not confirmed before the tool returns. The queue flushes later; flush failures emit a session warning notice and are not returned to the model. + +## Flow +1. `HindsightRetainTool.createIf(...)` only exposes the tool when `memory.backend == "hindsight"` in `packages/coding-agent/src/tools/hindsight-retain.ts`. +2. `execute(...)` fetches `session.getHindsightSessionState()` and throws if the Hindsight backend was not started. +3. Each input item is handed to `HindsightSessionState.enqueueRetain(...)` in `packages/coding-agent/src/hindsight/state.ts`. +4. `HindsightRetainQueue.enqueue(...)` appends the item and either: + - flushes immediately when the queue reaches `RETAIN_FLUSH_BATCH_SIZE`, or + - starts a debounce timer for `RETAIN_FLUSH_INTERVAL_MS`. +5. On flush, `HindsightRetainQueue.#doFlush(...)`: + - verifies the session still owns this state, + - calls `ensureBankMission(...)` once per bank/process before writing, + - maps queued items to `MemoryItemInput` with `context ?? config.retainContext`, `metadata.session_id`, and bank-scope tags, + - sends one async `retainBatch(...)` request. +6. The tool returns immediately after enqueueing; it does not await the HTTP write. + +## Modes / Variants +- Tool path: queued batch write only. +- Bank scoping comes from `computeBankScope(...)` in `packages/coding-agent/src/hindsight/bank.ts`: + - `global` — one shared bank, no project tags. + - `per-project` — bank id gets `-<cwd basename>` appended. + - `per-project-tagged` — shared bank plus `project:<cwd basename>` tags on retained memories. +- Session scope: + - tool-called retains are per-session queued work in `HindsightSessionState`, + - persisted memories are cross-session server-side bank data, + - subagents alias the parent `HindsightSessionState`, so their `retain` calls write into the same bank and queue. + +## Side Effects +- Filesystem + - None for retained memories. No local memory file is written. +- Network + - `POST /v1/default/banks/{bank_id}/memories` via `retainBatch(...)` in `packages/coding-agent/src/hindsight/client.ts`. + - Optional `PUT /v1/default/banks/{bank_id}` via `ensureBankMission(...)` before first write per bank/process. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Appends to the in-memory `HindsightRetainQueue` on the active `HindsightSessionState`. + - Includes `metadata.session_id` on each retained item. + - Shares parent state for subagents (`aliasOf` path in `packages/coding-agent/src/hindsight/backend.ts`). +- User-visible prompts / interactive UI + - On async flush failure, emits `session.emitNotice("warning", ...)`; the model is not told. +- Background work / cancellation + - Flush runs later on timer, queue-size threshold, `agent_end`, backend `enqueue(...)`, or backend `clear(...)`. + +## Limits & Caps +- Input schema requires `items.length >= 1` in `packages/coding-agent/src/tools/hindsight-retain.ts`. +- Queue flush threshold: `RETAIN_FLUSH_BATCH_SIZE = 16` in `packages/coding-agent/src/hindsight/state.ts`. +- Queue debounce: `RETAIN_FLUSH_INTERVAL_MS = 5_000` in `packages/coding-agent/src/hindsight/state.ts`. +- Queue writes use `retainBatch(..., { async: true })`; the client does not wait for server-side consolidation. +- Shared auto-retain settings on the same backend: + - `hindsight.retainEveryNTurns` default `3` + - `hindsight.retainOverlapTurns` default `2` + - `hindsight.retainContext` default `"omp"` + - `hindsight.retainMode` default `"full-session"` + from `packages/coding-agent/src/config/settings-schema.ts`. + +## Errors +- Throws `Hindsight backend is not initialised for this session.` when no state exists. +- Queue enqueue on disposed state throws `Hindsight retain queue is closed.` +- Flush-time API failures are caught in `HindsightRetainQueue.#doFlush(...)`, logged, and converted into a warning notice instead of a tool error. +- Mission creation failures are swallowed in `ensureBankMission(...)`; writes continue. + +## Notes +- Storage is server-side. `hindsightBackend.clear(...)` only clears local cache/state and warns that upstream deletion must happen in Hindsight UI or `deleteBank`; see `packages/coding-agent/src/hindsight/backend.ts`. +- Auto-retain uses the same bank but a different path than this tool: `retainSession(...)` extracts plain user/assistant transcript from `packages/coding-agent/src/hindsight/transcript.ts`, strips `<memories>` / `<mental_models>` blocks via `stripMemoryTags(...)`, and calls single-item `retain(...)`. +- `retain` itself does not seed or read mental models. Mental-model bootstrap lives in the shared backend: `HindsightSessionState.runMentalModelLoad(...)` optionally resolves seeds from `packages/coding-agent/src/hindsight/seeds.json`, creates missing models with `ensureMentalModels(...)`, then caches a rendered `<mental_models>` block for prompt injection. +- Built-in seeds are `user-preferences`, `project-conventions`, and `project-decisions`. `projectTagged: true` seeds inherit the active scope's retain tags; untagged seeds read the whole bank. +- Mental-model defaults from `packages/coding-agent/src/config/settings-schema.ts`: `hindsight.mentalModelsEnabled = true`, `hindsight.mentalModelAutoSeed = true`, `hindsight.mentalModelRefreshIntervalMs = 5 * 60 * 1000`, `hindsight.mentalModelMaxRenderChars = 16_000`. First-turn loading waits up to `MENTAL_MODEL_FIRST_TURN_DEADLINE_MS = 1500` in `packages/coding-agent/src/hindsight/mental-models.ts`. +- Seed lifecycle is create-only. Changing `packages/coding-agent/src/hindsight/seeds.json` does not mutate existing server-side models. +- `recall.md` and `reflect.md` rely on the same bank, scoping, and mental-model bootstrap; refer back here for the shared backend behavior. diff --git a/docs/tools/rewind.md b/docs/tools/rewind.md new file mode 100644 index 000000000..5ffa90945 --- /dev/null +++ b/docs/tools/rewind.md @@ -0,0 +1,96 @@ +# rewind + +> End an active checkpoint by pruning exploratory context and retaining a concise report. + +## Source +- Entry: `packages/coding-agent/src/tools/checkpoint.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/rewind.md` +- Key collaborators: + - `packages/coding-agent/src/session/agent-session.ts` — validates pending rewind state, applies the actual rewind, and injects the retained report. + - `packages/coding-agent/src/session/session-manager.ts` — branches the persisted session tree and appends persisted summary/report entries. + - `packages/coding-agent/src/session/messages.ts` — converts persisted `branch_summary` entries into LLM-visible branch-summary messages on rebuilt context. + - `packages/coding-agent/src/tools/index.ts` — registers the tool and shares the `checkpoint.enabled` gate. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `report` | `string` | Yes | Investigation findings. `execute()` trims it and rejects the empty result. | + +## Outputs +The tool returns a single text result plus structured details: + +- text body: + - `Rewind requested.` + - `Report captured for context replacement.` +- `details`: + - `report: string` — trimmed report text + - `rewound: true` + +The returned tool result is not the final rewind. `AgentSession` waits until `turn_end`, then applies the rewind side effects asynchronously. + +## Flow +1. `RewindTool.createIf()` in `packages/coding-agent/src/tools/checkpoint.ts` hides the tool from subagents. +2. `RewindTool.execute()` rejects subagent calls with `ToolError("Checkpoint not available in subagents.")`. +3. It rejects calls with no active checkpoint using `ToolError("No active checkpoint.")`. +4. It trims `params.report`; if empty, it throws `ToolError("Report cannot be empty.")`. +5. It returns a `toolResult()` with `details.report` and `details.rewound = true`. +6. On `tool_execution_end`, `AgentSession` extracts the report from `details.report` or the first text content block and stores it in `#pendingRewindReport`. +7. On `turn_end`, if `#pendingRewindReport` is set, `AgentSession.#applyRewind()` runs. +8. `#applyRewind()` computes `safeCount = clamp(checkpointMessageCount, 0, agent.state.messages.length)` and calls `agent.replaceMessages(agent.state.messages.slice(0, safeCount))`. +9. It then calls `sessionManager.branchWithSummary(checkpointEntryId, report, { startedAt })`. That moves the persisted session leaf back to the checkpoint entry and appends a new `branch_summary` entry whose `summary` is the rewind report. +10. If `checkpointEntryId` no longer resolves, it logs a warning and falls back to `branchWithSummary(null, report, { startedAt })`, branching from root instead. +11. `#applyRewind()` appends a hidden in-memory custom message `{ customType: "rewind-report", content: report, display: false }` and persists the same payload through `sessionManager.appendCustomMessageEntry("rewind-report", ...)` with `details = { startedAt, rewoundAt }`. +12. Finally it clears `#checkpointState` and `#pendingRewindReport`. + +## Modes / Variants +- Normal rewind: checkpoint entry exists; session history branches from that exact entry. +- Fallback rewind: checkpoint entry ID is missing from the current session tree; rewind branches from root and logs a warning. +- Immediate turn-end apply: rewind side effects happen only after the surrounding assistant turn finishes, not inside `RewindTool.execute()`. + +## Side Effects +- Session state (transcript, memory, jobs, checkpoints, registries) + - Replaces in-memory conversation history with the prefix ending at the checkpoint tool result. + - Adds a hidden custom message `rewind-report` carrying the retained report. + - Clears the active checkpoint state and pending rewind report. + - Repositions the persisted session leaf to the checkpoint branch point and appends new session entries. +- Filesystem + - Persists the new `branch_summary` and `custom_message` entries into the session `.jsonl` file through normal `SessionManager` append persistence. + - Session files are named `<ISO-timestamp-with-:-and-.-replaced>_<uuidv7>.jsonl` in the session directory; default directory selection is documented in `SessionManager.create()` as `~/.omp/agent/sessions/<encoded-cwd>/` when no override is passed. +- User-visible prompts / interactive UI + - The tool result itself is visible. + - The persisted `branch_summary` becomes an LLM-visible `branchSummary` message when context is rebuilt from `SessionManager.buildSessionContext()`; `messages.ts` renders it as a user-role text message using `prompts/compaction/branch-summary-context.md`. + - The persisted `rewind-report` custom message also participates in rebuilt LLM context because `custom_message` entries are converted through `createCustomMessage()`. +- Background work / cancellation + - Rewind application is deferred to `turn_end`. There is no separate job object or cancel handle. + +## Limits & Caps +- Availability is gated by `checkpoint.enabled`, default `false`, in `packages/coding-agent/src/config/settings-schema.ts`. +- Top-level sessions only. +- Requires exactly one active checkpoint; there is no path to name or choose among multiple checkpoints. +- Report text must be non-empty after `trim()`. +- Rewind restores only the message prefix recorded by `checkpointMessageCount`; there is no file restore, artifact restore, blob restore, or process restore path. +- Persisted report/summary content is still subject to the global session persistence cap `MAX_PERSIST_CHARS = 500_000` in `packages/coding-agent/src/session/session-manager.ts`. + +## Errors +- `ToolError("Checkpoint not available in subagents.")` — thrown for subagent sessions. +- `ToolError("No active checkpoint.")` — thrown when no checkpoint state is present. +- `ToolError("Report cannot be empty.")` — thrown when the trimmed report is empty. +- Missing checkpoint entry IDs during apply do not fail the tool call; `#applyRewind()` catches the error, logs `Rewind branch checkpoint missing, falling back to root`, and branches from root. +- If the agent turn is aborted while a checkpoint is active, `AgentSession` clears checkpoint state rather than applying a delayed rewind. + +## Notes +- Checkpoint selection is implicit. `rewind` always targets the single `#checkpointState` captured by the last successful `checkpoint`; there is no checkpoint list, label, or ID parameter. +- Restored state is transcript/session-tree state only: + - in-memory `agent.state.messages` prefix up to `checkpointMessageCount` + - persisted session leaf reset to `checkpointEntryId` or root fallback + - retained rewind report as `branch_summary` and hidden `rewind-report` custom message +- Not restored: + - filesystem contents + - git state + - artifacts under `packages/coding-agent/src/session/artifacts.ts` + - blob-store payloads under `packages/coding-agent/src/session/blob-store.ts` + - prompt history rows in `packages/coding-agent/src/session/history-storage.ts` + - auth or other agent storage in `packages/coding-agent/src/session/agent-storage.ts` +- There is no concurrent-edit reconciliation. If code or session-adjacent state changes during the checkpoint window, rewind does not merge or revert them; it only drops conversation context and rewires the session branch. +- Rewind is not destructive to persisted session history. `branchWithSummary()` appends a new `branch_summary` entry and moves the leaf; it does not delete the abandoned path from the `.jsonl` session log. The active context is cut over to the new branch, but the old entries remain in session storage. diff --git a/docs/tools/search.md b/docs/tools/search.md new file mode 100644 index 000000000..6fbe4badd --- /dev/null +++ b/docs/tools/search.md @@ -0,0 +1,143 @@ +# search + +> Search file contents with a regex across files, directories, globs, and internal URLs. + +## Source +- Entry: `packages/coding-agent/src/tools/search.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/search.md` +- Key collaborators: + - `packages/coding-agent/src/tools/match-line-format.ts` — model-facing anchor formatting. + - `packages/coding-agent/src/tools/path-utils.ts` — path normalization, glob splitting, internal URL resolution. + - `packages/coding-agent/src/tools/file-recorder.ts` — file ordering for grouped output. + - `packages/coding-agent/src/tools/grouped-file-output.ts` — grouped per-file text layout. + - `packages/coding-agent/src/session/streaming-output.ts` — line truncation and final byte truncation. + - `packages/coding-agent/src/config/settings-schema.ts` — default context lines. + - `packages/natives/native/index.d.ts` — native `grep()` types exposed to TS. + - `crates/pi-natives/src/grep.rs` — native regex/file search implementation. + - `docs/natives-text-search-pipeline.md` — native search pipeline overview. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `pattern` | `string` | Yes | Regex pattern. `search.ts` trims it and rejects empty input. The native matcher enables multiline only when the pattern text contains a literal newline or the two-character sequence `\\n`. The model prompt explicitly documents literal-brace escaping such as ``interface\\{\\}``, although the native layer also auto-escapes braces that cannot be valid repetition quantifiers. | +| `paths` | `string[]` | Yes | One or more file paths, directory paths, glob-like paths, or internal URLs. Empty strings are rejected after trimming/quote stripping. Internal URLs must resolve to a backing file and cannot contain glob characters. | +| `i` | `boolean` | No | Case-insensitive search. Defaults to `false`. Passed to native `ignoreCase`. | +| `gitignore` | `boolean` | No | Respect `.gitignore` during directory scans. Defaults to `true`. Passed to native `gitignore`. | +| `skip` | `number` | No | Global match offset. Defaults to `0`. `search.ts` floors finite numbers and rejects negative or non-finite values. | + +## Outputs +The tool returns a single text block in `content[0].text` plus structured `details`. + +- Match lines are formatted by `formatMatchLine()` as `*<anchor>|<line>` for matches and ` <anchor>|<line>` for context. + - Hashline mode: `*5th|content`, ` 9x}|content`. + - Plain mode: `*5|content`, ` 9|content`. +- Directory results are grouped by file, with `# <path>` headings and blank lines between groups. +- `details` may include: + - `scopePath` — formatted search scope. + - `matchCount`, `fileCount`, `files`, `fileMatches` — counts for the returned page, not necessarily total corpus counts. + - `matchLimitReached` — visible-page limit hit (`100`). + - `resultLimitReached` — native preselection limit hit (`500`). + - `linesTruncated` — one or more matched lines were shortened to `1024` chars plus `…`. + - `truncated` and `meta.truncation` — final text output was head-truncated by `truncateHead()`. + - `displayContent` — TUI-only rendering text with `│` gutters instead of model anchors. + - `missingPaths` — multi-path entries skipped because their base path did not exist. +- No-match result text is `No matches found`, optionally followed by `Skipped missing paths: ...`. + +## Flow +1. `SearchTool.execute()` validates and normalizes input in `packages/coding-agent/src/tools/search.ts`: + - trims `pattern`, rejects empty patterns; + - normalizes `skip` to a non-negative integer; + - reads `search.contextBefore` and `search.contextAfter` from session settings (`1` and `3` by default); + - enables multiline only when `pattern` contains `\n` or an actual newline. +2. Each `paths` entry is normalized with `normalizePathLikeInput()`. +3. Internal URLs are resolved through `session.internalRouter`: + - glob metacharacters (`*`, `?`, `[`, `{`) are rejected for internal URLs; + - URLs without `resource.sourcePath` fail; + - immutable sources are tracked so output can suppress editable hashline anchors per file. +4. For multi-path calls, `partitionExistingPaths()` skips only ENOENT entries. If every entry is missing, the tool errors. +5. Path resolution branches: + - one entry: `parseSearchPath()` splits `basePath` and optional glob; + - multiple entries: `resolveExplicitSearchPaths()` computes a common base directory, brace-union glob, exact-file list, or degenerate-root target list. +6. `search.ts` stats the resolved base path to decide file vs directory behavior. +7. It calls native `grep()` from `@oh-my-pi/pi-natives` with: + - `pattern`, `ignoreCase`, `multiline`, `gitignore`; + - `hidden: true`; + - `cache: false`; + - `contextBefore` / `contextAfter` from settings; + - `maxColumns: 1024`; + - `mode: content`. +8. Native execution happens in `crates/pi-natives/src/grep.rs`: + - `build_matcher()` sanitizes non-quantifier braces before regex compile; + - if compile fails with unopened/unclosed-group errors, it retries after escaping previously unescaped parentheses; + - directory scans use the grep pipeline described in `docs/natives-text-search-pipeline.md`. +9. Search dispatch differs by resolved path set: + - exact explicit files or degenerate-root multi-targets: JS loops over targets and merges `grep()` results itself; + - single file/directory base: one `grep()` call handles offset/limit natively. +10. JS output shaping then: + - round-robins directory matches down to `100` visible matches so one file does not monopolize the page; + - keeps the first `100` file matches for single-file searches; + - formats lines through `formatMatchLine()` for the model and `formatCodeFrameLine()` for TUI; + - records non-truncated matched/context lines into the session file-read cache with `recordSparse()`. +11. Final text is passed through `truncateHead(rawOutput, { maxLines: Number.MAX_SAFE_INTEGER })`, so the effective cap is the default byte cap from `streaming-output.ts`, not the default line cap. +12. `toolResult()` attaches text plus limit/truncation metadata. + +## Modes / Variants +1. **Single file path** + - `grep()` searches one file. + - Output is a flat list of match/context lines. + - Visible limit is the first `100` matches after native offset handling. +2. **Single directory path or single glob-like path** + - `parseSearchPath()` may split the input into `path` + `glob`. + - One native `grep()` scans the directory tree with `gitignore` and `hidden:true`. + - Native `offset` handles `skip` globally across files. + - JS round-robins the returned matches to `100` visible rows. +3. **Multiple explicit paths/globs** + - `resolveExplicitSearchPaths()` collapses them into a common base and either a brace-union glob, an explicit file list, or per-target searches when the only common base is the filesystem root. + - Missing entries are skipped non-fatally unless all are missing. +4. **Internal URL paths** + - Supported only when the internal resource resolves to a real backing file. + - No internal-URL globbing. + - Immutable sources switch to the immutable display mode when formatting anchors. + +## Side Effects +- Filesystem + - Stats resolved search roots and input paths. + - Reads matched files through native `grep()`. + - Records sparse matched/context lines into the session file-read cache via `getFileReadCache(...).recordSparse(...)`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Reads session settings for context defaults. + - Uses `session.internalRouter` to resolve internal URLs. + - Populates tool `details.meta` with truncation/limit metadata. +- Background work / cancellation + - Wrapped in `untilAborted(signal, ...)` at the JS level. + - `search.ts` does not pass `signal` or `timeoutMs` into native `grep()`, so native grep cancellation/timeouts are not used by this tool. + +## Limits & Caps +- Visible page limit: `100` matches (`DEFAULT_MATCH_LIMIT` in `packages/coding-agent/src/tools/search.ts`). +- Native preselection limit: `500` matches (`internalLimit = Math.min(DEFAULT_MATCH_LIMIT * 5, 2000)` in `packages/coding-agent/src/tools/search.ts`). +- Line truncation: `1024` characters per emitted line (`DEFAULT_MAX_COLUMN` in `packages/coding-agent/src/session/streaming-output.ts`). Native grep marks truncated lines; JS reports `linesTruncated`. +- Final text truncation: `truncateHead()` default byte cap `50 * 1024` bytes (`DEFAULT_MAX_BYTES` in `packages/coding-agent/src/session/streaming-output.ts`). `search.ts` overrides `maxLines` to `Number.MAX_SAFE_INTEGER`, so normal search output is byte-capped, not line-capped. +- Context defaults: `search.contextBefore = 1`, `search.contextAfter = 3` in `packages/coding-agent/src/config/settings-schema.ts`. +- Pagination: `skip` is a global match offset. In single-base searches it is pushed into native `offset`; in exact-file/multi-target aggregation it is applied in JS with `matches.slice(skip)`. +- Native directory-scan cache: available in `grep.rs`, but this tool always sets `cache: false`. + +## Errors +- `Pattern must not be empty` when trimmed `pattern` is empty. +- `Skip must be a non-negative number` for negative or non-finite `skip`. +- `` `paths` must contain non-empty paths or globs `` when any normalized path is empty. +- `Glob patterns are not supported for internal URLs: ...` for internal URL + glob metacharacters. +- `Cannot search internal URL without a backing file: ...` when the router resolves a virtual resource without `sourcePath`. +- `Path not found: ...` when the resolved base path is missing, or when every multi-path entry is missing. +- Regex compile failures bubble from native `grep()` as tool errors. `search.ts` has a special catch for messages beginning with `regex parse error`, then otherwise rethrows. +- Multi-file native scans skip per-file open/search failures inside `grep.rs`; the scan continues with surviving files. + +## Notes +- The model-facing prompt documents standard regex syntax plus two search-specific rules: escape literal braces, and use `\n` or a literal newline for cross-line matching. +- Native `build_matcher()` already auto-escapes braces that cannot be valid quantifiers, so patterns like `${platform}` become searchable instead of failing. Valid quantifiers like `a{2,4}` remain unchanged. +- Native compile retry also escapes unescaped literal parentheses only after an unopened/unclosed-group parse error. It is a fallback, not a general parser mode. +- Internal URLs are resolved before path existence checks. After resolution, the native layer sees ordinary filesystem paths. +- `hidden:true` is hard-coded in `search.ts`; there is no model-facing flag to exclude dotfiles. +- `gitignore:false` only affects native directory traversal. It does not disable the tool's own path normalization or explicit-file handling. +- When `paths` resolves to multiple exact files, `search.ts` does not apply the native `500` match cap and reports `totalMatches` internally as the post-skip length for that branch. +- The anchor suffix in hashline mode comes from `computeLineHash()` in `packages/coding-agent/src/hashline/hash.ts`; `search` itself only formats it. diff --git a/docs/tools/search_tool_bm25.md b/docs/tools/search_tool_bm25.md new file mode 100644 index 000000000..886f446bb --- /dev/null +++ b/docs/tools/search_tool_bm25.md @@ -0,0 +1,117 @@ +# search_tool_bm25 + +> Search the hidden tool-discovery index and activate the top matches for the current session. + +## Source +- Entry: `packages/coding-agent/src/tools/search-tool-bm25.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/search-tool-bm25.md` +- Key collaborators: + - `packages/coding-agent/src/tool-discovery/tool-index.ts` — discoverable-tool metadata and BM25 index/search. + - `packages/coding-agent/src/session/agent-session.ts` — session discovery mode, corpus assembly, activation, cache invalidation. + - `packages/coding-agent/src/sdk.ts` — initial hiding of discoverable built-ins and prompt-time discoverable summary. + - `packages/coding-agent/src/tools/index.ts` — tool-session discovery hooks, essential/discoverable load modes, registry wiring. + - `packages/coding-agent/src/config/settings-schema.ts` — `tools.discoveryMode` and legacy `mcp.discoveryMode` settings. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `query` | `string` | Yes | Natural-language or keyword query. Trimmed before search; empty-after-trim is rejected. | +| `limit` | `integer` | No | Max matches to return and activate. Minimum `1`. Defaults to `8` (`DEFAULT_LIMIT`). | + +## Outputs +- Single-shot `AgentToolResult`. +- Model-visible `content` is one text part containing JSON with: + +```json +{"query":"...","activated_tools":["..."],"match_count":2,"total_tools":17} +``` + +- Runtime-only `details` carries the ranked matches used by the TUI renderer: + - `query`, `limit`, `total_tools` + - `activated_tools`: tool names activated by this call + - `active_selected_tools`: cumulative discovered-tool selections still active + - `tools`: array of match objects with + - `name` + - `label` + - `description` (`tool.summary`; this is the only snippet-like field) + - optional `server_name` + - optional `mcp_tool_name` + - `schema_keys` + - `score` rounded to 6 decimals +- The renderer shows a status line plus up to 5 collapsed tree items by default (`COLLAPSED_MATCH_LIMIT`), each with label, optional server name, score to 3 decimals, and truncated description. The ranked match list is not serialized into `content`. + +## Flow +1. `SearchToolBm25Tool.createIf()` in `packages/coding-agent/src/tools/search-tool-bm25.ts` exposes the tool only when `tools.discoveryMode !== "off"` or legacy `mcp.discoveryMode === true`, and only if the session implements the discovery hooks. +2. `description` is rendered from `packages/coding-agent/src/prompts/tools/search-tool-bm25.md` via `renderSearchToolBm25Description()`, using the current discoverable-tool list plus per-server summary/count. +3. `execute()` re-checks capability and settings: + - missing discovery hooks -> `ToolError("Tool discovery is unavailable in this session.")` + - discovery disabled -> `ToolError("Tool discovery is disabled. Enable tools.discoveryMode or mcp.discoveryMode to use search_tool_bm25.")` +4. `query` is trimmed and validated; `limit` is defaulted/validated. +5. `getDiscoverableToolSearchIndexForExecution()` fetches the cached generic search index from the session when available, otherwise falls back to the legacy MCP cache, otherwise rebuilds an index from the current discoverable-tool list. +6. `getSelectedToolNames()` reads the current discovered selections so already-selected tools can be excluded from fresh results. +7. `searchDiscoverableTools()` in `packages/coding-agent/src/tool-discovery/tool-index.ts` tokenizes the query, scores every document with BM25, sorts by descending score then `tool.name`, and returns up to `searchIndex.documents.length` results; `execute()` then filters already-selected names and slices to `limit`. +8. If any matches remain, `activateTools()` activates all matched tool names through `session.activateDiscoveredTools()` or legacy `activateDiscoveredMCPTools()`. +9. `details` is assembled from the activated names, current selected names, corpus size, and formatted matches; `content` is reduced to the compact JSON summary from `buildSearchToolBm25Content()`. +10. `searchToolBm25Renderer` renders either: + - the structured `details` view, or + - a fallback text-only warning block if `details` is absent. + +## Modes / Variants +- Discovery-mode gating: + - `tools.discoveryMode = "all"`: searches hidden discoverable built-ins plus hidden MCP tools. + - `tools.discoveryMode = "mcp-only"`: searches hidden MCP tools only. + - legacy `mcp.discoveryMode = true` with `tools.discoveryMode = "off"`: same as MCP-only. +- Search-index source: + - generic cached discoverable index from the session + - legacy cached MCP index, cast to the generic shape + - rebuilt ad hoc from the current discoverable-tool list if neither cache path works +- Activation backend: + - generic `activateDiscoveredTools()` + - legacy `activateDiscoveredMCPTools()` fallback + +## Side Effects +- Session state + - Adds matched tools to the active session tool set through `activateDiscoveredTools()` / `activateDiscoveredMCPTools()`. + - Updates discovered-tool selection state so repeated searches accumulate selections instead of replacing them. + - Invalidates the cached discoverable search index when newly activated built-ins change the hidden corpus (`packages/coding-agent/src/session/agent-session.ts`). + - Tool availability changes before the next model call in the same turn; the prompt text says this explicitly. +- User-visible prompts / interactive UI + - The tool description includes discoverable server summaries and total discoverable-tool count. + - The TUI renderer shows ranked matches, but the model-visible text summary does not. + +## Limits & Caps +- Default result cap: `8` (`DEFAULT_LIMIT` in `packages/coding-agent/src/tools/search-tool-bm25.ts`). +- `limit` must be a positive integer; no tool-level upper bound beyond corpus size. +- Renderer collapsed list cap: `5` (`COLLAPSED_MATCH_LIMIT`). +- Renderer truncation widths: + - label: `72` chars (`MATCH_LABEL_LEN`) + - description: `96` chars (`MATCH_DESCRIPTION_LEN`) +- BM25 parameters in `packages/coding-agent/src/tool-discovery/tool-index.ts`: + - `BM25_K1 = 1.2` + - `BM25_B = 0.75` +- Weighted corpus fields (`FIELD_WEIGHTS`): + - `name`: `6` + - `label`: `4` + - `mcpToolName`: `4` + - `serverName`: `2` + - `summary`: `2` + - each `schemaKey`: `1` +- Summary fallback length for discoverable metadata: first `200` chars of `description` when no explicit summary exists (`getDiscoverableTool()` in `packages/coding-agent/src/tool-discovery/tool-index.ts`). + +## Errors +- `execute()` throws `ToolError` for unavailable discovery hooks, disabled discovery mode, empty trimmed query, and non-positive/non-integer `limit`. +- `searchDiscoverableTools()` throws `Error("Query must contain at least one letter or number.")` if tokenization produces no alphanumeric tokens; `execute()` catches `Error` and rethrows `ToolError(error.message)`. +- Empty corpus is not an error; search returns `[]`, activation is skipped, and the renderer message becomes either `No discoverable tools are currently loaded.` or `No matching tools found.` +- `getDiscoverableToolsForDescription()` and `getDiscoverableToolSearchIndexForExecution()` swallow discovery-hook/cache errors and fall back to an empty corpus or rebuilt index. + +## Notes +- The tool wire name stays `search_tool_bm25` for persisted-session back-compat, even though the source file is `search-tool-bm25.ts`. +- Corpus composition is session-dependent and excludes already-active tools: + - MCP entries come from `#discoverableMCPTools`, filtered to names not currently active, mapped with `summary = description`. + - Built-in entries appear only in `"all"` mode and only for registry tools whose `loadMode === "discoverable"` and are not currently active. + - Hidden/internal built-ins are intentionally excluded from the built-in corpus: `resolve`, `yield`, `exit_plan_mode`, `report_finding`, `report_tool_issue` are called out in the `#collectDiscoverableBuiltinTools()` comment. +- `DiscoverableToolSource` includes `"extension"` and `"custom"`, but `AgentSession.getDiscoverableTools()` currently assembles only built-in and MCP sources. +- On startup, `packages/coding-agent/src/sdk.ts` hides non-essential discoverable built-ins in `tools.discoveryMode = "all"`; defaults are `read`, `bash`, and `edit` unless `tools.essentialOverride` changes them. +- Query tokenization is simple and deterministic: camelCase is split, non-alphanumerics become spaces, tokens are lowercased, and only non-empty alphanumeric tokens survive. +- Scores are rounded differently by surface: `details.tools[].score` keeps 6 decimals; the TUI line renders 3. diff --git a/docs/tools/ssh.md b/docs/tools/ssh.md new file mode 100644 index 000000000..94e912d35 --- /dev/null +++ b/docs/tools/ssh.md @@ -0,0 +1,127 @@ +# ssh + +> Execute one remote command on a discovered SSH host. + +## Source +- Entry: `packages/coding-agent/src/tools/ssh.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/ssh.md` +- Key collaborators: + - `packages/coding-agent/src/ssh/ssh-executor.ts` — runs `ssh`, captures output + - `packages/coding-agent/src/ssh/connection-manager.ts` — master-connection reuse, host probing + - `packages/coding-agent/src/ssh/sshfs-mount.ts` — optional `sshfs` mount side effect + - `packages/coding-agent/src/discovery/ssh.ts` — discovers host configs + - `packages/coding-agent/src/capability/ssh.ts` — canonical host shape + - `packages/coding-agent/src/session/streaming-output.ts` — tail streaming, truncation, artifacts + - `packages/coding-agent/src/tools/tool-timeouts.ts` — timeout clamp rules + - `packages/utils/src/dirs.ts` — user/project ssh config paths + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `host` | `string` | Yes | Host name key from discovered SSH config entries, not an arbitrary hostname/IP. | +| `command` | `string` | Yes | Remote command string passed to `ssh` as the remote command. | +| `cwd` | `string` | No | Remote working directory. The tool prepends a shell-specific `cd`/`Set-Location` wrapper. | +| `timeout` | `number` | No | Timeout in seconds. Default `60`; clamped to `1..3600`. | + +## Outputs +The tool returns a standard text tool result built in `packages/coding-agent/src/tools/ssh.ts`: + +- `content`: one text block containing combined remote stdout+stderr, or `"(no output)"` when empty. +- `details.meta.truncation`: present when output exceeded the in-memory tail window; derived from the executor summary. + +Streaming behavior: + +- While the command runs, `onUpdate` receives tail-only text snapshots built from `TailBuffer` in `packages/coding-agent/src/session/streaming-output.ts`. +- Final output is single-shot after process exit. + +Side-channel artifacts: + +- When session artifact allocation is available and output exceeds the spill threshold, full output is written to a session artifact file and the returned summary carries its `artifactId` internally. +- The ssh tool itself does not print the `artifact://...` URI into the result text. + +Failure behavior: + +- Unknown host, missing host config, timeout, cancellation, SSH startup failure, key validation failure, or non-zero remote exit all surface as thrown `ToolError`s. +- Non-zero remote exit includes captured output plus `Command exited with code N`. + +## Flow +1. `loadSshTool()` in `packages/coding-agent/src/tools/ssh.ts` calls `loadCapability(sshCapability.id, { cwd: session.cwd })` to discover hosts. +2. `packages/coding-agent/src/discovery/ssh.ts` loads host entries from, in this order: project managed ssh config, user managed ssh config, `ssh.json` in the repo root, `.ssh.json` in the repo root. +3. `getSSHConfigPath("project")` and `getSSHConfigPath("user")` in `packages/utils/src/dirs.ts` resolve those managed files to `.omp/ssh.json` in the project and `~/.omp/agent/ssh.json` in the user config dir. This tool does not read `~/.ssh/config`. +4. Capability loading deduplicates by host name with first item winning; provider order is priority-sorted and the SSH JSON provider registers at priority `5`. +5. `loadHosts()` in `packages/coding-agent/src/tools/ssh.ts` builds `hostsByName` and drops later duplicates again with `if (!hostsByName.has(host.name))`. +6. Tool description text is built from `packages/coding-agent/src/prompts/tools/ssh.md` plus an `Available hosts:` list. Each host entry calls `getHostInfoForHost()` to show detected shell/OS when cached; otherwise it renders `detecting...`. +7. On execute, `SshTool.execute()` rejects any `host` not in the discovered host-name set. +8. `ensureHostInfo()` in `packages/coding-agent/src/ssh/connection-manager.ts` ensures an SSH master connection exists, loads cached host info from disk if present, and probes remote OS/shell when cache is missing or stale. +9. `buildRemoteCommand()` in `packages/coding-agent/src/tools/ssh.ts` prepends a cwd change when `cwd` is provided: + - Unix-like or Windows compat shells: `cd -- '<cwd>' && <command>` + - Windows PowerShell: `Set-Location -Path '<cwd>'; <command>` + - Windows cmd: `cd /d "<cwd>" && <command>` +10. `clampTimeout("ssh", rawTimeout)` applies the `1..3600` second clamp from `packages/coding-agent/src/tools/tool-timeouts.ts`. +11. `executeSSH()` in `packages/coding-agent/src/ssh/ssh-executor.ts` calls `ensureConnection(host)` again, opportunistically mounts the remote host root with `sshfs` if available, optionally wraps the command in `bash -c` or `sh -c` for Windows compat mode, then spawns `ssh` with `ptree.spawn`. +12. Output from both stdout and stderr is piped into one `OutputSink`; chunks are sanitized and forwarded to streaming updates through `streamTailUpdates()`. +13. On normal exit, the sink returns combined output plus truncation counters. On timeout or abort, `executeSSH()` returns `cancelled: true` and prefixes the output with a notice line such as `[SSH: ...]` or `[Command aborted: ...]`. +14. `SshTool.execute()` converts `cancelled: true` into `ToolError`, converts non-zero exit codes into `ToolError`, otherwise returns the text result with truncation metadata. + +## Modes / Variants +- **Tool unavailable**: `loadSshTool()` returns `null` when discovery finds no hosts, so the tool is not registered for that session. +- **Unix-like target**: remote command is passed through directly, with optional `cd -- ... &&` prefix. +- **Windows native shell**: cwd wrapper uses PowerShell `Set-Location` or cmd `cd /d`; command otherwise runs in the remote default Windows shell. +- **Windows compat shell**: if host probing finds `bash` or `sh` on Windows, `executeSSH()` wraps the remote command as `bash -c '...'` or `sh -c '...'`. Host config can force compat on/off with `compat`. +- **Cached vs probed host info**: shell/OS detection comes from in-memory cache, persisted JSON under the remote-host dir, or a fresh probe over SSH. +- **Truncated vs untruncated output**: small output stays in memory; large output keeps only the last 50 KiB in memory and may spill full output to an artifact file. + +## Side Effects +- Filesystem + - Reads managed SSH config JSON plus legacy `ssh.json` / `.ssh.json`. + - Validates private-key path existence and permissions before connecting. + - Persists probed host info as JSON under the remote-host cache dir via `persistHostInfo()`. + - May create the SSH control socket dir and, when `sshfs` exists, remote mount dirs. + - May write full command output to a session artifact file. +- Network + - Opens SSH connections to the selected host. + - May issue extra probe commands to detect OS/shell and compat shells. +- Subprocesses / native bindings + - Requires `ssh` on `PATH`; spawns it for connection checks, master startup, probing, and command execution. + - May call `sshfs`, `mountpoint`, `fusermount`/`fusermount3`, or `umount`. + - Sanitizes streamed text with `@oh-my-pi/pi-natives` text sanitization. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Uses session artifact allocation when available. + - Registers postmortem cleanup hooks for SSH master connections and sshfs mounts. + - Tool concurrency is `exclusive`, so the agent scheduler should not run multiple ssh tool calls concurrently. +- Background work / cancellation + - Process spawn receives the tool `AbortSignal`. + - Cancellation/timeout ends the running ssh process and returns a cancelled result that the tool turns into an error. + +## Limits & Caps +- Timeout defaults/clamps: `default=60`, `min=1`, `max=3600` in `packages/coding-agent/src/tools/tool-timeouts.ts`. +- Output tail window: `DEFAULT_MAX_BYTES = 50 * 1024` in `packages/coding-agent/src/session/streaming-output.ts`. +- Output sink spill threshold defaults to the same `50 KiB`; once exceeded, only the tail remains in memory. +- SSH master reuse persistence: `ControlPersist=3600` in `packages/coding-agent/src/ssh/connection-manager.ts` and `packages/coding-agent/src/ssh/sshfs-mount.ts`. +- SSH host info schema version: `HOST_INFO_VERSION = 2` in `packages/coding-agent/src/ssh/connection-manager.ts`; stale cache entries are reprobed. +- Streaming tail buffer compacts after more than `10` pending chunks (`MAX_PENDING`) before trimming. + +## Errors +- `Unknown SSH host: ... Available hosts: ...` when the model passes a host name not present in discovery. +- `SSH host not loaded: ...` if the discovered-name set and `hostsByName` map diverge. +- `ssh binary not found on PATH` when `ssh` is unavailable. +- `SSH key not found: ...`, `SSH key is not a file: ...`, or `SSH key permissions must be 600 or stricter: ...` from key validation. +- `Failed to start SSH master for <target>: <stderr>` when control-master startup fails. +- Non-zero remote command exit becomes `ToolError` with captured output and `Command exited with code N`. +- Timeout becomes a cancelled result with output notice `[SSH: <timeout message>]`, then `ToolError`. +- Abort becomes a cancelled result with output notice `[Command aborted: <message>]`, then `ToolError`. +- `sshfs` mount failures are logged and ignored in `executeSSH()`; they do not fail the tool call. +- Discovery parse problems do not fail tool loading; they become capability warnings. If all sources are empty/invalid, the tool simply does not load. + +## Notes +- Host discovery is JSON-based only. The tool does not parse OpenSSH config files. +- Discovery expands environment variables recursively in the parsed JSON and expands `~` in `key`/`keyPath`. +- Host names are capability keys; the model must pass the config key, not the raw hostname. +- Commands run without a PTY. `executeSSH()` uses `ptree.spawn(..., { stdin: "pipe", stderr: "full" })` and does not request an interactive terminal. +- The tool exposes `cwd` but no `env`, `pty`, upload, download, or explicit file-transfer fields. +- Lower layers support an `artifactId` for full output and a `remotePath` mount target, but `SshTool.execute()` does not expose those knobs. +- Both stdout and stderr are merged into one output stream; ordering is whatever arrives through the two streams. +- `StrictHostKeyChecking=accept-new` and `BatchMode=yes` are always set for connection checks, master startup, and command runs. +- Connection reuse is keyed by discovered host name, not by raw target tuple alone. +- `closeAllConnections()` and sshfs unmount cleanup run through postmortem hooks, not per-call teardown. diff --git a/docs/tools/task.md b/docs/tools/task.md new file mode 100644 index 000000000..58955e444 --- /dev/null +++ b/docs/tools/task.md @@ -0,0 +1,224 @@ +# task + +> Launch subagents for parallel, optionally isolated work. + +## Source +- Entry: `packages/coding-agent/src/task/index.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/task.md` +- Key collaborators: + - `packages/coding-agent/src/task/types.ts` — dynamic schema, progress/result types, output caps. + - `packages/coding-agent/src/task/discovery.ts` — discover project/user/plugin/bundled agents. + - `packages/coding-agent/src/task/agents.ts` — bundled agent definitions and frontmatter parsing. + - `packages/coding-agent/src/task/executor.ts` — create child sessions, run subagents, collect output. + - `packages/coding-agent/src/task/parallel.ts` — concurrency-limited scheduling and async semaphore. + - `packages/coding-agent/src/task/isolation-backend.ts` — isolation backend resolution and platform fallback. + - `packages/coding-agent/src/task/worktree.ts` — worktree / FUSE / ProjFS setup, patch capture, branch merge. + - `packages/coding-agent/src/task/output-manager.ts` — session-scoped `agent://` id allocation. + - `packages/coding-agent/src/task/simple-mode.ts` — `default` / `schema-free` / `independent` field gating. + - `packages/coding-agent/src/internal-urls/agent-protocol.ts` — resolve `agent://<id>` to saved subagent output. + - `packages/coding-agent/src/tools/index.ts` — tool registration and recursion-depth gating. + - `packages/coding-agent/src/sdk.ts` — child-session router/tool wiring and per-subagent `AgentOutputManager`. + - `docs/task-agent-discovery.md` — deeper discovery and precedence notes. + - `docs/handoff-generation-pipeline.md` — session artifact/handoff persistence patterns used by the wider session layer. + +## Inputs + +### Default mode (`task.simple = "default"`) + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `agent` | `string` | Yes | Exact agent name for every task item. Resolved at execution time through `discoverAgents(...)`. | +| `tasks` | `Array<{ id: string; description: string; assignment: string }>` | Yes | Batch of small, self-contained task items. `id` max length 48 in schema; duplicate ids are rejected case-insensitively at runtime. | +| `context` | `string` | No | Shared background prepended to every subagent system prompt. Trimmed before use. | +| `schema` | `string` | No | JSON-encoded JTD schema. Overrides agent/session output schema when this mode allows task-level schemas. | +| `isolated` | `boolean` | No | Only present when the tool is created with isolation enabled. Requests isolated execution for the whole batch. | + +`tasks[].description` is UI-only. `tasks[].assignment` is the actual per-task instruction. + +### Schema-free mode (`task.simple = "schema-free"`) + +Same as default, except `schema` is rejected by `validateTaskModeParams(...)` in `packages/coding-agent/src/task/index.ts`. + +### Independent mode (`task.simple = "independent"`) + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `agent` | `string` | Yes | Exact agent name. | +| `tasks` | `Array<{ id: string; description: string; assignment: string }>` | Yes | Same item shape, but each `assignment` must carry all required background because shared `context` is disabled. | +| `isolated` | `boolean` | No | Same conditional field as above. | + +In this mode both `context` and `schema` are rejected. + +## Outputs +The tool returns one text block plus `details: TaskToolDetails`. + +`details` fields: +- `projectAgentsDir: string | null` — nearest discovered project `agents/` dir. +- `results: SingleResult[]` — one entry per task in input order for synchronous execution; empty for async-launch responses. +- `totalDurationMs: number` +- `usage?: Usage` — sum of per-subagent assistant-message usage. +- `outputPaths?: string[]` — written `.md` artifact paths for completed subagent outputs. +- `progress?: AgentProgress[]` — live or final per-task progress snapshots. +- `async?: { state: "running" | "completed" | "failed"; jobId: string; type: "task" }` — present for background execution updates/results. + +`SingleResult` includes: +- identity: `index`, `id`, `agent`, `agentSource`, `description`, optional `assignment` +- status: `exitCode`, optional `error`, optional `aborted`, optional `abortReason` +- output: `output`, `stderr`, `truncated`, `durationMs`, `tokens` +- artifact metadata: `outputPath?`, `patchPath?`, `branchName?`, `nestedPatches?`, `outputMeta?` +- extracted tool data: `extractedToolData?` from registered subprocess tool handlers such as `yield` and `report_finding` + +Artifacts and side channels: +- Every subagent with an artifacts dir writes `<id>.md`; `agent://<id>` resolves to that file. +- If the output file is JSON, `agent://<id>/<path>` and `agent://<id>?q=<query>` perform JSON extraction in `packages/coding-agent/src/internal-urls/agent-protocol.ts`. +- When the parent session persists artifacts, each subagent also gets `<id>.jsonl` session history. +- Isolated patch mode writes `<id>.patch` per successful task before merge. +- Async mode returns immediately after job registration, then emits `onUpdate(...)` progress snapshots and later hands completion to the session async-job pipeline. + +## Flow +1. `TaskTool.create(...)` in `packages/coding-agent/src/task/index.ts` calls `discoverAgents(session.cwd)` once to build the dynamic prompt description from current agents and `task.simple` capabilities. +2. `execute(...)` validates mode-gated fields with `validateTaskModeParams(...)`. +3. It decides async vs sync: + - sync when `async.enabled` is false + - sync when the selected cached agent has `blocking === true` + - sync when `tasks.length === 0` + - otherwise async job scheduling +4. Async path: + - allocate unique output ids with `AgentOutputManager.allocateBatch(...)` + - create one async job per task through `session.asyncJobManager.register(...)` + - limit concurrent job bodies with `Semaphore(task.maxConcurrency)` from `packages/coding-agent/src/task/parallel.ts` + - each job body calls `#executeSync(...)` with a one-task batch and the preallocated id + - `onUpdate(...)` emits aggregate `progress` snapshots and `details.async` +5. Sync path (`#executeSync(...)`) rediscovers agents from disk via `discoverAgents(...)`, so runtime resolution can differ from the earlier prompt description. +6. It resolves the requested agent with `getAgent(...)`, rejects unknown or disabled agents, and enforces parent spawn policy plus `PI_BLOCKED_AGENT` self-recursion prevention. +7. It derives the effective output schema in priority order: task call `schema` (if allowed) → agent frontmatter `output` → inherited parent session schema. +8. It validates task ids: missing ids and case-insensitive duplicates are immediate errors. +9. If `isolated` was requested, it requires a git repo (`getRepoRoot(...)` / `captureBaseline(...)`) and resolves the actual backend through `resolveIsolationBackendForTaskExecution(...)`. +10. It chooses an artifacts dir from the parent session when available, otherwise a temp dir, and writes `context.md` there when `session.getCompactContext?.()` returns content. +11. It allocates unique ids again if the caller did not preallocate them, then builds `tasksWithUniqueIds`. +12. For each task, it seeds an `AgentProgress` entry and runs `runTask(...)` through `mapWithConcurrencyLimit(...)` using `task.maxConcurrency`. +13. Non-isolated `runTask(...)` calls `runSubprocess(...)` directly with parent cwd. +14. Isolated `runTask(...)`: + - creates an isolation workspace (`ensureWorktree(...)`, `ensureFuseOverlay(...)`, or `ensureProjfsOverlay(...)`) + - applies the captured baseline for worktrees + - runs `runSubprocess(...)` inside that workspace + - on success, either commits to a per-task branch (`mergeMode === "branch"`) or captures a patch with `captureDeltaPatch(...)` + - always cleans up the isolation workspace/backend +15. `runSubprocess(...)` in `packages/coding-agent/src/task/executor.ts` creates a child agent session with: + - isolated settings snapshot via `Settings.isolated(...)`, forcing `async.enabled = false` and `bash.autoBackground.enabled = false` + - child `agentId` / `parentTaskPrefix` equal to the allocated task id + - child internal URL router and `AgentOutputManager` from `packages/coding-agent/src/sdk.ts` + - the shared `context`, optional `context.md` reference, optional isolation worktree path, output schema, and IRC peer roster in the system prompt template +16. Child tool availability is derived from the agent definition plus runtime guards: + - explicit `agent.tools` if provided + - auto-add `task` when the agent has `spawns` and recursion depth allows it + - remove `task` at or past `task.maxRecursionDepth` + - expand `exec` to `eval` and `bash` + - strip parent-owned `todo_write` after session creation +17. `runSubprocess(...)` subscribes to child agent events, coalesces progress updates every 150 ms, forwards lifecycle/progress events on the parent event bus, and extracts tool data through `subprocessToolRegistry`. +18. The child must finish through the hidden `yield` tool. If it does not, `runSubprocess(...)` sends up to 3 reminder prompts; the last reminder forces `toolChoice = yield` when supported. +19. Finalization uses `finalizeSubprocessOutput(...)` to reconcile raw assistant text, `yield` payloads, structured schemas, `report_finding` data, and abort states. Output is truncated with `MAX_OUTPUT_BYTES` / `MAX_OUTPUT_LINES` before returning to the parent, but the full raw output is still written to `<id>.md`. +20. After all sync tasks finish, `#executeSync(...)` aggregates usage, collects artifact paths, and if isolation was used merges results back: + - branch mode: cherry-pick per-task branches with `mergeTaskBranches(...)`, then delete merged branches with `cleanupTaskBranches(...)` + - patch mode: combine non-empty patch artifacts, dry-check with `git.patch.canApplyText(...)`, then apply or leave manual artifacts + - nested repo patches are applied separately with `applyNestedPatches(...)` +21. The final text summary is rendered from `packages/coding-agent/src/prompts/tools/task-summary.md` and includes `agent://<id>` handles for outputs that exist. + +## Modes / Variants +- Execution mode + - Sync inline execution — default path. + - Async background execution — one async job per task item when `async.enabled` is on and the chosen agent is not marked `blocking`. +- Simple mode + - `default` — accepts shared `context` and per-call `schema`. + - `schema-free` — accepts `context`, rejects `schema`. + - `independent` — rejects `context` and `schema`; each assignment stands alone. +- Isolation backend + - `none` — no isolation. + - `worktree` — detached git worktree plus baseline replay. + - `fuse-overlay` — Unix FUSE overlay mount. + - `fuse-projfs` — Windows ProjFS overlay. +- Isolation merge strategy + - Patch mode — capture/apply root patches, keep patch artifacts when application fails. + - Branch mode — commit each task onto `omp/task/<id>` branch, cherry-pick into parent, preserve failed branches for manual resolution. +- Agent source + - Project custom agents — nearest project config/plugin agent directories, first by source-family precedence. + - User custom agents — user config/plugin agent directories after project dirs of the same source family. + - Bundled agents — appended last from `packages/coding-agent/src/task/agents.ts`. +- Bundled agent types + - `explore` — read-only scout with structured handoff output. + - `plan` — architecture/planning agent; may spawn `explore`. + - `designer` — UI/UX specialist. + - `reviewer` — review agent with `report_finding` extraction. + - `task` — general-purpose worker with full capabilities. + - `quick_task` — low-reasoning mechanical worker using the same task prompt body. + - `librarian` — source-grounded external API/library researcher. + +## Side Effects +- Filesystem + - Writes `context.md`, `<id>.jsonl`, and `<id>.md` under the session artifacts dir or a temp task dir. + - In isolated patch mode writes `<id>.patch` artifacts. + - Creates/removes worktrees or overlay mount directories. + - In branch mode creates temporary worktrees and task branches. +- Network + - Child sessions may use whichever networked tools/models their active tool set permits. + - MCP proxy tools can call existing parent MCP connections with a 60_000 ms timeout. +- Subprocesses / native bindings + - `fuse-overlayfs` and `fusermount`/`fusermount3` for FUSE isolation. + - ProjFS native bindings via `@oh-my-pi/pi-natives` on Windows. + - Git operations for baseline capture, patch apply, worktrees, branches, stash, cherry-pick, commits. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Creates child `AgentSession` instances with isolated settings snapshots. + - Registers async jobs in `session.asyncJobManager` for background task mode. + - Emits `task:subagent:event`, `task:subagent:progress`, and `task:subagent:lifecycle` on the parent event bus. + - Allocates session-scoped output ids through `AgentOutputManager` so `agent://` remains unique across invocations and resumes. + - Shares the parent `local://` root with subagents by passing `localProtocolOptions` through `createAgentSession(...)`. +- User-visible prompts / interactive UI + - Async mode streams aggregate progress updates. + - Missing-`yield` recovery sends up to three internal reminder prompts to the child session. + - Final summaries include `<system-notification>` blocks for isolation fallbacks or merge failures. +- Background work / cancellation + - Parent abort stops scheduling new work, aborts active child sessions, and marks unscheduled tasks as skipped. + - Async jobs keep their own cancellation via `AsyncJobManager`. + +## Limits & Caps +- Per-subagent output truncation: `MAX_OUTPUT_BYTES = 500_000` and `MAX_OUTPUT_LINES = 5000` in `packages/coding-agent/src/task/types.ts`. Full raw output is still written to `<id>.md` before truncation is returned to the caller. +- Progress coalescing in child execution: `PROGRESS_COALESCE_MS = 150` in `packages/coding-agent/src/task/executor.ts`. +- Recent output tail for progress: `RECENT_OUTPUT_TAIL_BYTES = 8 * 1024` and `recentOutput` keeps the last 8 non-empty lines in `packages/coding-agent/src/task/executor.ts`. +- Missing-`yield` reminder retries: `MAX_YIELD_RETRIES = 3` in `packages/coding-agent/src/task/executor.ts`. +- MCP proxy timeout: `MCP_CALL_TIMEOUT_MS = 60_000` in `packages/coding-agent/src/task/executor.ts`. +- Task id schema cap: `tasks[].id` `maxLength: 48` in `packages/coding-agent/src/task/types.ts`. +- Prompt text says ids should be `≤32` chars, but the runtime schema allows 48; this mismatch is real. +- Async/full sync parallelism both use `task.maxConcurrency` from settings: + - sync path: `mapWithConcurrencyLimit(...)` + - async path: `Semaphore(...)` around job bodies +- Recursion depth gate: `task.maxRecursionDepth` from settings; `packages/coding-agent/src/tools/index.ts` hides the `task` tool at or beyond the limit, and `runSubprocess(...)` also strips child `task` access at max depth. +- Final inline summary preview per task uses `fullOutputThreshold = 5000` chars in `packages/coding-agent/src/task/index.ts`; longer outputs are summarized while `agent://<id>` points to the full artifact. + +## Errors +- Most validation failures are returned as normal tool text with empty `results`, not thrown: + - invalid simple-mode fields + - unknown/disabled agent + - missing tasks + - missing/duplicate task ids + - spawn-policy denial + - requesting `isolated` while isolation mode is `none` +- Isolated execution without a git repo returns `Isolated task execution requires a git repository. ...`. +- Backend resolution can return a hard error (`ProjFS isolation initialization failed...`) or a non-fatal warning with fallback to `worktree`. +- `mapWithConcurrencyLimit(...)` fails fast on non-abort worker exceptions; already completed results are preserved only in the thrown path’s local state, not surfaced unless the caller catches and converts them. +- Child-session failures surface as `SingleResult.exitCode = 1` with `stderr`/`error` populated. +- If the child omits `yield`, `finalizeSubprocessOutput(...)` injects warnings such as `SYSTEM WARNING: Subagent exited without calling yield tool after 3 reminders.` +- Async scheduling failures are accumulated per task; if no jobs start, the tool returns `Failed to start background task jobs: ...`. +- `agent://<id>` resolution errors are model-visible when another tool reads them: no session, no artifacts dir, missing id, conflicting extraction syntax, or invalid JSON for extraction. + +## Notes +- Agent discovery precedence is first-wins by exact name: project dirs before user dirs within a source family, plugin agent dirs after config dirs, bundled agents last. See `packages/coding-agent/src/task/discovery.ts` and `docs/task-agent-discovery.md`. +- `TaskTool.create(...)` caches discovered agents only for description rendering and the async blocking-agent decision. `#executeSync(...)` rediscovers agents each call. +- Custom agent frontmatter can override bundled agents by name. Bundled definitions are embedded at build time in `packages/coding-agent/src/task/agents.ts`. +- Child sessions do not inherit conversation history automatically. The only built-in carry-over is shared `context`, optional `context.md`, workspace tree/skills/context files, and shared `local://` root. +- `Settings.isolated(...)` gives each child a session-isolated settings snapshot; tool enablement is recomputed inside the child session rather than sharing mutable parent tool state. +- When the parent passes `mcpManager`, child sessions disable standalone MCP discovery and instead get proxy tools that reuse the parent connections. +- Plan mode mutates an `effectiveAgent` with a read-only tool subset and plan-mode prompt text, but `runSubprocess(...)` is still invoked with `agent` rather than `effectiveAgent`. Model/thinking/schema overrides use the effective agent; prompt/tool/spawn restrictions do not fully flow through this call path. +- Branch-mode merge temporarily stashes the parent repo before cherry-picking task branches. A stash-pop conflict is treated as merge failure and leaves recovery state behind. +- Patch-mode only applies combined root patches if every successful task produced a patch and `git.patch.canApplyText(...)` succeeds. +- Nested git repos are handled separately from the root repo. They are copied into isolated worktrees, diffed independently, and merged later with `applyNestedPatches(...)` because parent git cannot track their file-level changes. +- `agent://` ids are numeric-prefixed (`0-Task`, `1-Task`, nested like `0-Parent.0-Child`) by `AgentOutputManager`; this is what prevents artifact collisions across repeated or nested task invocations. diff --git a/docs/tools/todo_write.md b/docs/tools/todo_write.md new file mode 100644 index 000000000..1dfaad130 --- /dev/null +++ b/docs/tools/todo_write.md @@ -0,0 +1,161 @@ +# todo_write + +> Applies ordered mutations to the session todo list and returns a text summary plus the full phase/task state. + +## Source +- Entry: `packages/coding-agent/src/tools/todo-write.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/todo-write.md` +- Key collaborators: + - `packages/coding-agent/src/tools/index.ts` — registers tool, exposes session hooks, gates availability. + - `packages/coding-agent/src/modes/controllers/event-controller.ts` — updates the visible todo UI on tool completion. + - `packages/coding-agent/src/session/agent-session.ts` — stores cached phases, auto-clears done/dropped tasks, emits failure reminders. + - `packages/coding-agent/src/modes/controllers/todo-command-controller.ts` — `/todo` command path, custom-entry persistence, transcript reminder injection. + - `packages/coding-agent/src/tools/render-utils.ts` — collapsed-preview cap for renderer trees. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `ops` | `TodoOpEntry[]` | Yes | Ordered operations to apply. `minItems: 1`. + +### `TodoOpEntry` + +| Op | Required fields | Optional fields | Effect | +| --- | --- | --- | --- | +| `init` | `list` | None of the other fields are used | Replaces the entire list with `list`; every new task starts `pending` before normalization. | +| `start` | `task` | None | Marks one task `in_progress`; any other `in_progress` task is demoted to `pending`. | +| `done` | `task` or `phase` or neither | None | Marks the target task, phase, or all tasks `completed`. | +| `drop` | `task` or `phase` or neither | None | Marks the target task, phase, or all tasks `abandoned`. | +| `rm` | `task` or `phase` or neither | None | Removes the target task, clears the phase's task list, or clears all task lists. | +| `append` | `phase`, `items` | None | Appends new `pending` tasks to a phase; creates the phase if missing. | +| `note` | `task`, `text` | None | Appends one trimmed note string to the task's `notes` array. | + +### Fields used inside ops + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `op` | `"init" | "start" | "done" | "rm" | "drop" | "append" | "note"` | Yes | Operation discriminator. | +| `list` | `{ phase: string; items: string[] }[]` | For `init` | Full replacement payload. Each `items` array has `minItems: 1`. | +| `task` | `string` | For `start`; for task-targeted `done`/`drop`/`rm`/`note` | Exact task content match. | +| `phase` | `string` | For `append`; for phase-targeted `done`/`drop`/`rm` | Exact phase name match, except `append` lazily creates a missing phase. | +| `items` | `string[]` | For `append` | Tasks to append. `minItems: 1`. | +| `text` | `string` | For `note` | Note text; trailing whitespace is stripped before storing. Empty-after-trim is rejected. | + +## Outputs +The tool returns a single-shot `AgentToolResult`: + +- `content`: one text part containing the summary from `formatSummary(...)`. + - Empty final state with no errors: `Todo list cleared.` + - Non-empty final state: remaining-item list, current phase progress, then a per-phase tree. + - If the active `in_progress` task has notes, the summary includes the note bodies inline. + - If any op produced validation/runtime errors, the summary starts with `Errors: ...` but still returns the mutated state. +- `details`: + - `phases: TodoPhase[]` + - `storage: "session" | "memory"` + +`TodoPhase` / `TodoItem` state model: + +- `TodoPhase`: `{ name: string, tasks: TodoItem[] }` +- `TodoItem`: `{ content: string, status: "pending" | "in_progress" | "completed" | "abandoned", notes?: string[] }` + +The TUI renderer (`todoWriteToolRenderer`) merges call and result into one transcript block, renders phases as a tree, shows note counts as superscripts, and renders the note bodies only for the current `in_progress` task. Collapsed transcript previews cap tree items at `PREVIEW_LIMITS.COLLAPSED_ITEMS` (`8`). + +## Flow +1. `TodoWriteTool.execute(...)` clones the current cached phases from `session.getTodoPhases?.() ?? []` (`packages/coding-agent/src/tools/todo-write.ts`). +2. `applyParams(...)` walks `params.ops` in order and applies each entry with `applyEntry(...)`. +3. Each op mutates the working phase array: + - `initPhases(...)` rebuilds the list from scratch. + - `start` resolves a task by exact `content`, demotes every other `in_progress` task to `pending`, then marks the target `in_progress`. + - `done` / `drop` use `getTaskTargets(...)` to target one task, one phase, or every task. + - `rm` removes one task, clears one phase's `tasks`, or clears all phases' task arrays. + - `appendItems(...)` resolves or creates the target phase and pushes new `pending` tasks unless the same task content already exists anywhere. + - `note` trims trailing whitespace, rejects empty text, and appends the note to `task.notes`. +4. Missing task/phase references are recorded in an `errors` array by `resolveTaskOrError(...)` / `resolvePhaseOrError(...)`; execution continues through the rest of the batch. +5. After the full batch, `normalizeInProgressTask(...)` enforces the single-active-task invariant: + - if multiple tasks are `in_progress`, only the first stays active and the rest become `pending`; + - if none are `in_progress`, the first `pending` task in phase/task order is auto-promoted to `in_progress`. +6. `execute(...)` stores the normalized phases with `session.setTodoPhases?.(...)` and reports `storage` as `"session"` when `session.getSessionFile()` exists, else `"memory"`. +7. The agent runtime also watches `todo_write` tool results in `packages/coding-agent/src/session/agent-session.ts`; successful results refresh cached todos, failed results inject a hidden next-turn reminder telling the model that todo progress is not visible until it retries. +8. The event controller updates the visible todo UI from `result.details.phases` on success, or shows a warning on error (`packages/coding-agent/src/modes/controllers/event-controller.ts`). + +## Modes / Variants +### State transitions + +| Current status | `start` | `done` | `drop` | `rm` | `append` | `note` | +| --- | --- | --- | --- | --- | --- | --- | +| `pending` | `in_progress` on target | `completed` | `abandoned` | Removed | New tasks enter as `pending` | No status change | +| `in_progress` | Target stays `in_progress`; non-target active tasks become `pending` | `completed` | `abandoned` | Removed | No status change | No status change | +| `completed` | Can be set back to `in_progress` if targeted | Stays `completed` | Becomes `abandoned` if targeted | Removed | No status change | No status change | +| `abandoned` | Can be set back to `in_progress` if targeted | Becomes `completed` if targeted | Stays `abandoned` | Removed | No status change | No status change | + +Normalization then re-applies the single-active-task rule after the full op batch. + +### Op targeting rules +- `done`, `drop`, `rm`: + - `task` set: affect one exact-content task. + - else `phase` set: affect every task in that exact-name phase. + - else: affect every task in every phase. +- `append` is the only op that creates a missing phase. +- `note` only targets a single task. +- `init` discards previous phases entirely. + +### Markdown round-trip helpers +The same file also exposes non-tool helpers used by `/todo`: +- `phasesToMarkdown(...)` serializes phases as headings plus checklist items (`[ ]`, `[/]`, `[x]`, `[-]`) with blockquote note bodies. +- `markdownToPhases(...)` parses that format, defaults orphan tasks into a `Todos` phase, accepts `>` as an `in_progress` marker and `~` as `abandoned`, and runs the same normalization step. + +## Side Effects +- Filesystem + - None in the tool itself. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Mutates the session todo cache through `setTodoPhases`. + - `storage` reports whether the session has a backing session file, but the tool does not append a custom session entry itself. + - Successful tool-result messages carry `details.phases`; `getLatestTodoPhasesFromEntries(...)` can reconstruct state later from those transcript entries. + - Failed `todo_write` results cause `agent-session` to enqueue a hidden next-turn reminder (`customType: "todo-write-error-reminder"`). +- User-visible prompts / interactive UI + - Transcript block is rendered by `todoWriteToolRenderer` and merged with the call line. + - `event-controller` updates the visible todo panel from successful results. + - On error, `event-controller` shows `Todo update failed...`; the visible panel may stay stale until a later successful call. +- Background work / cancellation + - `AgentSession.setTodoPhases(...)` schedules auto-clear timers for `completed` / `abandoned` tasks via `tasks.todoClearDelay`. + +## Limits & Caps +- `ops` array: `minItems: 1` (`todoWriteSchema`). +- `init.list[*].items`: `minItems: 1`. +- `append.items`: `minItems: 1`. +- Renderer collapsed preview: `PREVIEW_LIMITS.COLLAPSED_ITEMS = 8` (`packages/coding-agent/src/tools/render-utils.ts`). +- Auto-clear delay: `tasks.todoClearDelay` default `60` seconds; `< 0` disables auto-clear, `0` clears on the next microtask (`packages/coding-agent/src/session/agent-session.ts`). +- Tool execution mode: `concurrency = "exclusive"`, `strict = true`, `loadMode = "discoverable"`. + +## Errors +- The tool does not throw for ordinary bad op payloads; it accumulates human-readable strings in `errors` and still returns success with the mutated state. +- Error strings come from the helpers in `packages/coding-agent/src/tools/todo-write.ts`, including: + - `Missing list for init operation` + - `Missing task content` + - `Task "..." not found` with an extra empty-list hint when applicable + - `Missing phase name` + - `Phase "..." not found` + - `Missing phase name for append operation` + - `Missing items for append operation` + - `Task "..." already exists` + - `Missing text for note operation` +- Because ops are processed in order, earlier errors do not roll back later ops. +- Runtime-level tool failure is handled outside the tool body: `agent-session` injects a hidden reminder and the event controller warns the user that visible progress may be stale. +- Idempotency is op-specific: + - `init` is a full replacement; replaying the same payload yields the same state. + - `start`, `done`, and `drop` are effectively idempotent on an existing target state, but `start` also demotes any other active task. + - `rm` is not idempotent for targeted removals: the second call errors because the task or phase is gone. + - `append` is not idempotent: duplicate task content is rejected with `Task "..." already exists`. + - `note` is append-only and never idempotent; replaying it adds another note entry. + +## Notes +- Task lookup is exact string equality inside the tool. The model-facing prompt says task content and phase names are identifiers and should stay unique; `append` enforces task uniqueness globally, but `init` does not validate duplicate task or phase names. +- `findTaskByContent(...)` returns the first matching task across phases. Duplicate task contents make later targeted ops ambiguous. +- `normalizeInProgressTask(...)` runs after the whole batch, not after each op. A single call can intentionally build an intermediate invalid state and rely on final normalization. +- `storage: "session"` means the session has a session-file backing; it does not mean this tool wrote a durable custom entry. +- Reload persistence differs by path: + - plain `todo_write` calls survive in transcript tool-result details; + - `/todo` command edits additionally append `customType: "user_todo_edit"` entries and inject a visible-to-model `<system-reminder>` developer message describing the manual edit. +- On session resume, `AgentSession.#syncTodoPhasesFromBranch()` strips `completed` and `abandoned` tasks before restoring the cached list. The `/todo` command works around that by reading the latest transcript/custom-entry state so historical done/dropped tasks still appear to the user. +- Tool availability is gated by `todo.enabled`, and the registry excludes it when `includeYield` is enabled (`packages/coding-agent/src/tools/index.ts`). +- Subagents do not inherit `todo_write`; `packages/coding-agent/src/task/executor.ts` filters it out as a parent-owned tool. diff --git a/docs/tools/web_search.md b/docs/tools/web_search.md new file mode 100644 index 000000000..55e6b153c --- /dev/null +++ b/docs/tools/web_search.md @@ -0,0 +1,224 @@ +# web_search + +> Run one web query through the first available search provider and return LLM-formatted answer, source URLs, and optional citations. + +## Source +- Entry: `packages/coding-agent/src/web/search/index.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/web-search.md` +- Key collaborators: + - `packages/coding-agent/src/web/search/provider.ts` — lazy provider registry; availability chain. + - `packages/coding-agent/src/web/search/types.ts` — unified `SearchResponse` / `SearchProviderError` types. + - `packages/coding-agent/src/web/search/render.ts` — TUI renderer details type. + - `packages/coding-agent/src/web/search/providers/base.ts` — provider interface and shared params contract. + - `packages/coding-agent/src/web/search/providers/utils.ts` — credential lookup; source normalization. + - `packages/coding-agent/src/web/search/providers/anthropic.ts` — Claude web-search provider. + - `packages/coding-agent/src/web/search/providers/brave.ts` — Brave Search API adapter. + - `packages/coding-agent/src/web/search/providers/codex.ts` — OpenAI Codex SSE adapter. + - `packages/coding-agent/src/web/search/providers/exa.ts` — Exa API or MCP adapter. + - `packages/coding-agent/src/web/search/providers/gemini.ts` — Gemini grounding SSE adapter. + - `packages/coding-agent/src/web/search/providers/jina.ts` — Jina Reader search adapter. + - `packages/coding-agent/src/web/search/providers/kagi.ts` — Kagi provider wrapper. + - `packages/coding-agent/src/web/search/providers/kimi.ts` — Kimi search adapter. + - `packages/coding-agent/src/web/search/providers/parallel.ts` — Parallel provider wrapper. + - `packages/coding-agent/src/web/search/providers/perplexity.ts` — Perplexity API / OAuth adapter. + - `packages/coding-agent/src/web/search/providers/searxng.ts` — self-hosted SearXNG adapter. + - `packages/coding-agent/src/web/search/providers/synthetic.ts` — Synthetic search adapter. + - `packages/coding-agent/src/web/search/providers/tavily.ts` — Tavily search adapter. + - `packages/coding-agent/src/web/search/providers/zai.ts` — Z.AI remote MCP adapter. + - `packages/coding-agent/src/web/parallel.ts` — Parallel search/extract HTTP client. + - `packages/coding-agent/src/web/kagi.ts` — Kagi HTTP client. + - `packages/coding-agent/src/tools/index.ts` — built-in tool registration and enable flag. + +## Inputs + +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `query` | `string` | Yes | Search query. `executeSearch()` rewrites any `2020`-`2029` substring to the current year before dispatch. | +| `recency` | `"day" \| "week" \| "month" \| "year"` | No | Time filter. Only providers that implement it use it. Prompt text says Brave and Perplexity; code also maps it for Tavily and SearXNG. | +| `limit` | `number` | No | Max results to return. Usually becomes the provider request's result-count parameter when `num_search_results` is absent. | +| `max_tokens` | `number` | No | Passed through as `maxOutputTokens` / `max_tokens` only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. | +| `temperature` | `number` | No | Passed through only by Anthropic, Gemini, and Perplexity API-key mode. Ignored by the other providers. | +| `num_search_results` | `number` | No | Requested upstream search breadth. For most providers this is the same count used for returned sources. Perplexity is the only adapter that keeps it distinct from `limit`. | + +## Outputs +The tool returns a single text content block plus structured `details`. + +- `content`: `[{ type: "text", text: string }]` +- `details`: `SearchRenderDetails` from `packages/coding-agent/src/web/search/render.ts` + - `response: SearchResponse` + - `error?: string` + +`text` is produced by `formatForLLM()` in `packages/coding-agent/src/web/search/index.ts`: + +- If `response.answer` exists, it is emitted first. +- If sources exist, a `## Sources` section follows with a source count, then one entry per source: + - `[n] <title> (<formatted age or published date>)` + - ` <url>` + - optional snippet line truncated to 240 chars. +- If citations exist, a `## Citations` section follows with URL/title plus optional cited text truncated to 240 chars. +- If related questions exist, a `## Related` bullet list follows. +- If search queries exist, a `Search queries: <n>` section follows, capped to the first 3 queries and 120 chars each. + +Failure output is not thrown at the tool boundary when at least one provider was attempted. Instead the tool returns: + +- `content[0].text = "Error: ..."` +- `details.response.provider = <last attempted provider> | "none"` +- `details.error = ...` + +Streaming: none. `WebSearchTool.execute()` does not forward its `_signal` argument into `executeSearch()`, so provider cancellation is only available to internal callers that place `signal` inside `SearchQueryParams`. + +## Flow +1. `WebSearchTool.execute()` in `packages/coding-agent/src/web/search/index.ts` delegates directly to `executeSearch()`. +2. `executeSearch()` chooses a provider list: + - if `params.provider` is set and not `"auto"`, it loads that provider with `getSearchProvider()`; if `isAvailable()` returns true, the list is `[that provider]`, otherwise it falls back to `resolveProviderChain("auto")`. + - otherwise it calls `resolveProviderChain()` with the module-global preferred provider from `packages/coding-agent/src/web/search/provider.ts`. +3. `resolveProviderChain()` lazily loads each provider module on demand, checks `isAvailable()`, and returns only available providers. If a preferred provider is set, it is tried first, then the static `SEARCH_PROVIDER_ORDER` excluding that provider. +4. If no providers are available, `executeSearch()` returns `Error: No web search provider configured.` with `details.response.provider = "none"`. +5. For each provider in order, `executeSearch()` calls `provider.search()` with: + - `query` after year-rewrite, + - `limit`, `recency`, `temperature`, `maxOutputTokens`, `numSearchResults`, + - `systemPrompt` from `packages/coding-agent/src/prompts/tools/web-search.md`. +6. On the first successful `SearchResponse`, `formatForLLM()` renders answer/sources/citations/related/search-queries into one text block and returns it with `details.response`. +7. If a provider throws, `executeSearch()` records the error and tries the next provider. There is no provider-level parallel fan-out; fallback is sequential. +8. After all candidates fail, `formatProviderError()` normalizes the last error: + - Anthropic `404` becomes `Anthropic web search returned 404 (model or endpoint not found).` + - `401`/`403` become `<Provider> authorization failed ...` except Z.AI, which preserves its raw message. + - other `SearchProviderError`s surface `error.message`. +9. If more than one provider was attempted, the final message is `All web search providers failed (<labels>). Last error: <message>`; otherwise it is just the normalized last error. + +## Modes / Variants +- **Provider selection** + - **Forced provider**: internal callers may pass `provider`; unavailable forced providers fall back to the auto chain instead of hard-failing (`packages/coding-agent/src/web/search/index.ts`). This field is not in the model-facing schema. + - **Preferred provider**: `setPreferredSearchProvider()` sets a module-global default used by `resolveProviderChain()`. `packages/coding-agent/src/sdk.ts` and `packages/coding-agent/src/modes/controllers/selector-controller.ts` wire this from settings. + - **Auto chain order**: `tavily`, `perplexity`, `brave`, `jina`, `kimi`, `anthropic`, `gemini`, `codex`, `zai`, `exa`, `parallel`, `kagi`, `synthetic`, `searxng` (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/provider.ts`). +- **Provider adapters** + - **Tavily** — `packages/coding-agent/src/web/search/providers/tavily.ts` + - Availability: API key from env or `agent.db` via `findCredential()`. + - Querying: POST `https://api.tavily.com/search`. + - `recency` maps to Tavily `time_range`; code explicitly keeps `topic` at default general scope instead of narrowing to news. + - `limit` / `num_search_results`: adapter uses `params.numSearchResults ?? params.limit`, clamped to `5..20` with default `5`. + - Output: `answer`, `sources`, `requestId`, `authMode: "api_key"`. + - **Perplexity** — `packages/coding-agent/src/web/search/providers/perplexity.ts` + - Availability: auth precedence is `PERPLEXITY_COOKIES` -> OAuth token in `agent.db` -> `PERPLEXITY_API_KEY` / `PPLX_API_KEY`. + - OAuth/cookie mode: POSTs to `https://www.perplexity.ai/rest/sse/perplexity_ask`, consumes SSE, merges partial events, extracts answer and source URLs, sets `authMode: "oauth"`. + - API-key mode: POSTs to `https://api.perplexity.ai/chat/completions` with `model: "sonar-pro"`, `search_mode: "web"`, `num_search_results`, optional `search_recency_filter`, `max_tokens`, `temperature`. + - `num_search_results` controls upstream API breadth only in API-key mode. `limit` is preserved separately as `num_results` and slices returned `sources` after parsing in both auth modes. + - Output may include `answer`, `sources`, `citations`, `usage`, `model`, `requestId`, `authMode`. + - **Brave** — `packages/coding-agent/src/web/search/providers/brave.ts` + - Availability: `BRAVE_API_KEY` only. + - Querying: GET `https://api.search.brave.com/res/v1/web/search` with `count`, `extra_snippets=true`, and `freshness=pd|pw|pm|py` for `recency`. + - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. + - Output: `sources`, `requestId`. + - **Jina** — `packages/coding-agent/src/web/search/providers/jina.ts` + - Availability: `JINA_API_KEY` only. + - Querying: GET-like fetch to `https://s.jina.ai/<encoded query>` with bearer auth. + - Ignores `recency`, `max_tokens`, and `temperature`. + - `limit` / `num_search_results`: adapter slices sources to `params.numSearchResults ?? params.limit` when provided; otherwise returns all payload items. + - Output: `sources` only. + - **Kimi** — `packages/coding-agent/src/web/search/providers/kimi.ts` + - Availability: `MOONSHOT_SEARCH_API_KEY`, `KIMI_SEARCH_API_KEY`, `MOONSHOT_API_KEY`, or `agent.db` credentials for `moonshot` / `kimi-code`. + - Querying: POST to `MOONSHOT_SEARCH_BASE_URL` / `KIMI_SEARCH_BASE_URL` / default `https://api.kimi.com/coding/v1/search` with `text_query`, `limit`, `enable_page_crawling`, `timeout_seconds: 30`. + - `limit` / `num_search_results`: `params.numSearchResults ?? params.limit`, clamped to `1..20`, default `10`. + - Output: `sources`, `requestId`. + - **Anthropic** — `packages/coding-agent/src/web/search/providers/anthropic.ts` + - Availability: `findAnthropicAuth()` from `@oh-my-pi/pi-ai`. + - Querying: Claude Messages API with web-search tool enabled. + - `max_tokens` and `temperature` pass through. + - `limit` and `num_search_results` are collapsed together before dispatch: `num_results = params.numSearchResults ?? params.limit`. + - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage.searchRequests`, `model`, `requestId`. + - **Gemini** — `packages/coding-agent/src/web/search/providers/gemini.ts` + - Availability: OAuth credentials in `agent.db` for `google-gemini-cli` or `google-antigravity`. + - Querying: SSE `streamGenerateContent` call with Google Search grounding enabled. Antigravity auth tries two fallback endpoints and retries `401/403/400 invalid auth` once after token refresh; `429/5xx` retry with exponential backoff and server-provided retry delay, capped by a `5 * 60 * 1000` ms rate-limit budget. + - `max_tokens` and `temperature` pass through as `generationConfig.maxOutputTokens` / `generationConfig.temperature`. + - `limit` and `num_search_results` are collapsed together before dispatch. + - Output may include `answer`, `sources`, `citations`, `searchQueries`, `usage`, `model`. + - **Codex** — `packages/coding-agent/src/web/search/providers/codex.ts` + - Availability: non-expired OAuth credential for `openai-codex` in `agent.db`. + - Querying: SSE POST to `https://chatgpt.com/backend-api/codex/responses` with `tool_choice: { type: "web_search" }` and `search_context_size: "high"` by default. + - Ignores `recency`, `max_tokens`, and `temperature` in this tool path. + - `limit` and `num_search_results` are collapsed together before dispatch. + - Output may include `answer`, `sources`, `usage`, `model`, `requestId`. If the streamed response has no `url_citation` annotations, the adapter falls back to scraping markdown links and bare URLs from the answer text. + - **Z.AI** — `packages/coding-agent/src/web/search/providers/zai.ts` + - Availability: env or `agent.db` credential for `zai`. + - Querying: JSON-RPC `tools/call` against `https://api.z.ai/api/mcp/web_search_prime/mcp` for remote MCP tool `web_search_prime`. + - Fallback chain inside the provider: tries `{query,count}`, then `{search_query,count}`, then `{search_query, search_engine:"search-prime", count}` when earlier attempts fail with argument-shape errors. + - `limit` and `num_search_results` are collapsed together before dispatch. + - Output may include parsed free-text `answer`, `sources`, `requestId`. + - **Exa** — `packages/coding-agent/src/web/search/providers/exa.ts` + - Availability: always true unless settings explicitly disable `exa.enabled` or `exa.enableSearch`; the adapter can use public MCP even without `EXA_API_KEY`. + - Querying: with `EXA_API_KEY`, POST `https://api.exa.ai/search`; otherwise call MCP tool `web_search_exa`. + - `limit` and `num_search_results` are collapsed together before dispatch. + - Output: synthesized `answer` from up to 3 result summaries, `sources`, `requestId`. + - **Parallel** — `packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts` + - Availability: env or `agent.db` credential for `parallel`. + - Querying: POST `https://api.parallel.ai/v1beta/search` with `objective=query`, `search_queries=[query]`, `mode:"fast"`, `max_chars_per_result: 10000`, beta header `search-extract-2025-10-10`. + - There is no provider fan-out here despite the name; the current adapter always sends a one-element `search_queries` array. + - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. + - Output: `sources`, `requestId`. + - **Kagi** — `packages/coding-agent/src/web/search/providers/kagi.ts`, `packages/coding-agent/src/web/kagi.ts` + - Availability: env or `agent.db` credential for `kagi`. + - Querying: GET `https://kagi.com/api/v0/search?q=<query>&limit=<n>` with `Authorization: Bot <key>`. + - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..40`, default `10`. + - Output: `sources`, `relatedQuestions`, `requestId`. + - **Synthetic** — `packages/coding-agent/src/web/search/providers/synthetic.ts` + - Availability: env or `agent.db` credential for `synthetic`. + - Querying: POST `https://api.synthetic.new/v2/search` with `{ query }`. + - Ignores `recency`, `max_tokens`, and `temperature`. + - `limit` and `num_search_results` are collapsed together before dispatch. + - Output: `sources` only. + - **SearXNG** — `packages/coding-agent/src/web/search/providers/searxng.ts` + - Availability: endpoint from `searxng.endpoint` setting or `SEARXNG_ENDPOINT` env. + - Querying: GET `<endpoint>/search?format=json&q=...`; optional settings add `categories` and `language`. + - Auth precedence: Basic auth (`searxng.basicUsername` / `searxng.basicPassword` or env equivalents) over bearer token (`searxng.token` / `SEARXNG_TOKEN`). Basic credentials are validated for RFC 7617 restrictions. + - `recency` maps to `time_range`; `week` is downgraded to `month` because SearXNG does not support week. + - `limit` and `num_search_results` are collapsed together before dispatch, clamped to `1..20`, default `10`. + - Output: `sources`, `relatedQuestions` from `suggestions`. + +## Side Effects +- Network + - Calls one or more external search providers over HTTPS until one succeeds or all fail. + - Provider-specific transports include JSON POST, JSON GET, SSE streaming (Perplexity OAuth/API, Gemini, Codex), and JSON-RPC over HTTP (Z.AI). +- Subprocesses / native bindings + - None. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Uses a module-global provider-instance cache in `packages/coding-agent/src/web/search/provider.ts`. + - Uses a module-global preferred-provider setting in the same file. + - `packages/coding-agent/src/tools/index.ts` gates tool availability behind `session.settings.get("web_search.enabled")`. +- Background work / cancellation + - Many provider adapters accept `AbortSignal`, but `WebSearchTool.execute()` does not pass its `_signal` into `executeSearch()`. Internal callers can still use cancellation by calling `runSearchQuery()` / `executeSearch()` with `signal` embedded in params. + +## Limits & Caps +- Provider auto-order length: 14 providers (`SEARCH_PROVIDER_ORDER` in `packages/coding-agent/src/web/search/provider.ts`). +- `formatForLLM()` truncates source snippets and citation text to 240 chars (`packages/coding-agent/src/web/search/index.ts`). +- `formatForLLM()` emits at most 3 search queries, each truncated to 120 chars (`packages/coding-agent/src/web/search/index.ts`). +- Brave result count: default `10`, max `20` (`DEFAULT_NUM_RESULTS`, `MAX_NUM_RESULTS` in `packages/coding-agent/src/web/search/providers/brave.ts`). +- Tavily result count: default `5`, max `20` (`packages/coding-agent/src/web/search/providers/tavily.ts`). +- Kimi result count: default `10`, max `20`; request timeout field fixed to `30` seconds (`packages/coding-agent/src/web/search/providers/kimi.ts`). +- Parallel result count: default `10`, max `40`; per-result excerpt cap `10_000` chars (`packages/coding-agent/src/web/search/providers/parallel.ts`, `packages/coding-agent/src/web/parallel.ts`). +- Kagi result count: default `10`, max `40` (`packages/coding-agent/src/web/search/providers/kagi.ts`). +- SearXNG result count: default `10`, max `20` (`packages/coding-agent/src/web/search/providers/searxng.ts`). +- Perplexity API-key mode defaults: `max_tokens = 8192`, `temperature = 0.2`, `num_search_results = 10` (`packages/coding-agent/src/web/search/providers/perplexity.ts`). +- Anthropic defaults: model `claude-haiku-4-5`, `DEFAULT_MAX_TOKENS = 4096` when the provider omits `max_tokens` (`packages/coding-agent/src/web/search/providers/anthropic.ts`). +- Gemini retries: up to `3` retries per endpoint, base delay `1000` ms, rate-limit delay budget `5 * 60 * 1000` ms (`packages/coding-agent/src/web/search/providers/gemini.ts`). + +## Errors +- Tool-level no-provider case returns a normal tool result with `Error: No web search provider configured.`; it does not throw. +- Tool-level all-failed case also returns a normal tool result with `Error: ...`; failures are summarized from the last attempted provider. +- Provider adapters usually throw `SearchProviderError(provider, message, status)` for HTTP or protocol failures. +- Availability probes intentionally swallow lookup errors and report `false` in many providers via `isApiKeyAvailable()`. +- Per-provider notable failures: + - Anthropic: missing credentials throw a plain `Error`; a `404` is remapped to a special final message by `formatProviderError()`. + - Perplexity: missing auth throws a plain `Error`; OAuth stream `error_code` events become `SearchProviderError("perplexity", ...)`. + - Gemini: auth refresh, endpoint fallback, and retry logic are internal; final exhausted failures surface as `SearchProviderError("gemini", ...)`. + - Codex and Gemini both fail if the HTTP response has no body after a `200`. + - Z.AI treats malformed SSE/JSON-RPC payloads as provider errors and retries only argument-shape failures across request variants. + - SearXNG `findAuth()` can throw configuration errors before any HTTP call if Basic auth fields are incomplete or invalid. + +## Notes +- The model-facing schema does not expose `provider`, but internal callers can force one through `SearchQueryParams`. +- `resolveProviderChain()` lazily imports provider modules and caches singleton instances. Just asking for labels via `getSearchProviderLabel()` does not trigger those imports. +- Most providers treat `limit` and `num_search_results` as the same number because adapters pass `params.numSearchResults ?? params.limit`. Perplexity is the only implementation that preserves both concepts. +- The prompt says `recency` is for Brave and Perplexity, but code also implements it for Tavily and SearXNG. +- The year rewrite in `executeSearch()` is blunt: any `2020`-`2029` substring is replaced with the current year. +- `packages/coding-agent/src/config/settings-schema.ts` exposes provider preferences for `auto`, `exa`, `brave`, `jina`, `kimi`, `perplexity`, `anthropic`, `zai`, `tavily`, `kagi`, `synthetic`, `parallel`, and `searxng`. Gemini and Codex are in the registry and auto chain but not in that settings enum. +- Exa availability is optimistic. Unless settings disable it, the provider stays in the chain even without an API key because it can fall back to MCP. diff --git a/docs/tools/write.md b/docs/tools/write.md new file mode 100644 index 000000000..18b5aa97b --- /dev/null +++ b/docs/tools/write.md @@ -0,0 +1,177 @@ +# write + +> Create or overwrite a file, archive entry, or SQLite row. + +## Source +- Entry: `packages/coding-agent/src/tools/write.ts` +- Model-facing prompt: `packages/coding-agent/src/prompts/tools/write.md` +- Key collaborators: + - `packages/coding-agent/src/tools/archive-reader.ts` — parse `archive.ext:entry` selectors. + - `packages/coding-agent/src/tools/sqlite-reader.ts` — detect SQLite paths and perform row insert/update/delete. + - `packages/coding-agent/src/lsp/index.ts` — format-on-write and diagnostics writethrough. + - `packages/coding-agent/src/tools/auto-generated-guard.ts` — block overwriting generated files. + - `packages/coding-agent/src/tools/fs-cache-invalidation.ts` — invalidate shared FS scan caches after writes. + - `packages/coding-agent/src/tools/plan-mode-guard.ts` — resolve paths and enforce plan-mode write policy. + +## Inputs +| Field | Type | Required | Description | +| --- | --- | --- | --- | +| `path` | `string` | Yes | Target path. Plain file path writes a filesystem file. `archive.ext:inner/path` writes an archive entry for `.tar`, `.tar.gz`, `.tgz`, or `.zip`. `db.sqlite:table` inserts a row. `db.sqlite:table:key` updates or deletes a row. | +| `content` | `string` | Yes | Full replacement file content, archive entry content, or SQLite row payload. SQLite non-delete writes must parse as a JSON5 object. Empty or whitespace-only content deletes a SQLite row when `path` includes a row key. | + +Worked examples: + +```text +path: "src/generated/config.json" +content: "{\n \"enabled\": true\n}\n" +``` + +```text +path: "fixtures/archive.zip:templates/email.txt" +content: "hello\n" +``` + +```text +path: "data/app.sqlite:users:42" +content: "{name: 'Ada', active: true}" +``` + +## Outputs +Single-shot result. + +- Success always returns a text block. + - Plain file write: `Successfully wrote <bytes> bytes to <relative-path>`. + - Archive write: `Successfully wrote <bytes> bytes to <relative-archive-path>:<entry-path>`. + - SQLite write: one of `Inserted row into <table>`, `Updated row '<key>' in <table>`, `No row updated ...`, `Deleted row ...`, `No row deleted ...`. +- If hashline prefixes were copied from `read` output and stripped first, the first text block gets an extra note. +- Plain file writes may also return `details.diagnostics` plus `details.meta.diagnostics` when LSP diagnostics-on-write is enabled. +- SQLite writes use `toolResult(...).sourcePath(...)`, so `details.meta.sourcePath` points at the database file. +- Archive writes return empty `details`. + +## Flow +1. `WriteTool.execute()` in `packages/coding-agent/src/tools/write.ts` strips `LINE+ID|` hashline prefixes from `content` when the session is in hashline display mode. +2. It calls `#resolveArchiveWritePath()` first. That uses `parseArchivePathCandidates()` from `packages/coding-agent/src/tools/archive-reader.ts`, checks candidate archive files on disk, and falls back to the longest matching archive suffix even when the archive file does not exist yet. +3. Archive writes call `enforcePlanModeWrite(..., { op: exists ? "update" : "create" })`, then `#writeArchiveEntry()`. + - The parent directory of the archive file is created with `fs.mkdir(..., { recursive: true })`. + - `.zip` archives are read with `fflate.unzipSync()`, the target entry is replaced in an in-memory map, and the archive is rewritten with `fflate.zipSync()` + `Bun.write()`. + - `.tar`, `.tar.gz`, and `.tgz` archives are read with `Bun.Archive`, existing entries are copied into an object map, the target entry is replaced, and `Bun.Archive.write()` rewrites the archive. + - `invalidateFsScanAfterWrite()` runs on the archive file path. +4. If the path is not treated as an archive, `execute()` calls `#resolveSqliteWritePath()`. That uses `parseSqlitePathCandidates()` and `isSqliteFile()` from `packages/coding-agent/src/tools/sqlite-reader.ts`. Existing non-SQLite files suppress the SQLite path interpretation. +5. SQLite writes call `enforcePlanModeWrite(..., { op: "update" })`, then `#writeSqliteRow()`. + - The database must already exist; missing DBs throw `SQLite database '<path>' not found`. + - The tool opens `new Database(..., { create: false, strict: true })` and sets `PRAGMA busy_timeout = 3000`. + - Whitespace-only `content` with a row key deletes a row. + - Non-empty `content` is parsed with `Bun.JSON5.parse()`, must be a JSON object, and is routed to insert/update helpers from `packages/coding-agent/src/tools/sqlite-reader.ts`. + - `invalidateFsScanAfterWrite()` runs on the DB path and the connection is closed in `finally`. +6. Otherwise the tool treats `path` as a plain filesystem file. + - `enforcePlanModeWrite(..., { op: "create" })` runs before path resolution. + - Existing files are checked by `assertEditableFile()` to block overwriting detected generated files. + - The session’s writethrough callback writes content. With LSP enabled and `lsp.formatOnWrite` / `lsp.diagnosticsOnWrite` settings on, `createLspWritethrough()` may format content, sync it through LSP servers, save it, and collect diagnostics. Otherwise `writethroughNoop()` writes directly with `Bun.write()` or `file.write()`. + - `invalidateFsScanAfterWrite()` runs on the file path. +7. The tool returns a text result and optional diagnostics metadata. + +## Modes / Variants +### Plain file path +- Target is any path that does not resolve as an archive selector and does not resolve as an existing-or-new SQLite selector. +- Existing files are overwritten. +- `write.ts` does not call `fs.mkdir()` on this path; parent-directory creation is only implemented in the archive branch. + +Example: + +```text +path: "tmp/output.txt" +content: "hello\n" +``` + +### Archive entry write +- Selector syntax: `archive.ext:inner/path`. +- Supported archive suffixes come from `parseArchivePathCandidates()`: `.tar`, `.tar.gz`, `.tgz`, `.zip`. +- The inner path is normalized to `/`, strips empty and `.` segments, rejects `..`, and rejects directory targets ending in `/`. +- Rewrites the whole archive file after replacing one entry. +- Creates the parent directory for the archive file if needed. + +Example: + +```text +path: "build/assets.tar.gz:css/app.css" +content: "body { color: black; }\n" +``` + +### SQLite table insert +- Selector syntax: `db.sqlite:table`. +- `content` must parse as a JSON5 object. +- Empty object is allowed and becomes `INSERT INTO <table> DEFAULT VALUES`. +- Query parameters are rejected for SQLite writes. + +Example: + +```text +path: "data/app.db:users" +content: "{name: 'Ada', active: true}" +``` + +### SQLite row update / delete +- Selector syntax: `db.sqlite:table:key`. +- Non-empty `content` updates the row. +- Empty or whitespace-only `content` deletes the row. +- Row lookup uses the single-column primary key if present; otherwise it falls back to `rowid`. Composite primary keys and `WITHOUT ROWID` tables are rejected for key-based writes. + +Example update: + +```text +path: "data/app.sqlite:users:42" +content: "{email: 'ada@example.com'}" +``` + +Example delete: + +```text +path: "data/app.sqlite:users:42" +content: "" +``` + +## Side Effects +- Filesystem + - Creates or overwrites plain files. + - Rewrites entire archive files when writing an archive entry. + - Creates parent directories for archive files only. + - Mutates existing SQLite databases; never creates a new SQLite DB. +- Subprocesses / native bindings + - Uses Bun SQLite bindings via `bun:sqlite`. + - Uses Bun archive APIs and lazily imports `fflate` for ZIP reads/writes. + - May talk to configured LSP servers through `packages/coding-agent/src/lsp/index.ts`. +- Session state (transcript, memory, jobs, checkpoints, registries) + - Invalidates shared filesystem scan cache entries through `invalidateFsScanAfterWrite()`. + - Enforces plan-mode write restrictions before mutating the target. +- Background work / cancellation + - Marks the tool `nonAbortable = true` and `concurrency = "exclusive"` in `WriteTool`. + - LSP writethrough can schedule deferred diagnostics fetches after a timeout, but plain `write.ts` only consumes the immediate return value. + +## Limits & Caps +- `WriteTool` itself exposes no byte cap beyond storing `content` in memory and, for archives, rebuilding the archive in memory. +- Generated-file detection reads at most `CHECK_BYTE_COUNT = 1024` bytes and `HEADER_LINE_LIMIT = 40` header lines from an existing file in `packages/coding-agent/src/tools/auto-generated-guard.ts`. +- SQLite writes set `PRAGMA busy_timeout = 3000`. +- LSP writethrough uses a `5_000` ms operation timeout in `runLspWritethrough()` and may schedule a deferred diagnostics fetch with `AbortSignal.timeout(25_000)` in `scheduleDeferredDiagnosticsFetch()`. + +## Errors +- Invalid archive subpaths throw `ToolError` with messages such as: + - `Archive write path must target a file inside the archive` + - `Archive write path must target a file, not a directory` + - `Archive path cannot contain '..'` +- SQLite path parsing throws on unsupported forms: + - `SQLite write paths do not support query parameters` + - `SQLite write path must target a table` + - `SQLite row writes require a non-empty row key` +- Missing SQLite DBs surface as `SQLite database '<path>' not found`. +- SQLite content errors are model-visible `ToolError`s, including invalid JSON5, non-object payloads, unknown columns, non-scalar values, empty update objects, composite primary keys, and `WITHOUT ROWID` tables. +- Existing plain files may be rejected by `assertEditableFile()` when they look generated. +- Archive read/write failures and unexpected SQLite exceptions are wrapped in `ToolError(error.message)`. +- If no LSP server matches or LSP formatting/diagnostics times out, file writes still fall back to writing content; diagnostics may be omitted. + +## Notes +- Archive path detection runs before SQLite detection. A path that matches an archive selector is never treated as SQLite. +- SQLite detection declines when an existing file with a `.sqlite` / `.db` suffix is present but does not have SQLite magic bytes; then the path falls back to a plain file write. +- ZIP entry content is encoded with `new TextEncoder().encode(content)` in `#writeArchiveEntry()`. Non-ZIP archive writes pass the string directly to `Bun.Archive.write()`. +- The prompt forbids two common anti-patterns: using `write` for routine edits that should use `edit`, and creating `*.md` / `README` files unless explicitly requested. It also forbids emojis unless requested. +- Plain file writes report byte count using `cleanContent.length`, which is UTF-16 code units in JS, not an on-disk byte measurement. +- `stripWriteContent()` only removes hashline prefixes when the session’s file display mode has `hashLines` enabled; otherwise content is written unchanged. diff --git a/packages/agent/CHANGELOG.md b/packages/agent/CHANGELOG.md index ee32dc23c..e8e9584b5 100644 --- a/packages/agent/CHANGELOG.md +++ b/packages/agent/CHANGELOG.md @@ -2,6 +2,10 @@ ## [Unreleased] +### Added + +- Added an `isError?: boolean` field on `AgentToolResult` so tools can flag a non-throwing failure (e.g. an aggregator that catches per-entry errors). `coerceToolResult` preserves the flag and the agent loop surfaces it as a tool error on the wire. + ## [14.9.3] - 2026-05-10 ### Added diff --git a/packages/agent/src/agent-loop.ts b/packages/agent/src/agent-loop.ts index ed9a19d49..63f1dd3dc 100644 --- a/packages/agent/src/agent-loop.ts +++ b/packages/agent/src/agent-loop.ts @@ -58,12 +58,17 @@ function coerceToolResult(raw: unknown): { result: AgentToolResult<any>; malform const rawObj = raw && typeof raw === "object" ? (raw as Record<string, unknown>) : null; const rawContent = rawObj?.content; const details = rawObj && "details" in rawObj ? rawObj.details : {}; + // Tools may flag a non-throwing failure on the result itself (e.g. an + // aggregator that catches per-entry errors and synthesizes a combined + // result). Preserve the flag so agent-loop can surface it on the wire. + const explicitError = Boolean(rawObj && "isError" in rawObj && rawObj.isError); if (!Array.isArray(rawContent)) { return { result: { content: [{ type: "text", text: "Tool returned an invalid result: missing content array." }], details, + isError: true, }, malformed: true, }; @@ -82,7 +87,7 @@ function coerceToolResult(raw: unknown): { result: AgentToolResult<any>; malform content.push(block as { type: "image"; data: string; mimeType: string }); } } - return { result: { content, details }, malformed: false }; + return { result: { content, details, ...(explicitError ? { isError: true } : {}) }, malformed: false }; } /** @@ -827,7 +832,7 @@ async function executeToolCalls( ); const coerced = coerceToolResult(rawResult); result = coerced.result; - if (coerced.malformed) isError = true; + if (coerced.malformed || result.isError) isError = true; } catch (e) { result = { content: [{ type: "text", text: e instanceof Error ? e.message : String(e) }], diff --git a/packages/agent/src/types.ts b/packages/agent/src/types.ts index 44042bdc7..edbfc05ba 100644 --- a/packages/agent/src/types.ts +++ b/packages/agent/src/types.ts @@ -223,6 +223,9 @@ export interface AgentToolResult<T = any, _TInput = unknown> { content: (TextContent | ImageContent)[]; // Details to be displayed in a UI or logged details?: T; + // Marks a non-throwing failure (e.g. an aggregator catching per-entry errors). + // agent-loop honors this and surfaces it as a tool error on the wire. + isError?: boolean; } // Callback for streaming tool execution updates diff --git a/packages/agent/test/agent.test.ts b/packages/agent/test/agent.test.ts index b82a5e81d..eb37be0d9 100644 --- a/packages/agent/test/agent.test.ts +++ b/packages/agent/test/agent.test.ts @@ -1,6 +1,6 @@ import { describe, expect, it } from "bun:test"; import { Agent, type AgentTool, ThinkingLevel } from "@oh-my-pi/pi-agent-core"; -import { getBundledModel, type SimpleStreamOptions, type ThinkingBudgets } from "@oh-my-pi/pi-ai"; +import { getBundledModel, type SimpleStreamOptions } from "@oh-my-pi/pi-ai"; import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; import { Type } from "@sinclair/typebox"; import { createAssistantMessage, pushAlphaThenDoneEvent } from "./helpers"; @@ -8,96 +8,6 @@ import { createAssistantMessage, pushAlphaThenDoneEvent } from "./helpers"; class MockAssistantStream extends AssistantMessageEventStream {} describe("Agent", () => { - it("should create an agent instance with default state", () => { - const agent = new Agent(); - - expect(agent.state).toBeDefined(); - expect(agent.state.systemPrompt).toEqual([]); - expect(agent.state.model).toBeDefined(); - expect(agent.state.thinkingLevel).toBeUndefined(); - expect(agent.state.tools).toEqual([]); - expect(agent.state.messages).toEqual([]); - expect(agent.state.isStreaming).toBe(false); - expect(agent.state.streamMessage).toBe(null); - expect(agent.state.pendingToolCalls).toEqual(new Set()); - expect(agent.state.error).toBeUndefined(); - }); - - it("should create an agent instance with custom initial state", () => { - const customModel = getBundledModel("openai", "gpt-4o-mini"); - const agent = new Agent({ - initialState: { - systemPrompt: ["You are a helpful assistant."], - model: customModel, - thinkingLevel: ThinkingLevel.Low, - }, - }); - - expect(agent.state.systemPrompt).toEqual(["You are a helpful assistant."]); - expect(agent.state.model).toBe(customModel); - expect(agent.state.thinkingLevel).toBe(ThinkingLevel.Low); - }); - - it("should subscribe to events", () => { - const agent = new Agent(); - - let eventCount = 0; - const unsubscribe = agent.subscribe(_event => { - eventCount++; - }); - - // No initial event on subscribe - expect(eventCount).toBe(0); - - // State mutators don't emit events - agent.setSystemPrompt(["Test prompt"]); - expect(eventCount).toBe(0); - expect(agent.state.systemPrompt).toEqual(["Test prompt"]); - - // Unsubscribe should work - unsubscribe(); - agent.setSystemPrompt(["Another prompt"]); - expect(eventCount).toBe(0); // Should not increase - }); - - it("should update state with mutators", () => { - const agent = new Agent(); - - // Test setSystemPrompt - agent.setSystemPrompt(["Custom prompt"]); - expect(agent.state.systemPrompt).toEqual(["Custom prompt"]); - - // Test setModel - const newModel = getBundledModel("google", "gemini-2.5-flash"); - agent.setModel(newModel); - expect(agent.state.model).toBe(newModel); - - // Test setThinkingLevel - agent.setThinkingLevel(ThinkingLevel.High); - expect(agent.state.thinkingLevel).toBe(ThinkingLevel.High); - - // Test setTools - const tools = [{ name: "test", description: "test tool" } as any]; - agent.setTools(tools); - expect(agent.state.tools).toBe(tools); - - // Test replaceMessages - const messages = [{ role: "user" as const, content: "Hello", timestamp: Date.now() }]; - agent.replaceMessages(messages); - expect(agent.state.messages).toEqual(messages); - expect(agent.state.messages).not.toBe(messages); // Should be a copy - - // Test appendMessage - const newMessage = createAssistantMessage([{ type: "text", text: "Hi" }]); - agent.appendMessage(newMessage); - expect(agent.state.messages).toHaveLength(2); - expect(agent.state.messages[1]).toBe(newMessage); - - // Test clearMessages - agent.clearMessages(); - expect(agent.state.messages).toEqual([]); - }); - it("should support steering message queueing", async () => { const agent = new Agent(); @@ -108,13 +18,6 @@ describe("Agent", () => { expect(agent.state.messages).not.toContainEqual(message); }); - it("should handle abort controller", () => { - const agent = new Agent(); - - // Should not throw even if nothing is running - expect(() => agent.abort()).not.toThrow(); - }); - it("continue() should process queued follow-up messages after an assistant turn", async () => { const agent = new Agent({ streamFn: () => { @@ -323,63 +226,6 @@ describe("Agent", () => { ]); }); - it("forwards sessionId and thinkingBudgets to streamFn options", async () => { - let receivedSessionId: string | undefined; - let receivedBudgets: ThinkingBudgets | undefined; - - const agent = new Agent({ - sessionId: "session-abc", - thinkingBudgets: { minimal: 64, low: 256 }, - streamFn: (_model, _context, options) => { - receivedSessionId = options?.sessionId; - receivedBudgets = options?.thinkingBudgets; - const stream = new MockAssistantStream(); - queueMicrotask(() => { - const message = createAssistantMessage([{ type: "text", text: "ok" }]); - stream.push({ type: "done", reason: "stop", message }); - }); - return stream; - }, - }); - - await agent.prompt("hello"); - expect(receivedSessionId).toBe("session-abc"); - expect(receivedBudgets).toEqual({ minimal: 64, low: 256 }); - - agent.sessionId = "session-def"; - agent.thinkingBudgets = { medium: 512 }; - - await agent.prompt("hello again"); - expect(receivedSessionId).toBe("session-def"); - expect(receivedBudgets).toEqual({ medium: 512 }); - }); - - it("forwards onPayload to streamFn options", async () => { - let receivedOnPayload: SimpleStreamOptions["onPayload"] | undefined; - - const agent = new Agent({ - onPayload: async (payload, model) => ({ payload, provider: model?.provider }), - streamFn: (_model, _context, options) => { - receivedOnPayload = options?.onPayload; - const stream = new MockAssistantStream(); - queueMicrotask(() => { - const message = createAssistantMessage([{ type: "text", text: "ok" }]); - stream.push({ type: "done", reason: "stop", message }); - }); - return stream; - }, - }); - - await agent.prompt("hello"); - expect(receivedOnPayload).toBeDefined(); - - const replacementPayload = await receivedOnPayload?.({ request: true }, getBundledModel("openai", "gpt-4o-mini")); - expect(replacementPayload).toEqual({ - payload: { request: true }, - provider: "openai", - }); - }); - it("re-reads thinking level for each model call within a run", async () => { const toolSchema = Type.Object({ value: Type.String() }); type Details = { value: string }; diff --git a/packages/ai/src/model-thinking.ts b/packages/ai/src/model-thinking.ts index 9c6f02fa1..510837693 100644 --- a/packages/ai/src/model-thinking.ts +++ b/packages/ai/src/model-thinking.ts @@ -319,7 +319,7 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void { (model.provider === "minimax-code" || model.provider === "minimax-code-cn") ) { model.compat = { - ...model.compat, + ...(model.compat ?? {}), supportsStore: false, supportsDeveloperRole: false, supportsReasoningEffort: false, @@ -327,6 +327,18 @@ function applyGeneratedModelPolicy(model: ApiModel<Api>): void { }; delete model.compat.thinkingFormat; } + if ( + model.api === "openai-completions" && + model.provider === "opencode-go" && + (model.id === "deepseek-v4-flash" || model.id === "deepseek-v4-pro") + ) { + model.compat = { + ...(model.compat ?? {}), + supportsToolChoice: false, + reasoningContentField: "reasoning_content", + requiresReasoningContentForToolCalls: true, + }; + } const parsedModel = parseKnownModel(model.id); const applyPatchToolType = inferGeneratedApplyPatchToolType(model, parsedModel); if (applyPatchToolType) { diff --git a/packages/ai/src/models.json b/packages/ai/src/models.json index 51d45d316..2d01b5740 100644 --- a/packages/ai/src/models.json +++ b/packages/ai/src/models.json @@ -403,6 +403,31 @@ "maxLevel": "xhigh" } }, + "au.anthropic.claude-haiku-4-5-20251001-v1:0": { + "id": "au.anthropic.claude-haiku-4-5-20251001-v1:0", + "name": "Claude Haiku 4.5 (AU)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1, + "output": 5, + "cacheRead": 0.1, + "cacheWrite": 1.25 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "au.anthropic.claude-opus-4-6-v1": { "id": "au.anthropic.claude-opus-4-6-v1", "name": "AU Anthropic Claude Opus 4.6", @@ -428,6 +453,31 @@ "maxLevel": "xhigh" } }, + "au.anthropic.claude-sonnet-4-5-20250929-v1:0": { + "id": "au.anthropic.claude-sonnet-4-5-20250929-v1:0", + "name": "Claude Sonnet 4.5 (AU)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "anthropic-budget-effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "au.anthropic.claude-sonnet-4-6": { "id": "au.anthropic.claude-sonnet-4-6", "name": "AU Anthropic Claude Sonnet 4.6", @@ -1163,6 +1213,81 @@ "contextWindow": 128000, "maxTokens": 4096 }, + "jp.anthropic.claude-opus-4-7": { + "id": "jp.anthropic.claude-opus-4-7", + "name": "Claude Opus 4.7 (JP)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 25, + "cacheRead": 0.5, + "cacheWrite": 6.25 + }, + "contextWindow": 1000000, + "maxTokens": 128000, + "thinking": { + "mode": "anthropic-adaptive", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, + "jp.anthropic.claude-sonnet-4-5-20250929-v1:0": { + "id": "jp.anthropic.claude-sonnet-4-5-20250929-v1:0", + "name": "Claude Sonnet 4.5 (JP)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 + }, + "contextWindow": 200000, + "maxTokens": 64000, + "thinking": { + "mode": "anthropic-budget-effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, + "jp.anthropic.claude-sonnet-4-6": { + "id": "jp.anthropic.claude-sonnet-4-6", + "name": "Claude Sonnet 4.6 (JP)", + "api": "bedrock-converse-stream", + "provider": "amazon-bedrock", + "baseUrl": "https://bedrock-runtime.us-east-1.amazonaws.com", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 3, + "output": 15, + "cacheRead": 0.3, + "cacheWrite": 3.75 + }, + "contextWindow": 1000000, + "maxTokens": 64000, + "thinking": { + "mode": "anthropic-budget-effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "meta.llama3-1-405b-instruct-v1:0": { "id": "meta.llama3-1-405b-instruct-v1:0", "name": "Llama 3.1 405B Instruct", @@ -4136,7 +4261,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro": { @@ -4161,7 +4290,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gpt-5.1-codex-max": { @@ -5341,7 +5474,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro-preview": { @@ -5374,7 +5511,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gpt-4.1": { @@ -5520,7 +5661,7 @@ }, "gpt-5.1-codex": { "id": "gpt-5.1-codex", - "name": "GPT-5.1-Codex", + "name": "GPT-5.1 Codex", "api": "openai-responses", "provider": "github-copilot", "baseUrl": "https://api.githubcopilot.com", @@ -5548,7 +5689,7 @@ }, "gpt-5.1-codex-max": { "id": "gpt-5.1-codex-max", - "name": "GPT-5.1-Codex-max", + "name": "GPT-5.1 Codex Max", "api": "openai-responses", "provider": "github-copilot", "baseUrl": "https://api.githubcopilot.com", @@ -5576,7 +5717,7 @@ }, "gpt-5.1-codex-mini": { "id": "gpt-5.1-codex-mini", - "name": "GPT-5.1-Codex-mini", + "name": "GPT-5.1 Codex mini", "api": "openai-responses", "provider": "github-copilot", "baseUrl": "https://api.githubcopilot.com", @@ -6606,6 +6747,35 @@ "thinking": { "mode": "google-level", "minLevel": "low", + "maxLevel": "high", + "levels": [ + "low", + "high" + ] + } + }, + "gemini-3.1-flash-lite": { + "id": "gemini-3.1-flash-lite", + "name": "Gemini 3.1 Flash Lite", + "api": "google-generative-ai", + "provider": "google", + "baseUrl": "https://generativelanguage.googleapis.com/v1beta", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 1.5, + "cacheRead": 0.025, + "cacheWrite": 1 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "google-level", + "minLevel": "minimal", "maxLevel": "high" } }, @@ -6656,7 +6826,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro-preview-customtools": { @@ -6681,7 +6855,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-flash-latest": { @@ -7202,7 +7380,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3-pro-low": { @@ -7227,7 +7409,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro-high": { @@ -7252,7 +7438,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro-low": { @@ -7277,7 +7467,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gpt-oss-120b-medium": { @@ -7423,7 +7617,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-flash-lite-preview": { @@ -7473,7 +7671,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } } }, @@ -7725,7 +7927,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro-preview": { @@ -7750,7 +7956,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro-preview-customtools": { @@ -7775,7 +7985,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } } }, @@ -9319,6 +9533,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "baidu/cobuddy:free": { + "id": "baidu/cobuddy:free", + "name": "Baidu Qianfan: CoBuddy (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "baidu/ernie-4.5-21b-a3b": { "id": "baidu/ernie-4.5-21b-a3b", "name": "Baidu: ERNIE 4.5 21B A3B", @@ -10254,7 +10487,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3-pro-preview": { @@ -10279,7 +10516,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3.1-flash-image-preview": { @@ -10301,6 +10542,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "google/gemini-3.1-flash-lite": { + "id": "google/gemini-3.1-flash-lite", + "name": "Google: Gemini 3.1 Flash Lite", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "google/gemini-3.1-flash-lite-preview": { "id": "google/gemini-3.1-flash-lite-preview", "name": "Gemini 3.1 Flash Lite Preview", @@ -10343,7 +10603,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3.1-pro-preview-customtools": { @@ -10682,6 +10946,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "inclusionai/ling-2.6-1t": { + "id": "inclusionai/ling-2.6-1t", + "name": "inclusionAI: Ling-2.6-1T", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "inclusionai/ling-2.6-1t:free": { "id": "inclusionai/ling-2.6-1t:free", "name": "inclusionAI: Ling-2.6-1T (free)", @@ -10739,6 +11022,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "inclusionai/ring-2.6-1t:free": { + "id": "inclusionai/ring-2.6-1t:free", + "name": "inclusionAI: Ring-2.6-1T (free)", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "inflection/inflection-3-pi": { "id": "inflection/inflection-3-pi", "name": "Inflection: Inflection 3 Pi", @@ -11366,6 +11668,31 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "microsoft/phi-4-mini-instruct": { + "id": "microsoft/phi-4-mini-instruct", + "name": "Phi-4-Mini", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 8192, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "microsoft/wizardlm-2-8x22b": { "id": "microsoft/wizardlm-2-8x22b", "name": "WizardLM-2 8x22B", @@ -11842,6 +12169,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "mistralai/mistral-medium-3-5": { + "id": "mistralai/mistral-medium-3-5", + "name": "Mistral: Mistral Medium 3.5", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "mistralai/mistral-medium-3.1": { "id": "mistralai/mistral-medium-3.1", "name": "Mistral: Mistral Medium 3.1", @@ -13647,6 +13993,25 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "openai/gpt-chat-latest": { + "id": "openai/gpt-chat-latest", + "name": "OpenAI: GPT Chat Latest", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 222222, + "maxTokens": 8888 + }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", @@ -15660,6 +16025,30 @@ "contextWindow": 222222, "maxTokens": 8888 }, + "tencent/hy3-preview": { + "id": "tencent/hy3-preview", + "name": "Hy3 preview", + "api": "openai-completions", + "provider": "kilo", + "baseUrl": "https://api.kilo.ai/api/gateway", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 256000, + "maxTokens": 64000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "tencent/hy3-preview:free": { "id": "tencent/hy3-preview:free", "name": "Tencent: Hy3 preview (free)", @@ -16103,7 +16492,7 @@ }, "x-ai/grok-code-fast-1:optimized:free": { "id": "x-ai/grok-code-fast-1:optimized:free", - "name": "xAI: Grok Code Fast 1 Optimized (free)", + "name": "xAI: Grok Code Fast 1, retiring May 15 (free)", "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", @@ -16136,8 +16525,8 @@ "cacheRead": 0, "cacheWrite": 0 }, - "contextWindow": 262000, - "maxTokens": 64000, + "contextWindow": 262144, + "maxTokens": 65536, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -16150,7 +16539,7 @@ "api": "openai-completions", "provider": "kilo", "baseUrl": "https://api.kilo.ai/api/gateway", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" @@ -16162,7 +16551,12 @@ "cacheWrite": 0 }, "contextWindow": 265000, - "maxTokens": 265000 + "maxTokens": 265000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "xiaomi/mimo-v2-omni:free": { "id": "xiaomi/mimo-v2-omni:free", @@ -16191,8 +16585,7 @@ "baseUrl": "https://api.kilo.ai/api/gateway", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 0, @@ -16235,7 +16628,8 @@ "baseUrl": "https://api.kilo.ai/api/gateway", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -16259,8 +16653,7 @@ "baseUrl": "https://api.kilo.ai/api/gateway", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 0, @@ -17315,7 +17708,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3-pro-preview": { @@ -17340,7 +17737,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "glm-4.5": { @@ -18509,7 +18910,8 @@ "compat": { "supportsDeveloperRole": false, "supportsReasoningEffort": false, - "reasoningContentField": "reasoning_content" + "reasoningContentField": "reasoning_content", + "supportsStore": false }, "contextWindow": 1000000, "maxTokens": 32000, @@ -18598,7 +19000,8 @@ "compat": { "supportsDeveloperRole": false, "supportsReasoningEffort": false, - "reasoningContentField": "reasoning_content" + "reasoningContentField": "reasoning_content", + "supportsStore": false }, "contextWindow": 204800, "maxTokens": 32000, @@ -18749,7 +19152,8 @@ "compat": { "supportsDeveloperRole": false, "supportsReasoningEffort": false, - "reasoningContentField": "reasoning_content" + "reasoningContentField": "reasoning_content", + "supportsStore": false }, "contextWindow": 1000000, "maxTokens": 32000, @@ -18838,7 +19242,8 @@ "compat": { "supportsDeveloperRole": false, "supportsReasoningEffort": false, - "reasoningContentField": "reasoning_content" + "reasoningContentField": "reasoning_content", + "supportsStore": false }, "contextWindow": 204800, "maxTokens": 32000, @@ -22943,7 +23348,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3-pro-preview-thinking": { @@ -23621,7 +24030,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3.1-pro-preview-customtools": { @@ -23646,7 +24059,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3.1-pro-preview-high": { @@ -31907,7 +32324,7 @@ "api": "openai-completions", "provider": "nanogpt", "baseUrl": "https://nano-gpt.com/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" @@ -31919,7 +32336,12 @@ "cacheWrite": 0 }, "contextWindow": 265000, - "maxTokens": 265000 + "maxTokens": 265000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "xiaomi/mimo-v2-pro": { "id": "xiaomi/mimo-v2-pro", @@ -31929,8 +32351,7 @@ "baseUrl": "https://nano-gpt.com/api/v1", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 0, @@ -31954,7 +32375,8 @@ "baseUrl": "https://nano-gpt.com/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0, @@ -31978,8 +32400,7 @@ "baseUrl": "https://nano-gpt.com/api/v1", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 0, @@ -35709,6 +36130,11 @@ "mode": "effort", "minLevel": "minimal", "maxLevel": "xhigh" + }, + "compat": { + "supportsToolChoice": false, + "reasoningContentField": "reasoning_content", + "requiresReasoningContentForToolCalls": true } }, "deepseek-v4-pro": { @@ -35733,6 +36159,11 @@ "mode": "effort", "minLevel": "minimal", "maxLevel": "xhigh" + }, + "compat": { + "supportsToolChoice": false, + "reasoningContentField": "reasoning_content", + "requiresReasoningContentForToolCalls": true } }, "glm-5": { @@ -35810,7 +36241,7 @@ }, "kimi-k2.6": { "id": "kimi-k2.6", - "name": "Kimi K2.6 (3x limits)", + "name": "Kimi K2.6", "api": "openai-completions", "provider": "opencode-go", "baseUrl": "https://opencode.ai/zen/go/v1", @@ -35820,9 +36251,9 @@ "image" ], "cost": { - "input": 0.32, - "output": 1.34, - "cacheRead": 0.054, + "input": 0.95, + "output": 4, + "cacheRead": 0.16, "cacheWrite": 0 }, "contextWindow": 262144, @@ -35835,7 +36266,7 @@ }, "mimo-v2-omni": { "id": "mimo-v2-omni", - "name": "MiMo V2 Omni", + "name": "MiMo-V2-Omni", "api": "openai-completions", "provider": "opencode-go", "baseUrl": "https://opencode.ai/zen/go/v1", @@ -35860,7 +36291,7 @@ }, "mimo-v2-pro": { "id": "mimo-v2-pro", - "name": "MiMo V2 Pro", + "name": "MiMo-V2-Pro", "api": "openai-completions", "provider": "opencode-go", "baseUrl": "https://opencode.ai/zen/go/v1", @@ -36034,9 +36465,9 @@ "big-pickle": { "id": "big-pickle", "name": "Big Pickle", - "api": "anthropic-messages", + "api": "openai-completions", "provider": "opencode-zen", - "baseUrl": "https://opencode.ai/zen", + "baseUrl": "https://opencode.ai/zen/v1", "reasoning": true, "input": [ "text" @@ -36050,7 +36481,7 @@ "contextWindow": 200000, "maxTokens": 128000, "thinking": { - "mode": "budget", + "mode": "effort", "minLevel": "minimal", "maxLevel": "xhigh" } @@ -36322,7 +36753,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemini-3.1-pro": { @@ -36347,7 +36782,11 @@ "thinking": { "mode": "google-level", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "glm-4.6": { @@ -36508,9 +36947,9 @@ "image" ], "cost": { - "input": 0, - "output": 0, - "cacheRead": 0, + "input": 0.05, + "output": 0.4, + "cacheRead": 0.005, "cacheWrite": 0 }, "contextWindow": 400000, @@ -37276,6 +37715,30 @@ "maxLevel": "high" } }, + "ring-2.6-1t-free": { + "id": "ring-2.6-1t-free", + "name": "Ring 2.6 1T Free", + "api": "openai-completions", + "provider": "opencode-zen", + "baseUrl": "https://opencode.ai/zen/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262000, + "maxTokens": 66000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } + }, "trinity-large-preview-free": { "id": "trinity-large-preview-free", "name": "Trinity Large Preview", @@ -37434,13 +37897,13 @@ "image" ], "cost": { - "input": 0.74, - "output": 3.49, - "cacheRead": 0.14, + "input": 0.75, + "output": 3.5, + "cacheRead": 0.15, "cacheWrite": 0 }, - "contextWindow": 262142, - "maxTokens": 262142, + "contextWindow": 262144, + "maxTokens": 16384, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -38190,6 +38653,33 @@ "maxLevel": "high" } }, + "baidu/cobuddy:free": { + "id": "baidu/cobuddy:free", + "name": "Baidu Qianfan: CoBuddy (free)", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 131072, + "maxTokens": 65536, + "compat": { + "supportsToolChoice": false + }, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "baidu/ernie-4.5-21b-a3b": { "id": "baidu/ernie-4.5-21b-a3b", "name": "Baidu: ERNIE 4.5 21B A3B", @@ -38623,8 +39113,8 @@ "cacheRead": 0.003625, "cacheWrite": 0 }, - "contextWindow": 131000, - "maxTokens": 131000, + "contextWindow": 1048576, + "maxTokens": 384000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -38667,7 +39157,7 @@ "cacheRead": 0.024999999999999998, "cacheWrite": 0.08333333333333334 }, - "contextWindow": 1048576, + "contextWindow": 1000000, "maxTokens": 8192 }, "google/gemini-2.0-flash-lite-001": { @@ -38907,6 +39397,35 @@ "thinking": { "mode": "effort", "minLevel": "low", + "maxLevel": "high", + "levels": [ + "low", + "high" + ] + } + }, + "google/gemini-3.1-flash-lite": { + "id": "google/gemini-3.1-flash-lite", + "name": "Google: Gemini 3.1 Flash Lite", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 1.5, + "cacheRead": 0.024999999999999998, + "cacheWrite": 0.08333333333333334 + }, + "contextWindow": 1048576, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", "maxLevel": "high" } }, @@ -38952,7 +39471,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3.1-pro-preview-customtools": { @@ -38977,7 +39500,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemma-3-12b-it": { @@ -39225,6 +39752,25 @@ "contextWindow": 128000, "maxTokens": 32000 }, + "inclusionai/ling-2.6-1t": { + "id": "inclusionai/ling-2.6-1t", + "name": "inclusionAI: Ling-2.6-1T", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": false, + "input": [ + "text" + ], + "cost": { + "input": 0.3, + "output": 2.5, + "cacheRead": 0.06, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 32768 + }, "inclusionai/ling-2.6-1t:free": { "id": "inclusionai/ling-2.6-1t:free", "name": "inclusionAI: Ling-2.6-1T (free)", @@ -39282,6 +39828,30 @@ "contextWindow": 262144, "maxTokens": 32768 }, + "inclusionai/ring-2.6-1t:free": { + "id": "inclusionai/ring-2.6-1t:free", + "name": "inclusionAI: Ring-2.6-1T (free)", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0, + "output": 0, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "kwaipilot/kat-coder-pro": { "id": "kwaipilot/kat-coder-pro", "name": "Kwaipilot: KAT-Coder-Pro V1", @@ -39582,7 +40152,7 @@ "cacheWrite": 0 }, "contextWindow": 196608, - "maxTokens": 131072, + "maxTokens": 196608, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -39627,13 +40197,13 @@ "text" ], "cost": { - "input": 0.3, + "input": 0.29900000000000004, "output": 1.2, "cacheRead": 0.059, "cacheWrite": 0 }, "contextWindow": 196608, - "maxTokens": 131070, + "maxTokens": 131072, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -39873,6 +40443,31 @@ "contextWindow": 131072, "maxTokens": 8888 }, + "mistralai/mistral-medium-3-5": { + "id": "mistralai/mistral-medium-3-5", + "name": "Mistral: Mistral Medium 3.5", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 1.5, + "output": 7.5, + "cacheRead": 0, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 8888, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "mistralai/mistral-medium-3.1": { "id": "mistralai/mistral-medium-3.1", "name": "Mistral: Mistral Medium 3.1", @@ -40224,13 +40819,13 @@ "image" ], "cost": { - "input": 0.44, - "output": 2, - "cacheRead": 0.22, + "input": 0.39999999999999997, + "output": 1.9800000000000002, + "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65535, + "maxTokens": 262144, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -40249,13 +40844,13 @@ "image" ], "cost": { - "input": 0.74, - "output": 3.49, - "cacheRead": 0.14, + "input": 0.75, + "output": 3.5, + "cacheRead": 0.15, "cacheWrite": 0 }, - "contextWindow": 262142, - "maxTokens": 262142, + "contextWindow": 262144, + "maxTokens": 16384, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -41548,6 +42143,26 @@ "contextWindow": 128000, "maxTokens": 16384 }, + "openai/gpt-chat-latest": { + "id": "openai/gpt-chat-latest", + "name": "OpenAI: GPT Chat Latest", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": false, + "input": [ + "text", + "image" + ], + "cost": { + "input": 5, + "output": 30, + "cacheRead": 0.5, + "cacheWrite": 0 + }, + "contextWindow": 400000, + "maxTokens": 128000 + }, "openai/gpt-oss-120b": { "id": "openai/gpt-oss-120b", "name": "GPT OSS 120B", @@ -42484,12 +43099,12 @@ ], "cost": { "input": 0.08, - "output": 0.24, + "output": 0.28, "cacheRead": 0.04, "cacheWrite": 0 }, "contextWindow": 40960, - "maxTokens": 40960, + "maxTokens": 16384, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -42636,7 +43251,7 @@ "text" ], "cost": { - "input": 0.12, + "input": 0.11, "output": 0.7999999999999999, "cacheRead": 0.07, "cacheWrite": 0 @@ -43028,13 +43643,13 @@ "image" ], "cost": { - "input": 0.15, + "input": 0.14, "output": 1, "cacheRead": 0.049999999999999996, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 262144, + "maxTokens": 81920, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -43078,13 +43693,13 @@ "image" ], "cost": { - "input": 0.09999999999999999, + "input": 0.04, "output": 0.15, "cacheRead": 0, "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 8888, + "maxTokens": 81920, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -43508,6 +44123,30 @@ "maxLevel": "high" } }, + "tencent/hy3-preview": { + "id": "tencent/hy3-preview", + "name": "Hy3 preview", + "api": "openai-completions", + "provider": "openrouter", + "baseUrl": "https://openrouter.ai/api/v1", + "reasoning": true, + "input": [ + "text" + ], + "cost": { + "input": 0.06599999999999999, + "output": 0.26, + "cacheRead": 0.029, + "cacheWrite": 0 + }, + "contextWindow": 262144, + "maxTokens": 262144, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } + }, "tencent/hy3-preview:free": { "id": "tencent/hy3-preview:free", "name": "Tencent: Hy3 preview (free)", @@ -43937,9 +44576,9 @@ "text" ], "cost": { - "input": 0.09, - "output": 0.29, - "cacheRead": 0.045, + "input": 0.09999999999999999, + "output": 0.3, + "cacheRead": 0.01, "cacheWrite": 0 }, "contextWindow": 262144, @@ -43956,7 +44595,7 @@ "api": "openai-completions", "provider": "openrouter", "baseUrl": "https://openrouter.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" @@ -43968,7 +44607,12 @@ "cacheWrite": 0 }, "contextWindow": 262144, - "maxTokens": 65536 + "maxTokens": 65536, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "high" + } }, "xiaomi/mimo-v2-pro": { "id": "xiaomi/mimo-v2-pro", @@ -43978,8 +44622,7 @@ "baseUrl": "https://openrouter.ai/api/v1", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 1, @@ -44003,7 +44646,8 @@ "baseUrl": "https://openrouter.ai/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.39999999999999997, @@ -44027,8 +44671,7 @@ "baseUrl": "https://openrouter.ai/api/v1", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 1, @@ -44037,7 +44680,7 @@ "cacheWrite": 0 }, "contextWindow": 1048576, - "maxTokens": 131072, + "maxTokens": 16384, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -44244,13 +44887,13 @@ "text" ], "cost": { - "input": 0.38, - "output": 1.74, - "cacheRead": 0.195, + "input": 0.39999999999999997, + "output": 1.75, + "cacheRead": 0.08, "cacheWrite": 0 }, "contextWindow": 202752, - "maxTokens": 64000, + "maxTokens": 131072, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -44293,12 +44936,12 @@ ], "cost": { "input": 0.6, - "output": 2.08, + "output": 1.92, "cacheRead": 0.12, "cacheWrite": 0 }, "contextWindow": 202752, - "maxTokens": 16384, + "maxTokens": 128000, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -45714,7 +46357,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "gemma-4-uncensored": { @@ -48384,6 +49031,35 @@ "thinking": { "mode": "budget", "minLevel": "low", + "maxLevel": "high", + "levels": [ + "low", + "high" + ] + } + }, + "google/gemini-3.1-flash-lite": { + "id": "google/gemini-3.1-flash-lite", + "name": "Gemini 3.1 Flash Lite", + "api": "anthropic-messages", + "provider": "vercel-ai-gateway", + "baseUrl": "https://ai-gateway.vercel.sh", + "reasoning": true, + "input": [ + "text", + "image" + ], + "cost": { + "input": 0.25, + "output": 1.5, + "cacheRead": 0.03, + "cacheWrite": 0 + }, + "contextWindow": 1000000, + "maxTokens": 65000, + "thinking": { + "mode": "budget", + "minLevel": "minimal", "maxLevel": "high" } }, @@ -48429,7 +49105,11 @@ "thinking": { "mode": "budget", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemma-4-26b-a4b-it": { @@ -50540,8 +51220,8 @@ "image" ], "cost": { - "input": 2, - "output": 6, + "input": 1.25, + "output": 2.5, "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, @@ -50565,8 +51245,8 @@ "image" ], "cost": { - "input": 2, - "output": 6, + "input": 1.25, + "output": 2.5, "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, @@ -50590,8 +51270,8 @@ "image" ], "cost": { - "input": 2, - "output": 6, + "input": 1.25, + "output": 2.5, "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, @@ -50610,8 +51290,8 @@ "image" ], "cost": { - "input": 2, - "output": 6, + "input": 1.25, + "output": 2.5, "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, @@ -50630,8 +51310,8 @@ "image" ], "cost": { - "input": 2, - "output": 6, + "input": 1.25, + "output": 2.5, "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, @@ -50655,8 +51335,8 @@ "image" ], "cost": { - "input": 2, - "output": 6, + "input": 1.25, + "output": 2.5, "cacheRead": 0.19999999999999998, "cacheWrite": 0 }, @@ -50749,8 +51429,7 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 1, @@ -50774,7 +51453,8 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.39999999999999997, @@ -50798,8 +51478,7 @@ "baseUrl": "https://ai-gateway.vercel.sh", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 1, @@ -51757,8 +52436,8 @@ "cacheRead": 0.01, "cacheWrite": 0 }, - "contextWindow": 256000, - "maxTokens": 64000, + "contextWindow": 262144, + "maxTokens": 65536, "thinking": { "mode": "budget", "minLevel": "minimal", @@ -51782,8 +52461,8 @@ "cacheRead": 0.08, "cacheWrite": 0 }, - "contextWindow": 256000, - "maxTokens": 128000, + "contextWindow": 262144, + "maxTokens": 131072, "thinking": { "mode": "budget", "minLevel": "minimal", @@ -51806,8 +52485,8 @@ "cacheRead": 0.2, "cacheWrite": 0 }, - "contextWindow": 1000000, - "maxTokens": 128000, + "contextWindow": 1048576, + "maxTokens": 131072, "thinking": { "mode": "budget", "minLevel": "minimal", @@ -51822,7 +52501,8 @@ "baseUrl": "https://api.xiaomimimo.com/anthropic", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.4, @@ -51846,8 +52526,7 @@ "baseUrl": "https://api.xiaomimimo.com/anthropic", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 1, @@ -53053,7 +53732,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3-pro-preview": { @@ -53078,7 +53761,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemini-3.1-flash-lite-preview": { @@ -53123,7 +53810,11 @@ "thinking": { "mode": "effort", "minLevel": "low", - "maxLevel": "high" + "maxLevel": "high", + "levels": [ + "low", + "high" + ] } }, "google/gemma-3-12b-it": { @@ -55301,8 +55992,8 @@ "cacheRead": 0.01, "cacheWrite": 0 }, - "contextWindow": 262000, - "maxTokens": 64000, + "contextWindow": 262144, + "maxTokens": 65536, "thinking": { "mode": "effort", "minLevel": "minimal", @@ -55339,7 +56030,7 @@ "api": "openai-completions", "provider": "zenmux", "baseUrl": "https://zenmux.ai/api/v1", - "reasoning": false, + "reasoning": true, "input": [ "text", "image" @@ -55347,11 +56038,16 @@ "cost": { "input": 0.4, "output": 2, - "cacheRead": 0, + "cacheRead": 0.08, "cacheWrite": 0 }, "contextWindow": 265000, - "maxTokens": 265000 + "maxTokens": 265000, + "thinking": { + "mode": "effort", + "minLevel": "minimal", + "maxLevel": "xhigh" + } }, "xiaomi/mimo-v2-pro": { "id": "xiaomi/mimo-v2-pro", @@ -55361,13 +56057,12 @@ "baseUrl": "https://zenmux.ai/api/v1", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { - "input": 1.5, - "output": 4.5, - "cacheRead": 0, + "input": 1, + "output": 3, + "cacheRead": 0.2, "cacheWrite": 0 }, "contextWindow": 1000000, @@ -55386,7 +56081,8 @@ "baseUrl": "https://zenmux.ai/api/v1", "reasoning": true, "input": [ - "text" + "text", + "image" ], "cost": { "input": 0.4, @@ -55410,8 +56106,7 @@ "baseUrl": "https://zenmux.ai/api/v1", "reasoning": true, "input": [ - "text", - "image" + "text" ], "cost": { "input": 1, diff --git a/packages/ai/test/alibaba-coding-plan-provider.test.ts b/packages/ai/test/alibaba-coding-plan-provider.test.ts deleted file mode 100644 index 50a5b6cfe..000000000 --- a/packages/ai/test/alibaba-coding-plan-provider.test.ts +++ /dev/null @@ -1,34 +0,0 @@ -import { afterEach, describe, expect, test } from "bun:test"; -import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "../src/provider-models/descriptors"; -import { alibabaCodingPlanModelManagerOptions } from "../src/provider-models/openai-compat"; -import { getEnvApiKey } from "../src/stream"; - -const originalAlibabaApiKey = Bun.env.ALIBABA_CODING_PLAN_API_KEY; - -afterEach(() => { - if (originalAlibabaApiKey === undefined) { - delete Bun.env.ALIBABA_CODING_PLAN_API_KEY; - return; - } - Bun.env.ALIBABA_CODING_PLAN_API_KEY = originalAlibabaApiKey; -}); - -describe("alibaba-coding-plan provider support", () => { - test("resolves ALIBABA_CODING_PLAN_API_KEY from environment", () => { - Bun.env.ALIBABA_CODING_PLAN_API_KEY = "alibaba-test-key"; - expect(getEnvApiKey("alibaba-coding-plan")).toBe("alibaba-test-key"); - }); - - test("registers built-in descriptor and default model", () => { - const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "alibaba-coding-plan"); - expect(descriptor).toBeDefined(); - expect(descriptor?.defaultModel).toBe("qwen3.5-plus"); - expect(DEFAULT_MODEL_PER_PROVIDER["alibaba-coding-plan"]).toBe("qwen3.5-plus"); - }); - - test("builds model manager options with alibaba-coding-plan defaults", () => { - const options = alibabaCodingPlanModelManagerOptions(); - expect(options.providerId).toBe("alibaba-coding-plan"); - expect(options.fetchDynamicModels).toBeDefined(); - }); -}); diff --git a/packages/ai/test/anthropic-alignment.test.ts b/packages/ai/test/anthropic-alignment.test.ts index 252f9dfb0..60b51e6f1 100644 --- a/packages/ai/test/anthropic-alignment.test.ts +++ b/packages/ai/test/anthropic-alignment.test.ts @@ -9,8 +9,6 @@ import { buildAnthropicClientOptions, buildAnthropicHeaders, buildAnthropicSystemBlocks, - claudeCodeHeaders, - claudeCodeSystemInstruction, claudeCodeVersion, generateClaudeCloakingUserId, isClaudeCloakingUserId, @@ -83,21 +81,6 @@ function captureAnthropicPayload( } describe("Anthropic request fingerprint alignment", () => { - it("uses updated Claude Code header defaults", () => { - const headers = buildAnthropicHeaders({ - apiKey: "sk-ant-oat-test", - isOAuth: true, - stream: true, - }); - - expect(headers["Anthropic-Beta"]).toContain("context-management-2025-06-27"); - expect(headers["Anthropic-Beta"]).toContain("prompt-caching-scope-2026-01-05"); - expect(headers["Anthropic-Beta"]).not.toContain("fine-grained-tool-streaming-2025-05-14"); - expect(headers["User-Agent"]).toBe(`claude-cli/${claudeCodeVersion} (external, cli)`); - expect(claudeCodeHeaders["X-Stainless-Package-Version"]).toBe("0.74.0"); - expect("X-Stainless-Helper-Method" in claudeCodeHeaders).toBe(false); - }); - it("maps Stainless OS and arch values from explicit inputs", () => { expect(mapStainlessOs("darwin")).toBe("MacOS"); expect(mapStainlessOs("windows")).toBe("Windows"); @@ -124,29 +107,6 @@ describe("Anthropic request fingerprint alignment", () => { expect(headers["X-Stainless-Arch"]).toBe(mapStainlessArch(process.arch)); }); - it("injects billing header and Claude Agent SDK identity block", () => { - const blocks = buildAnthropicSystemBlocks(["Stay concise."], { - includeClaudeCodeInstruction: true, - extraInstructions: ["Use citations when possible"], - }); - - expect(blocks).toBeDefined(); - expect(blocks?.[0]?.text.startsWith(`x-anthropic-billing-header: cc_version=${claudeCodeVersion}.`)).toBe(true); - expect(blocks?.[0]?.text).toMatch(/cc_entrypoint=cli; cch=[0-9a-f]{5};$/); - expect(blocks?.[1]).toEqual({ - type: "text", - text: claudeCodeSystemInstruction, - }); - expect(blocks?.[2]).toEqual({ - type: "text", - text: "Use citations when possible", - }); - expect(blocks?.[3]).toEqual({ - type: "text", - text: "Stay concise.", - }); - }); - it("attaches cache_control only to the last emitted system block when cacheControl is set", () => { const blocks = buildAnthropicSystemBlocks(["Stay concise."], { includeClaudeCodeInstruction: true, @@ -607,21 +567,6 @@ describe("Anthropic request fingerprint alignment", () => { expect(strictNames).toEqual(["python"]); }); - it("drops fine-grained tool-streaming beta from default Anthropic client options", () => { - const options = buildAnthropicClientOptions({ - model: ANTHROPIC_MODEL, - apiKey: "sk-ant-oat-test", - extraBetas: [], - stream: true, - interleavedThinking: false, - dynamicHeaders: {}, - }); - - const beta = options.defaultHeaders["Anthropic-Beta"]; - expect(beta).toContain("context-management-2025-06-27"); - expect(beta).not.toContain("fine-grained-tool-streaming-2025-05-14"); - }); - it("adds legacy fine-grained tool-streaming beta only for tool requests on incompatible models", () => { const incompatibleModel: Model<"anthropic-messages"> = { ...ANTHROPIC_MODEL, diff --git a/packages/ai/test/api-registry.test.ts b/packages/ai/test/api-registry.test.ts index 31306df97..f62840076 100644 --- a/packages/ai/test/api-registry.test.ts +++ b/packages/ai/test/api-registry.test.ts @@ -14,14 +14,6 @@ afterEach(() => { describe("custom API registry", () => { const streamSimple: CustomStreamSimpleFn = () => ({}) as unknown as AssistantMessageEventStream; - test("registers and resolves a custom API provider", () => { - registerCustomApi("custom-provider", streamSimple, "ext-a"); - - const provider = getCustomApi("custom-provider"); - expect(provider).toBeDefined(); - expect(provider?.streamSimple).toBe(streamSimple); - expect(provider?.sourceId).toBe("ext-a"); - }); test("rejects registrations that collide with built-in API names", () => { expect(() => registerCustomApi("openai-responses", streamSimple)).toThrow( diff --git a/packages/ai/test/apply-patch-freeform.test.ts b/packages/ai/test/apply-patch-freeform.test.ts index ca37a473c..3665712e4 100644 --- a/packages/ai/test/apply-patch-freeform.test.ts +++ b/packages/ai/test/apply-patch-freeform.test.ts @@ -107,11 +107,6 @@ function makeUnionTool(strict: boolean): Tool { } describe("supportsFreeformApplyPatch", () => { - test("absent flag returns false", () => { - // No runtime auto-detection — requires generated model metadata. - expect(supportsFreeformApplyPatch(makeModel())).toBe(false); - }); - test("applyPatchToolType: freeform enables", () => { expect(supportsFreeformApplyPatch(makeModel({ applyPatchToolType: "freeform" }))).toBe(true); }); diff --git a/packages/ai/test/auth-storage-credential-disabled-event.test.ts b/packages/ai/test/auth-storage-credential-disabled-event.test.ts index 57f42ca8c..2534198db 100644 --- a/packages/ai/test/auth-storage-credential-disabled-event.test.ts +++ b/packages/ai/test/auth-storage-credential-disabled-event.test.ts @@ -4,6 +4,12 @@ import * as os from "node:os"; import * as path from "node:path"; import { AuthCredentialStore, AuthStorage, type CredentialDisabledEvent } from "../src/auth-storage"; import * as oauthUtils from "../src/utils/oauth"; +import { withEnv } from "./helpers"; + +const SUPPRESS_ANTHROPIC_ENV = { + ANTHROPIC_API_KEY: undefined, + ANTHROPIC_OAUTH_TOKEN: undefined, +} as const; describe("AuthStorage onCredentialDisabled callback", () => { let tempDir = ""; @@ -51,12 +57,14 @@ describe("AuthStorage onCredentialDisabled callback", () => { ); }); - const apiKey = await authStorage.getApiKey("anthropic", "session-disabled-event"); + await withEnv(SUPPRESS_ANTHROPIC_ENV, async () => { + const apiKey = await authStorage!.getApiKey("anthropic", "session-disabled-event"); - expect(apiKey).toBeUndefined(); - expect(events).toHaveLength(1); - expect(events[0]?.provider).toBe("anthropic"); - expect(events[0]?.disabledCause).toContain("invalid_grant"); + expect(apiKey).toBeUndefined(); + expect(events).toHaveLength(1); + expect(events[0]?.provider).toBe("anthropic"); + expect(events[0]?.disabledCause).toContain("invalid_grant"); + }); }); test("does not fire for transient (non-definitive) refresh failures", async () => { @@ -75,9 +83,10 @@ describe("AuthStorage onCredentialDisabled callback", () => { throw new Error("fetch failed: ECONNRESET"); }); - await authStorage.getApiKey("anthropic", "session-transient-failure"); - - expect(events).toHaveLength(0); + await withEnv(SUPPRESS_ANTHROPIC_ENV, async () => { + await authStorage!.getApiKey("anthropic", "session-transient-failure"); + expect(events).toHaveLength(0); + }); }); test("swallows handler exceptions so disable still completes", async () => { @@ -104,8 +113,10 @@ describe("AuthStorage onCredentialDisabled callback", () => { throw new Error("invalid_grant"); }); - await expect(authStorage.getApiKey("anthropic", "session-handler-throws")).resolves.toBeUndefined(); - expect(authStorage.list()).not.toContain("anthropic"); + await withEnv(SUPPRESS_ANTHROPIC_ENV, async () => { + await expect(authStorage!.getApiKey("anthropic", "session-handler-throws")).resolves.toBeUndefined(); + expect(authStorage!.list()).not.toContain("anthropic"); + }); }); test("swallows async handler rejections so the disable path still completes", async () => { @@ -143,12 +154,14 @@ describe("AuthStorage onCredentialDisabled callback", () => { }; process.on("unhandledRejection", onUnhandled); try { - await expect(authStorage.getApiKey("anthropic", "session-async-handler-throws")).resolves.toBeUndefined(); - // Wait for the handler's microtask + our internal .catch to run. - await settled.promise; - await Bun.sleep(0); - expect(authStorage.list()).not.toContain("anthropic"); - expect(unhandled).toHaveLength(0); + await withEnv(SUPPRESS_ANTHROPIC_ENV, async () => { + await expect(authStorage!.getApiKey("anthropic", "session-async-handler-throws")).resolves.toBeUndefined(); + // Wait for the handler's microtask + our internal .catch to run. + await settled.promise; + await Bun.sleep(0); + expect(authStorage!.list()).not.toContain("anthropic"); + expect(unhandled).toHaveLength(0); + }); } finally { process.off("unhandledRejection", onUnhandled); } diff --git a/packages/ai/test/empty.test.ts b/packages/ai/test/empty.test.ts deleted file mode 100644 index 0c29787c3..000000000 --- a/packages/ai/test/empty.test.ts +++ /dev/null @@ -1,763 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { getBundledModel } from "@oh-my-pi/pi-ai/models"; -import { complete } from "@oh-my-pi/pi-ai/stream"; -import type { Api, AssistantMessage, Context, Model, OptionsForApi, UserMessage } from "@oh-my-pi/pi-ai/types"; -import { e2eApiKey, resolveApiKey } from "./oauth"; - -// Resolve OAuth tokens at module level (async, runs before tests) -const oauthTokens = await Promise.all([ - resolveApiKey("anthropic"), - resolveApiKey("github-copilot"), - resolveApiKey("google-gemini-cli"), - resolveApiKey("google-antigravity"), - resolveApiKey("openai-codex"), -]); -const [anthropicOAuthToken, githubCopilotToken, geminiCliToken, antigravityToken, openaiCodexToken] = oauthTokens; - -async function testEmptyMessage<TApi extends Api>(llm: Model<TApi>, options: OptionsForApi<TApi> = {}) { - // Test with completely empty content array - const emptyMessage: UserMessage = { - role: "user", - content: [], - timestamp: Date.now(), - }; - - const context: Context = { - messages: [emptyMessage], - }; - - const response = await complete(llm, context, options); - - // Should either handle gracefully or return an error - expect(response).toBeDefined(); - expect(response.role).toBe("assistant"); - // Should handle empty string gracefully - if (response.stopReason === "error") { - expect(response.errorMessage).toBeDefined(); - } else { - expect(response.content).toBeDefined(); - } -} - -async function testEmptyStringMessage<TApi extends Api>(llm: Model<TApi>, options: OptionsForApi<TApi> = {}) { - // Test with empty string content - const context: Context = { - messages: [ - { - role: "user", - content: "", - timestamp: Date.now(), - }, - ], - }; - - const response = await complete(llm, context, options); - - expect(response).toBeDefined(); - expect(response.role).toBe("assistant"); - - // Should handle empty string gracefully - if (response.stopReason === "error") { - expect(response.errorMessage).toBeDefined(); - } else { - expect(response.content).toBeDefined(); - } -} - -async function testWhitespaceOnlyMessage<TApi extends Api>(llm: Model<TApi>, options: OptionsForApi<TApi> = {}) { - // Test with whitespace-only content - const context: Context = { - messages: [ - { - role: "user", - content: " \n\t ", - timestamp: Date.now(), - }, - ], - }; - - const response = await complete(llm, context, options); - - expect(response).toBeDefined(); - expect(response.role).toBe("assistant"); - - // Should handle whitespace-only gracefully - if (response.stopReason === "error") { - expect(response.errorMessage).toBeDefined(); - } else { - expect(response.content).toBeDefined(); - } -} - -async function testEmptyAssistantMessage<TApi extends Api>(llm: Model<TApi>, options: OptionsForApi<TApi> = {}) { - // Test with empty assistant message in conversation flow - // User -> Empty Assistant -> User - const emptyAssistant: AssistantMessage = { - role: "assistant", - content: [], - api: llm.api, - provider: llm.provider, - model: llm.id, - usage: { - input: 10, - output: 0, - cacheRead: 0, - cacheWrite: 0, - totalTokens: 10, - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, - }, - stopReason: "stop", - timestamp: Date.now(), - }; - - const context: Context = { - messages: [ - { - role: "user", - content: "Hello, how are you?", - timestamp: Date.now(), - }, - emptyAssistant, - { - role: "user", - content: "Please respond this time.", - timestamp: Date.now(), - }, - ], - }; - - const response = await complete(llm, context, options); - - expect(response).toBeDefined(); - expect(response.role).toBe("assistant"); - - // Should handle empty assistant message in context gracefully - if (response.stopReason === "error") { - expect(response.errorMessage).toBeDefined(); - } else { - expect(response.content).toBeDefined(); - expect(response.content.length).toBeGreaterThan(0); - } -} - -describe("AI Providers Empty Message Tests", () => { - describe.skipIf(!e2eApiKey("GEMINI_API_KEY"))("Google Provider Empty Messages", () => { - const llm = getBundledModel("google", "gemini-2.5-flash"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("OPENAI_API_KEY"))("OpenAI Completions Provider Empty Messages", () => { - const llm = getBundledModel("openai", "gpt-4o-mini"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("OPENAI_API_KEY"))("OpenAI Responses Provider Empty Messages", () => { - const llm = getBundledModel("openai", "gpt-5-mini"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("ANTHROPIC_API_KEY"))("Anthropic Provider Empty Messages", () => { - const llm = getBundledModel("anthropic", "claude-haiku-4-5-20251001"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("XAI_API_KEY"))("xAI Provider Empty Messages", () => { - const llm = getBundledModel("xai", "grok-3"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("GROQ_API_KEY"))("Groq Provider Empty Messages", () => { - const llm = getBundledModel("groq", "openai/gpt-oss-20b"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("CEREBRAS_API_KEY"))("Cerebras Provider Empty Messages", () => { - const llm = getBundledModel("cerebras", "gpt-oss-120b"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("ZAI_API_KEY"))("zAI Provider Empty Messages", () => { - const llm = getBundledModel("zai", "glm-4.5-air"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe.skipIf(!e2eApiKey("MISTRAL_API_KEY"))("Mistral Provider Empty Messages", () => { - const llm = getBundledModel("mistral", "devstral-medium-latest"); - - it( - "should handle empty content array", - async () => { - await testEmptyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - - it( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe("Anthropic OAuth Provider Empty Messages", () => { - const llm = getBundledModel("anthropic", "claude-haiku-4-5-20251001"); - - it.skipIf(!anthropicOAuthToken)( - "should handle empty content array", - async () => { - await testEmptyMessage(llm, { apiKey: anthropicOAuthToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!anthropicOAuthToken)( - "should handle empty string content", - async () => { - await testEmptyStringMessage(llm, { apiKey: anthropicOAuthToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!anthropicOAuthToken)( - "should handle whitespace-only content", - async () => { - await testWhitespaceOnlyMessage(llm, { apiKey: anthropicOAuthToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!anthropicOAuthToken)( - "should handle empty assistant message in conversation", - async () => { - await testEmptyAssistantMessage(llm, { apiKey: anthropicOAuthToken }); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe("GitHub Copilot Provider Empty Messages", () => { - it.skipIf(!githubCopilotToken)( - "gpt-4o - should handle empty content array", - async () => { - const llm = getBundledModel("github-copilot", "gpt-4o"); - await testEmptyMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!githubCopilotToken)( - "gpt-4o - should handle empty string content", - async () => { - const llm = getBundledModel("github-copilot", "gpt-4o"); - await testEmptyStringMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!githubCopilotToken)( - "gpt-4o - should handle whitespace-only content", - async () => { - const llm = getBundledModel("github-copilot", "gpt-4o"); - await testWhitespaceOnlyMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!githubCopilotToken)( - "gpt-4o - should handle empty assistant message in conversation", - async () => { - const llm = getBundledModel("github-copilot", "gpt-4o"); - await testEmptyAssistantMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!githubCopilotToken)( - "claude-sonnet-4 - should handle empty content array", - async () => { - const llm = getBundledModel("github-copilot", "claude-sonnet-4"); - await testEmptyMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!githubCopilotToken)( - "claude-sonnet-4 - should handle empty string content", - async () => { - const llm = getBundledModel("github-copilot", "claude-sonnet-4"); - await testEmptyStringMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!githubCopilotToken)( - "claude-sonnet-4 - should handle whitespace-only content", - async () => { - const llm = getBundledModel("github-copilot", "claude-sonnet-4"); - await testWhitespaceOnlyMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!githubCopilotToken)( - "claude-sonnet-4 - should handle empty assistant message in conversation", - async () => { - const llm = getBundledModel("github-copilot", "claude-sonnet-4"); - await testEmptyAssistantMessage(llm, { apiKey: githubCopilotToken }); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe("Google Gemini CLI Provider Empty Messages", () => { - it.skipIf(!geminiCliToken)( - "gemini-2.5-flash - should handle empty content array", - async () => { - const llm = getBundledModel("google-gemini-cli", "gemini-2.5-flash"); - await testEmptyMessage(llm, { apiKey: geminiCliToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!geminiCliToken)( - "gemini-2.5-flash - should handle empty string content", - async () => { - const llm = getBundledModel("google-gemini-cli", "gemini-2.5-flash"); - await testEmptyStringMessage(llm, { apiKey: geminiCliToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!geminiCliToken)( - "gemini-2.5-flash - should handle whitespace-only content", - async () => { - const llm = getBundledModel("google-gemini-cli", "gemini-2.5-flash"); - await testWhitespaceOnlyMessage(llm, { apiKey: geminiCliToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!geminiCliToken)( - "gemini-2.5-flash - should handle empty assistant message in conversation", - async () => { - const llm = getBundledModel("google-gemini-cli", "gemini-2.5-flash"); - await testEmptyAssistantMessage(llm, { apiKey: geminiCliToken }); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe("Google Antigravity Provider Empty Messages", () => { - it.skipIf(!antigravityToken)( - "gemini-3-flash - should handle empty content array", - async () => { - const llm = getBundledModel("google-antigravity", "gemini-3-flash"); - await testEmptyMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "gemini-3-flash - should handle empty string content", - async () => { - const llm = getBundledModel("google-antigravity", "gemini-3-flash"); - await testEmptyStringMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "gemini-3-flash - should handle whitespace-only content", - async () => { - const llm = getBundledModel("google-antigravity", "gemini-3-flash"); - await testWhitespaceOnlyMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "gemini-3-flash - should handle empty assistant message in conversation", - async () => { - const llm = getBundledModel("google-antigravity", "gemini-3-flash"); - await testEmptyAssistantMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "claude-sonnet-4-5 - should handle empty content array", - async () => { - const llm = getBundledModel("google-antigravity", "claude-sonnet-4-5"); - await testEmptyMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "claude-sonnet-4-5 - should handle empty string content", - async () => { - const llm = getBundledModel("google-antigravity", "claude-sonnet-4-5"); - await testEmptyStringMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "claude-sonnet-4-5 - should handle whitespace-only content", - async () => { - const llm = getBundledModel("google-antigravity", "claude-sonnet-4-5"); - await testWhitespaceOnlyMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "claude-sonnet-4-5 - should handle empty assistant message in conversation", - async () => { - const llm = getBundledModel("google-antigravity", "claude-sonnet-4-5"); - await testEmptyAssistantMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "gpt-oss-120b-medium - should handle empty content array", - async () => { - const llm = getBundledModel("google-antigravity", "gpt-oss-120b-medium"); - await testEmptyMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "gpt-oss-120b-medium - should handle empty string content", - async () => { - const llm = getBundledModel("google-antigravity", "gpt-oss-120b-medium"); - await testEmptyStringMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "gpt-oss-120b-medium - should handle whitespace-only content", - async () => { - const llm = getBundledModel("google-antigravity", "gpt-oss-120b-medium"); - await testWhitespaceOnlyMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!antigravityToken)( - "gpt-oss-120b-medium - should handle empty assistant message in conversation", - async () => { - const llm = getBundledModel("google-antigravity", "gpt-oss-120b-medium"); - await testEmptyAssistantMessage(llm, { apiKey: antigravityToken }); - }, - { retry: 3, timeout: 30000 }, - ); - }); - - describe("OpenAI Codex Provider Empty Messages", () => { - it.skipIf(!openaiCodexToken)( - "gpt-5.2-codex - should handle empty content array", - async () => { - const llm = getBundledModel("openai-codex", "gpt-5.2-codex"); - await testEmptyMessage(llm, { apiKey: openaiCodexToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!openaiCodexToken)( - "gpt-5.2-codex - should handle empty string content", - async () => { - const llm = getBundledModel("openai-codex", "gpt-5.2-codex"); - await testEmptyStringMessage(llm, { apiKey: openaiCodexToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!openaiCodexToken)( - "gpt-5.2-codex - should handle whitespace-only content", - async () => { - const llm = getBundledModel("openai-codex", "gpt-5.2-codex"); - await testWhitespaceOnlyMessage(llm, { apiKey: openaiCodexToken }); - }, - { retry: 3, timeout: 30000 }, - ); - - it.skipIf(!openaiCodexToken)( - "gpt-5.2-codex - should handle empty assistant message in conversation", - async () => { - const llm = getBundledModel("openai-codex", "gpt-5.2-codex"); - await testEmptyAssistantMessage(llm, { apiKey: openaiCodexToken }); - }, - { retry: 3, timeout: 30000 }, - ); - }); -}); diff --git a/packages/ai/test/github-copilot-claude-messages-routing.test.ts b/packages/ai/test/github-copilot-claude-messages-routing.test.ts deleted file mode 100644 index 9c42922df..000000000 --- a/packages/ai/test/github-copilot-claude-messages-routing.test.ts +++ /dev/null @@ -1,49 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { getBundledModel } from "../src/models"; - -describe("Copilot Claude model routing", () => { - it("routes claude-sonnet-4 via anthropic-messages API", () => { - const model = getBundledModel("github-copilot", "claude-sonnet-4"); - expect(model).toBeDefined(); - expect(model.api).toBe("anthropic-messages"); - }); - - it("routes claude-sonnet-4.5 via anthropic-messages API", () => { - const model = getBundledModel("github-copilot", "claude-sonnet-4.5"); - expect(model).toBeDefined(); - expect(model.api).toBe("anthropic-messages"); - }); - - it("routes claude-haiku-4.5 via anthropic-messages API", () => { - const model = getBundledModel("github-copilot", "claude-haiku-4.5"); - expect(model).toBeDefined(); - expect(model.api).toBe("anthropic-messages"); - }); - - it("routes claude-opus-4.5 via anthropic-messages API", () => { - const model = getBundledModel("github-copilot", "claude-opus-4.5"); - expect(model).toBeDefined(); - expect(model.api).toBe("anthropic-messages"); - }); - - it("does not have compat block on Claude models (completions-API-specific)", () => { - const sonnet = getBundledModel("github-copilot", "claude-sonnet-4"); - expect("compat" in sonnet).toBe(false); - }); - - it("preserves static Copilot headers on Claude models", () => { - const model = getBundledModel("github-copilot", "claude-sonnet-4"); - expect(model.headers).toBeDefined(); - expect(model.headers?.["User-Agent"]).toContain("opencode"); - }); - - it("keeps non-Claude Copilot models on their existing APIs", () => { - const gpt4o = getBundledModel("github-copilot", "gpt-4o"); - expect(gpt4o).toBeDefined(); - expect(gpt4o.api).toBe("openai-completions"); - - const gpt5 = getBundledModel("github-copilot", "gpt-5"); - expect(gpt5).toBeDefined(); - expect(gpt5.api).toBe("openai-responses"); - }); -}); diff --git a/packages/ai/test/gitlab-duo-model-mapping.test.ts b/packages/ai/test/gitlab-duo-model-mapping.test.ts deleted file mode 100644 index 8815a68c0..000000000 --- a/packages/ai/test/gitlab-duo-model-mapping.test.ts +++ /dev/null @@ -1,20 +0,0 @@ -import { describe, expect, test } from "bun:test"; -import { getModelMapping } from "../src/providers/gitlab-duo"; - -describe("gitlab duo model mapping", () => { - test("resolves Duo alias IDs", () => { - const mapping = getModelMapping("duo-chat-gpt-5-codex"); - expect(mapping).toBeDefined(); - expect(mapping?.model).toBe("gpt-5-codex"); - }); - - test("resolves canonical model IDs", () => { - const mapping = getModelMapping("gpt-5-codex"); - expect(mapping).toBeDefined(); - expect(mapping?.model).toBe("gpt-5-codex"); - }); - - test("returns undefined for unknown IDs", () => { - expect(getModelMapping("totally-unknown-model")).toBeUndefined(); - }); -}); diff --git a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts index 06f73e7d2..9e1caf756 100644 --- a/packages/ai/test/google-gemini-cli-3x-thinking.test.ts +++ b/packages/ai/test/google-gemini-cli-3x-thinking.test.ts @@ -2,7 +2,6 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { Effort } from "@oh-my-pi/pi-ai"; import { enrichModelThinking } from "@oh-my-pi/pi-ai/model-thinking"; import { hookFetch } from "@oh-my-pi/pi-utils"; -import { getBundledModel } from "../src/models"; import { streamSimple } from "../src/stream"; import type { Context, Model } from "../src/types"; @@ -48,10 +47,6 @@ describe("google-gemini-cli Gemini 3.x thinking mapping", () => { afterEach(() => { vi.restoreAllMocks(); }); - - it("includes gemini-3.1-pro-preview in bundled google-gemini-cli models", () => { - expect(getBundledModel("google-gemini-cli", "gemini-3.1-pro-preview")?.id).toBe("gemini-3.1-pro-preview"); - }); it("uses thinkingLevel for gemini-3.1-pro-preview when the effort is supported", async () => { let requestBody: string | undefined; using _hook = hookFetch((_input, init) => { diff --git a/packages/ai/test/issue-887-repro.test.ts b/packages/ai/test/issue-887-repro.test.ts index 6ae43112b..7b4f91350 100644 --- a/packages/ai/test/issue-887-repro.test.ts +++ b/packages/ai/test/issue-887-repro.test.ts @@ -16,12 +16,6 @@ const OPENCODE_GO_BASE = "https://opencode.ai/zen/go/v1"; describe("opencode-go resolver routes 404-ing ids to openai-completions (issue #887)", () => { const descriptor = MODELS_DEV_PROVIDER_DESCRIPTORS.find(d => d.providerId === "opencode-go"); - test("descriptor exists and exposes resolveApi", () => { - expect(descriptor).toBeDefined(); - expect(descriptor?.modelsDevKey).toBe("opencode-go"); - expect(descriptor?.resolveApi).toBeTypeOf("function"); - }); - // Per upstream models.dev (verified 2026-05-02 against // https://models.dev/api.json["opencode-go"].models), these three ids carry // `provider.npm = "@ai-sdk/anthropic"`. The naive @ai-sdk/anthropic rule diff --git a/packages/ai/test/issue-945-repro.test.ts b/packages/ai/test/issue-945-repro.test.ts index 03da0848c..6f5e361f5 100644 --- a/packages/ai/test/issue-945-repro.test.ts +++ b/packages/ai/test/issue-945-repro.test.ts @@ -1,6 +1,6 @@ import { afterEach, describe, expect, it } from "bun:test"; import { getBundledModel } from "@oh-my-pi/pi-ai/models"; -import { detectCompat, streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; +import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions"; import type { Context, Model, Tool } from "@oh-my-pi/pi-ai/types"; import { Type } from "@sinclair/typebox"; @@ -38,19 +38,16 @@ async function capturePayload(opts: Parameters<typeof streamOpenAICompletions>[2 return (await promise) as Record<string, unknown>; } -describe("issue #945 — OpenCode Go DeepSeek disables reasoning when tool_choice is used", () => { - it("detects deepseek-v4-pro as supporting tool_choice with per-request reasoning suppression", () => { +describe("issue #945 — OpenCode Go DeepSeek tool_choice is disabled", () => { + it("marks deepseek-v4-pro as not supporting tool_choice via compat override", () => { const model = getBundledModel("opencode-go", "deepseek-v4-pro") as Model<"openai-completions">; - expect(model.compat?.supportsToolChoice).toBeUndefined(); - const compat = detectCompat(model); - expect(compat.supportsToolChoice).toBe(true); - expect(compat.disableReasoningOnToolChoice).toBe(true); + expect(model.compat?.supportsToolChoice).toBe(false); }); - it("preserves tool_choice and tools while omitting reasoning_effort", async () => { + it("omits tool_choice from payload but preserves tools and reasoning_effort", async () => { const body = await capturePayload({ reasoning: "high", toolChoice: "auto" }); expect(body.tools).toBeDefined(); - expect(body.tool_choice).toBe("auto"); - expect(body.reasoning_effort).toBeUndefined(); + expect(body.tool_choice).toBeUndefined(); + expect(body.reasoning_effort).toBe("high"); }); }); diff --git a/packages/ai/test/kilo-provider.test.ts b/packages/ai/test/kilo-provider.test.ts deleted file mode 100644 index 44421840b..000000000 --- a/packages/ai/test/kilo-provider.test.ts +++ /dev/null @@ -1,41 +0,0 @@ -import { afterEach, describe, expect, test } from "bun:test"; -import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "../src/provider-models/descriptors"; -import { kiloModelManagerOptions } from "../src/provider-models/openai-compat"; -import { getEnvApiKey } from "../src/stream"; -import { getOAuthProviders } from "../src/utils/oauth"; - -const originalKiloApiKey = Bun.env.KILO_API_KEY; - -afterEach(() => { - if (originalKiloApiKey === undefined) { - delete Bun.env.KILO_API_KEY; - return; - } - Bun.env.KILO_API_KEY = originalKiloApiKey; -}); - -describe("kilo provider support", () => { - test("resolves KILO_API_KEY from environment", () => { - Bun.env.KILO_API_KEY = "kilo-test-key"; - expect(getEnvApiKey("kilo")).toBe("kilo-test-key"); - }); - - test("registers built-in descriptor and default model", () => { - const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "kilo"); - expect(descriptor).toBeDefined(); - expect(descriptor?.defaultModel).toBe("anthropic/claude-sonnet-4.5"); - expect(descriptor?.catalogDiscovery?.envVars).toContain("KILO_API_KEY"); - expect(descriptor?.catalogDiscovery?.allowUnauthenticated).toBe(true); - expect(DEFAULT_MODEL_PER_PROVIDER.kilo).toBe("anthropic/claude-sonnet-4.5"); - }); - - test("registers Kilo in OAuth provider selector", () => { - const provider = getOAuthProviders().find(item => item.id === "kilo"); - expect(provider?.name).toBe("Kilo Gateway"); - }); - test("builds model manager options with kilo defaults", () => { - const options = kiloModelManagerOptions(); - expect(options.providerId).toBe("kilo"); - expect(options.fetchDynamicModels).toBeDefined(); - }); -}); diff --git a/packages/ai/test/model-thinking.test.ts b/packages/ai/test/model-thinking.test.ts index e80c1c355..ac0ac3dab 100644 --- a/packages/ai/test/model-thinking.test.ts +++ b/packages/ai/test/model-thinking.test.ts @@ -10,8 +10,6 @@ import { requireSupportedEffort, } from "@oh-my-pi/pi-ai/model-thinking"; import type { Api, Model, Provider } from "@oh-my-pi/pi-ai/types"; -import { getBundledModel } from "../src/models"; -import MODELS from "../src/models.json" with { type: "json" }; function createModel<TApi extends Api>(overrides: { id: string; @@ -133,52 +131,6 @@ describe("model thinking metadata", () => { }); }); -describe("bundled GPT-5.4 model metadata", () => { - it("stores raw GPT-5.4 mini/nano catalog metadata for OpenAI, OpenAI Codex, and Copilot", () => { - const openAiMini = MODELS.openai["gpt-5.4-mini"]; - const openAiNano = MODELS.openai["gpt-5.4-nano"]; - const openAiCodexMini = MODELS["openai-codex"]["gpt-5.4-mini"]; - const openAiCodexNano = MODELS["openai-codex"]["gpt-5.4-nano"]; - const copilotMini = MODELS["github-copilot"]["gpt-5.4-mini"]; - - expect(openAiMini?.thinking).toEqual({ mode: "effort", minLevel: "low", maxLevel: "xhigh" }); - expect(openAiNano?.thinking).toEqual({ mode: "effort", minLevel: "low", maxLevel: "xhigh" }); - expect(openAiCodexMini?.thinking).toEqual({ mode: "effort", minLevel: "low", maxLevel: "xhigh" }); - expect(openAiCodexNano?.thinking).toEqual({ mode: "effort", minLevel: "low", maxLevel: "xhigh" }); - expect(copilotMini?.thinking).toEqual({ mode: "effort", minLevel: "low", maxLevel: "xhigh" }); - expect(openAiCodexMini?.api).toBe("openai-codex-responses"); - expect(openAiCodexNano?.api).toBe("openai-codex-responses"); - expect(openAiCodexMini?.contextWindow).toBe(272000); - expect(openAiCodexNano?.contextWindow).toBe(272000); - expect(openAiCodexMini?.preferWebsockets).toBe(true); - expect(openAiCodexNano?.preferWebsockets).toBe(true); - expect(openAiCodexMini?.priority).toBe(1); - expect(openAiCodexNano?.priority).toBe(2); - }); - - it("exposes xhigh support for bundled GPT-5.4 mini/nano runtime models across supported providers", () => { - const openAiMini = getBundledModel("openai", "gpt-5.4-mini"); - const openAiNano = getBundledModel("openai", "gpt-5.4-nano"); - const openAiCodexMini = getBundledModel("openai-codex", "gpt-5.4-mini"); - const openAiCodexNano = getBundledModel("openai-codex", "gpt-5.4-nano"); - const copilotMini = getBundledModel("github-copilot", "gpt-5.4-mini"); - - expect(openAiCodexMini.contextWindow).toBe(272000); - expect(openAiCodexNano.contextWindow).toBe(272000); - expect(requireSupportedEffort(openAiMini, Effort.XHigh)).toBe(Effort.XHigh); - expect(requireSupportedEffort(openAiNano, Effort.XHigh)).toBe(Effort.XHigh); - expect(requireSupportedEffort(openAiCodexMini, Effort.XHigh)).toBe(Effort.XHigh); - expect(requireSupportedEffort(openAiCodexNano, Effort.XHigh)).toBe(Effort.XHigh); - expect(requireSupportedEffort(copilotMini, Effort.XHigh)).toBe(Effort.XHigh); - }); - - it("does not bundle GitHub Copilot GPT-5.4 nano", () => { - const copilotModels = MODELS["github-copilot"] as Record<string, unknown>; - expect(copilotModels["gpt-5.4-nano"]).toBeUndefined(); - expect(getBundledModel("github-copilot", "gpt-5.4-nano")).toBeUndefined(); - }); -}); - describe("generated model policies", () => { it("refreshes thinking metadata and applies parsed catalog corrections", () => { const models: Model<Api>[] = [ diff --git a/packages/ai/test/ollama-cloud-provider.test.ts b/packages/ai/test/ollama-cloud-provider.test.ts index 922fc9349..4c07d567e 100644 --- a/packages/ai/test/ollama-cloud-provider.test.ts +++ b/packages/ai/test/ollama-cloud-provider.test.ts @@ -1,9 +1,7 @@ import { afterEach, describe, expect, test, vi } from "bun:test"; -import { DEFAULT_MODEL_PER_PROVIDER, PROVIDER_DESCRIPTORS } from "../src/provider-models/descriptors"; import { ollamaCloudModelManagerOptions } from "../src/provider-models/ollama"; import { completeSimple, getEnvApiKey, stream, streamSimple } from "../src/stream"; import type { Context, Model, Tool } from "../src/types"; -import { getOAuthProviders } from "../src/utils/oauth"; const originalApiKey = Bun.env.OLLAMA_CLOUD_API_KEY; const originalFetch = global.fetch; @@ -64,18 +62,6 @@ describe("ollama-cloud provider support", () => { expect(getEnvApiKey("ollama-cloud")).toBe("ollama-cloud-test-key"); }); - test("registers built-in descriptor, default model, and oauth selector entry", () => { - const descriptor = PROVIDER_DESCRIPTORS.find(item => item.providerId === "ollama-cloud"); - expect(descriptor).toBeDefined(); - expect(descriptor?.defaultModel).toBe("gpt-oss:120b"); - expect(descriptor?.catalogDiscovery?.envVars).toContain("OLLAMA_CLOUD_API_KEY"); - expect(descriptor?.catalogDiscovery?.allowUnauthenticated).toBeUndefined(); - expect(DEFAULT_MODEL_PER_PROVIDER["ollama-cloud"]).toBe("gpt-oss:120b"); - - const provider = getOAuthProviders().find(item => item.id === "ollama-cloud"); - expect(provider?.name).toBe("Ollama Cloud"); - }); - test("discovers ollama-cloud models from native cloud endpoints", async () => { global.fetch = vi.fn(async (input, init) => { const url = String(input); diff --git a/packages/ai/test/openai-responses-cache-affinity.test.ts b/packages/ai/test/openai-responses-cache-affinity.test.ts index cc0d6e994..31a8cb7dc 100644 --- a/packages/ai/test/openai-responses-cache-affinity.test.ts +++ b/packages/ai/test/openai-responses-cache-affinity.test.ts @@ -1,10 +1,6 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import { getBundledModel } from "../src/models"; -import { - normalizeOpenAIResponsesPromptCacheKey, - type OpenAIResponsesOptions, - streamOpenAIResponses, -} from "../src/providers/openai-responses"; +import { type OpenAIResponsesOptions, streamOpenAIResponses } from "../src/providers/openai-responses"; import type { Context, Model } from "../src/types"; const originalFetch = global.fetch; @@ -115,23 +111,4 @@ describe("openai-responses cache affinity", () => { expect(captured.clientRequestId).toBeNull(); expect(captured.body?.prompt_cache_key).toBeUndefined(); }); - - it("normalizes long prompt cache keys while preserving ordered system prompts", async () => { - const longSessionId = "session-".repeat(20); - const expectedCacheKey = normalizeOpenAIResponsesPromptCacheKey(longSessionId); - if (!expectedCacheKey) throw new Error("Expected normalized prompt cache key"); - const captured = await captureOpenAIResponseHeaders({ sessionId: longSessionId }); - - expect(captured.sessionId).toBe(expectedCacheKey); - expect(captured.clientRequestId).toBe(expectedCacheKey); - expect(captured.body?.prompt_cache_key).toBe(expectedCacheKey); - expect(expectedCacheKey?.length).toBeLessThanOrEqual(64); - - const input = captured.body?.input; - expect(Array.isArray(input)).toBe(true); - expect((input as Array<{ role?: string; content?: string }>).slice(0, 2)).toEqual([ - { role: "developer", content: "stable system" }, - { role: "developer", content: "stable durable context" }, - ]); - }); }); diff --git a/packages/ai/test/rate-limit-utils.test.ts b/packages/ai/test/rate-limit-utils.test.ts index 485f5bc53..264a0bb4c 100644 --- a/packages/ai/test/rate-limit-utils.test.ts +++ b/packages/ai/test/rate-limit-utils.test.ts @@ -48,14 +48,6 @@ describe("parseRateLimitReason", () => { }); describe("calculateRateLimitBackoffMs", () => { - it("returns 30 minutes for QUOTA_EXHAUSTED", () => { - expect(calculateRateLimitBackoffMs("QUOTA_EXHAUSTED")).toBe(30 * 60 * 1000); - }); - - it("returns 30s for RATE_LIMIT_EXCEEDED", () => { - expect(calculateRateLimitBackoffMs("RATE_LIMIT_EXCEEDED")).toBe(30_000); - }); - it("returns 45–75s range for MODEL_CAPACITY_EXHAUSTED (jitter)", () => { for (let i = 0; i < 20; i++) { const ms = calculateRateLimitBackoffMs("MODEL_CAPACITY_EXHAUSTED"); @@ -63,12 +55,4 @@ describe("calculateRateLimitBackoffMs", () => { expect(ms).toBeLessThanOrEqual(75_000); } }); - - it("returns 20s for SERVER_ERROR", () => { - expect(calculateRateLimitBackoffMs("SERVER_ERROR")).toBe(20_000); - }); - - it("returns conservative fallback for UNKNOWN", () => { - expect(calculateRateLimitBackoffMs("UNKNOWN")).toBe(30 * 60 * 1000); - }); }); diff --git a/packages/ai/test/zen.test.ts b/packages/ai/test/zen.test.ts deleted file mode 100644 index 616ef3fe9..000000000 --- a/packages/ai/test/zen.test.ts +++ /dev/null @@ -1,26 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import MODELS from "@oh-my-pi/pi-ai/models.json" with { type: "json" }; -import { complete } from "@oh-my-pi/pi-ai/stream"; -import type { Model } from "@oh-my-pi/pi-ai/types"; -import { e2eApiKey } from "./oauth"; - -describe.skipIf(!e2eApiKey("OPENCODE_API_KEY"))("OpenCode Models Smoke Test", () => { - const providers = [ - { key: "opencode-zen", label: "OpenCode Zen" }, - { key: "opencode-go", label: "OpenCode Go" }, - ] as const; - - providers.forEach(({ key, label }) => { - const providerModels = Object.values(MODELS[key]); - providerModels.forEach(model => { - it(`${label}: ${model.id}`, async () => { - const response = await complete(model as unknown as Model, { - messages: [{ role: "user", content: "Say hello.", timestamp: Date.now() }], - }); - - expect(response.content).toBeTruthy(); - expect(response.stopReason).toBe("stop"); - }, 60000); - }); - }); -}); diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 953da574d..c2b958baa 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -1,11 +1,48 @@ # Changelog ## [Unreleased] +### Breaking Changes + +- Removed the `jobs://` internal URL protocol; inspect background jobs via the `job` tool's `list: true` operation instead + +### Added + +- Added `since` and `until` date-range filters to `search_issues`, `search_prs`, `search_commits`, and `search_repos`, accepting relative durations (`m`/`h`/`d`/`w`/`mo`/`y`), ISO dates, and ISO datetimes +- Added `dateField` support for date filtering (`created` or `updated`) so search results can be constrained by creation, update, pushed (for repos), or committer date (for commits) +- Added owner-based scoping to async job registration and queries so background jobs can be registered with an `ownerId` and filtered per agent in `getRunningJobs`, `getRecentJobs`, `getAllJobs`, and `cancelAll` +- Added agent ownership metadata to async jobs started by `task` and `bash` tools so their lifecycle and cancellation is attributed to the creating agent +- Added `list: true` operation to the `job` tool, returning an immediate snapshot of every job spawned by the calling agent without waiting (replaces the deleted `jobs://` URL) +- Added per-agent visibility scoping to the `job` tool so `list`, `poll`, and `cancel` only see and act on jobs owned by the calling agent; cross-agent operations now return `not_found` + +### Changed + +- Changed `search_issues`, `search_prs`, `search_commits`, and `search_repos` to allow date-only queries where `query` is omitted if `since`/`until` is provided +- Changed `search_code` to return a validation error when `since`/`until` is supplied because GitHub code search does not support date qualifiers +- Changed async job manager ownership so subagents inherit the parent session’s global `AsyncJobManager` instead of creating and owning separate instances +- Changed session lifecycle cleanup so the global async-job manager is disposed only by the owning top-level session +- Changed subagent session switches and handoff paths to stop global async-job cancellation and cancel only jobs owned by that session +- Changed `agent://` and `artifact://` URL resolution to search artifact outputs across all active sessions instead of only the current session, allowing parent and subagent sessions to read each other’s generated outputs by ID +- Changed `memory://` URL resolution to walk all active sessions’ memory roots and return the first matching file, so worktree-based subagents can access their own memory views as well as shared roots +- Changed internal URL routing to use a shared process-global `InternalUrlRouter` and protocol handlers, so built-in tools resolve `agent://`, `artifact://`, `memory://`, `skill://`, `rule://`, `mcp://`, and `local://` URLs without requiring session-specific router wiring +- Changed `mcp://` handler to use the globally registered MCP manager so MCP resource links work for agents sharing session context + +### Changed + +- Changed the `ask.timeout` default from `30` (seconds) to `0` (wait indefinitely). Auto-selecting the recommended option after a fixed delay was surprising users mid-deliberation; the timer is now strictly opt-in. The legacy auto-select behavior is preserved when `ask.timeout` is set to a non-zero value, and the `ask` tool's prompt has been updated so the model expects unlimited reply time by default. ### Fixed +- Added `ModelRegistry.hasConfiguredAuth(model)` to mirror the upstream `@mariozechner/pi-coding-agent` API surface; external plugins and downstream wrappers that pre-flight auth before launching a subagent no longer crash with `this._modelRegistry.hasConfiguredAuth is not a function` on the direct agent-launch path. ([#993](https://github.com/can1357/oh-my-pi/issues/993)) +- Fixed an ESM circular-import TDZ that crashed test suites when modules from the `task/` and `tools/` graphs were evaluated together (e.g. `executor-warnings.test.ts` + `task-simple-mode.test.ts`) by deferring `BUILTIN_TOOLS.task`'s `TaskTool.create` dereference to factory-call time and sourcing `truncateTail` from `session/streaming-output` instead of the `tools/` barrel +- Treat keyless-by-design providers (llama.cpp, ollama, lm-studio) as authenticated in subagent model resolution; fixes silent fallback to parent remote model when a local model is configured. ([#1008](https://github.com/can1357/oh-my-pi/issues/1008)) +- Fixed subagent disposal and session transitions that previously canceled all running async jobs, preventing inadvertent termination of a parent agent’s background work +- Fixed multi-entry edits silently rendering a fake success when every entry failed (e.g. all hit the auto-generated guard), by surfacing `isError: true` from the single-path edit orchestrator so the renderer takes the error branch instead of falling through to the streaming-preview fallback that displays the *proposed* diff +- Fixed the auto-generated streaming guard being gated behind `edit.streamingAbort` (default false), so it now pre-empts streaming edit tool calls targeting auto-generated files regardless of that setting - Fixed subagents launched in the same parallel batch not seeing each other in their initial `# IRC Peers` system-prompt block by pre-registering the agent in the global `AgentRegistry` before `rebuildSystemPrompt` runs and attaching the live session afterwards - Fixed plugin manifest extensions whose entry points at a directory (e.g. `pi-goal`'s `"pi": { "extensions": [".pi/extensions/pi-goal"] }`) failing to load with `Failed to load extension: Directories cannot be read like files`. The plugin path resolver now resolves directory entries to their `index.{ts,js,mjs,cjs}` file, matching the behavior of native auto-discovery via `resolveExtensionEntries`. +- Fixed the SSH tool on native Windows by avoiding OpenSSH ControlMaster multiplexing, which Win32-OpenSSH does not support and reports as `getsockname failed` ([#154](https://github.com/can1357/oh-my-pi/issues/154)). +- Fixed `/export` and `/tree` not showing developer-role messages (including the plan content injected after `/plan` approval) so the HTML export and TUI session tree now render developer messages dimmed with their actual content instead of hiding them entirely ([#753](https://github.com/can1357/oh-my-pi/issues/753)) +- Fixed `Timed out initializing browser tab worker` on prebuilt binaries by rewriting `spawnTabWorker` to import the worker entry with `with { type: "file" }` so Bun's `--compile` bundler statically discovers and embeds `tab-worker-entry.ts` in the single-file binary ([#1011](https://github.com/can1357/oh-my-pi/issues/1011)) ## [14.9.3] - 2026-05-10 ### Breaking Changes diff --git a/packages/coding-agent/bunfig.toml b/packages/coding-agent/bunfig.toml deleted file mode 100644 index 1bf5937d4..000000000 --- a/packages/coding-agent/bunfig.toml +++ /dev/null @@ -1,8 +0,0 @@ -[install] -linker = "isolated" -exact = true -saveTextLockfile = true - -[loader] -".md" = "text" -".py" = "text" diff --git a/packages/coding-agent/src/async/job-manager.ts b/packages/coding-agent/src/async/job-manager.ts index 3fc70ea3d..779d5c4d4 100644 --- a/packages/coding-agent/src/async/job-manager.ts +++ b/packages/coding-agent/src/async/job-manager.ts @@ -16,6 +16,13 @@ export interface AsyncJob { promise: Promise<void>; resultText?: string; errorText?: string; + /** + * Registry id of the agent that registered the job (e.g. "0-Main", + * "3-AuthLoader"). Used by scoped cancel/list APIs so a subagent's teardown + * does not cancel its parent's jobs. Undefined for callers that don't + * supply an id (e.g. legacy tests, SDK consumers without an agent context). + */ + ownerId?: string; } export interface AsyncJobManagerOptions { @@ -41,10 +48,38 @@ export interface AsyncJobDeliveryState { export interface AsyncJobRegisterOptions { id?: string; + /** Registry id of the agent that owns this job; used to scope cancelAll. */ + ownerId?: string; onProgress?: (text: string, details?: Record<string, unknown>) => void | Promise<void>; } +/** + * Filter applied to job query/cancel APIs. With `ownerId`, results are + * restricted to jobs registered by that agent (registry id from + * `AgentRegistry`, e.g. "0-Main", "3-AuthLoader"). + */ +export interface AsyncJobFilter { + ownerId?: string; +} + export class AsyncJobManager { + static #instance: AsyncJobManager | undefined; + + /** Process-global instance shared by internal URL protocol handlers and tools. */ + static instance(): AsyncJobManager | undefined { + return AsyncJobManager.#instance; + } + + /** Install or clear the process-global instance. */ + static setInstance(value: AsyncJobManager | undefined): void { + AsyncJobManager.#instance = value; + } + + /** Reset the process-global instance. Test-only. */ + static resetForTests(): void { + AsyncJobManager.#instance = undefined; + } + readonly #jobs = new Map<string, AsyncJob>(); readonly #deliveries: AsyncJobDelivery[] = []; readonly #suppressedDeliveries = new Set<string>(); @@ -56,6 +91,16 @@ export class AsyncJobManager { #deliveryLoop: Promise<void> | undefined; #disposed = false; + #filterJobs(jobs: Iterable<AsyncJob>, filter?: AsyncJobFilter): AsyncJob[] { + const ownerId = filter?.ownerId; + if (!ownerId) return Array.from(jobs); + const out: AsyncJob[] = []; + for (const job of jobs) { + if (job.ownerId === ownerId) out.push(job); + } + return out; + } + constructor(options: AsyncJobManagerOptions) { this.#onJobComplete = options.onJobComplete; this.#maxRunningJobs = Math.max(1, Math.floor(options.maxRunningJobs ?? DEFAULT_MAX_RUNNING_JOBS)); @@ -95,6 +140,7 @@ export class AsyncJobManager { label, abortController, promise: Promise.resolve(), + ownerId: options?.ownerId, }; const reportProgress = async (text: string, details?: Record<string, unknown>): Promise<void> => { @@ -138,9 +184,15 @@ export class AsyncJobManager { return id; } - cancel(id: string): boolean { + /** + * Cancel a single job by id. When `filter.ownerId` is set and does not + * match the job's owner, the call is treated as not-found (returns false) + * so cross-agent cancellation is rejected at the manager level. + */ + cancel(id: string, filter?: AsyncJobFilter): boolean { const job = this.#jobs.get(id); if (!job) return false; + if (filter?.ownerId && job.ownerId !== filter.ownerId) return false; if (job.status !== "running") return false; job.status = "cancelled"; job.abortController.abort(); @@ -152,19 +204,19 @@ export class AsyncJobManager { return this.#jobs.get(id); } - getRunningJobs(): AsyncJob[] { - return Array.from(this.#jobs.values()).filter(job => job.status === "running"); + getRunningJobs(filter?: AsyncJobFilter): AsyncJob[] { + return this.#filterJobs(this.#jobs.values(), filter).filter(job => job.status === "running"); } - getRecentJobs(limit = 10): AsyncJob[] { - return Array.from(this.#jobs.values()) + getRecentJobs(limit = 10, filter?: AsyncJobFilter): AsyncJob[] { + return this.#filterJobs(this.#jobs.values(), filter) .filter(job => job.status !== "running") .sort((a, b) => b.startTime - a.startTime) .slice(0, limit); } - getAllJobs(): AsyncJob[] { - return Array.from(this.#jobs.values()); + getAllJobs(filter?: AsyncJobFilter): AsyncJob[] { + return this.#filterJobs(this.#jobs.values(), filter); } getDeliveryState(): AsyncJobDeliveryState { @@ -221,8 +273,13 @@ export class AsyncJobManager { return before - this.#deliveries.length; } - cancelAll(): void { - for (const job of this.getRunningJobs()) { + /** + * Cancel running jobs. With `filter.ownerId` set, cancels only jobs the + * matching agent registered; with no filter, cancels every running job + * (used by `dispose()` to nuke the manager's state). + */ + cancelAll(filter?: AsyncJobFilter): void { + for (const job of this.getRunningJobs(filter)) { job.status = "cancelled"; job.abortController.abort(); this.#scheduleEviction(job.id); diff --git a/packages/coding-agent/src/capability/rule.ts b/packages/coding-agent/src/capability/rule.ts index 84caf469b..8b4623dd2 100644 --- a/packages/coding-agent/src/capability/rule.ts +++ b/packages/coding-agent/src/capability/rule.ts @@ -209,6 +209,26 @@ export function parseRuleConditionAndScope(frontmatter: RuleFrontmatter): Pick<R }; } +let activeRules: readonly Rule[] = []; + +/** + * Process-global snapshot of rules the active session loaded. + * Read by internal URL protocol handlers (rule://). + */ +export function getActiveRules(): readonly Rule[] { + return activeRules; +} + +/** Replace the active rule snapshot. Called once per top-level session. */ +export function setActiveRules(value: readonly Rule[]): void { + activeRules = value; +} + +/** Reset the active rule snapshot. Test-only. */ +export function resetActiveRulesForTests(): void { + activeRules = []; +} + export const ruleCapability = defineCapability<Rule>({ id: "rules", displayName: "Rules", diff --git a/packages/coding-agent/src/config/model-registry.ts b/packages/coding-agent/src/config/model-registry.ts index f190763ec..6b00432aa 100644 --- a/packages/coding-agent/src/config/model-registry.ts +++ b/packages/coding-agent/src/config/model-registry.ts @@ -2072,6 +2072,19 @@ export class ModelRegistry { return this.#models.filter(model => this.#isModelAvailable(model)); } + /** + * Check whether auth is configured for a model's provider. + * + * Mirrors the upstream `@mariozechner/pi-coding-agent` API surface so that + * external plugins/extensions and downstream wrappers (e.g. subagent launch + * paths that pre-flight auth before model resolution) can probe a model + * without resolving an API key. Returns true for keyless providers as well + * as providers with stored credentials. See issue #993. + */ + hasConfiguredAuth(model: Model<Api>): boolean { + return this.#keylessProviders.has(model.provider) || this.authStorage.hasAuth(model.provider); + } + getDiscoverableProviders(): string[] { const disabledProviders = getDisabledProviderIdsFromSettings(); return this.#discoverableProviders diff --git a/packages/coding-agent/src/config/model-resolver.ts b/packages/coding-agent/src/config/model-resolver.ts index 391096dac..00f9a30d8 100644 --- a/packages/coding-agent/src/config/model-resolver.ts +++ b/packages/coding-agent/src/config/model-resolver.ts @@ -16,7 +16,7 @@ import chalk from "chalk"; import MODEL_PRIO from "../priority.json" with { type: "json" }; import { parseThinkingLevel, resolveThinkingLevelForModel } from "../thinking"; import { fuzzyMatch } from "../utils/fuzzy"; -import { isAuthenticated, MODEL_ROLE_IDS, type ModelRegistry, type ModelRole } from "./model-registry"; +import { isAuthenticated, kNoAuth, MODEL_ROLE_IDS, type ModelRegistry, type ModelRole } from "./model-registry"; import type { Settings } from "./settings"; /** Default model IDs for each known provider */ @@ -743,6 +743,12 @@ export function resolveModelOverride( * `modelRoles.task` pointing at an unqualified id whose only available * provider variant has no configured credentials — see #985). * + * Keyless-by-design providers (llama.cpp, ollama, lm-studio) advertise the + * `kNoAuth` sentinel from `getApiKey` to signal that they do not require + * credentials. Those are treated as authenticated here so an explicitly + * configured local model is never silently rerouted to the parent's remote + * provider (see #1008). + * * If neither the subagent nor the parent has working auth, returns the * primary resolution unchanged so the existing error path still surfaces * a meaningful failure downstream. @@ -764,7 +770,7 @@ export async function resolveModelOverrideWithAuthFallback( } const primaryKey = await modelRegistry.getApiKey(primary.model); - if (isAuthenticated(primaryKey)) { + if (primaryKey === kNoAuth || isAuthenticated(primaryKey)) { return { ...primary, authFallbackUsed: false }; } diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index eaa651631..c93397dcf 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -905,7 +905,7 @@ export const SETTINGS_SCHEMA = { "ask.timeout": { type: "number", - default: 30, + default: 0, ui: { tab: "interaction", label: "Ask Timeout", diff --git a/packages/coding-agent/src/edit/index.ts b/packages/coding-agent/src/edit/index.ts index 4410eed62..5cc9b3962 100644 --- a/packages/coding-agent/src/edit/index.ts +++ b/packages/coding-agent/src/edit/index.ts @@ -204,6 +204,7 @@ async function executeSinglePathEntries( const contentTexts: string[] = []; const diffTexts: string[] = []; let firstChangedLine: number | undefined; + let errorCount = 0; for (let i = 0; i < runs.length; i++) { const isLast = i === runs.length - 1; @@ -221,6 +222,7 @@ async function executeSinglePathEntries( } catch (err) { const errorText = err instanceof Error ? err.message : String(err); contentTexts.push(`Error editing ${path}: ${errorText}`); + errorCount++; } if (!isLast && onUpdate) { @@ -230,6 +232,7 @@ async function executeSinglePathEntries( diff: diffTexts.join("\n"), firstChangedLine, }, + ...(errorCount > 0 ? { isError: true } : {}), }); } } @@ -240,6 +243,11 @@ async function executeSinglePathEntries( diff: diffTexts.join("\n"), firstChangedLine, }, + // Any per-entry failure marks the aggregate result as an error so the + // renderer takes the error branch instead of falling through to the + // streaming-edit preview (which displays the *proposed* diff and looks + // indistinguishable from success). + ...(errorCount > 0 ? { isError: true } : {}), }; } diff --git a/packages/coding-agent/src/edit/renderer.ts b/packages/coding-agent/src/edit/renderer.ts index 3b8b3f30c..6d7163e15 100644 --- a/packages/coding-agent/src/edit/renderer.ts +++ b/packages/coding-agent/src/edit/renderer.ts @@ -148,6 +148,8 @@ export interface EditRenderContext { editDiffPreview?: DiffResult | DiffError; /** Multi-file streaming diff preview (edits spanning several files) */ perFileDiffPreview?: PerFileDiffPreview[]; + /** Raw in-flight edit text shown while a computed diff preview is unavailable */ + editStreamingFallback?: string; /** Function to render diff text with syntax highlighting */ renderDiff?: (diffText: string, options?: { filePath?: string }) => string; } @@ -272,7 +274,7 @@ function formatMultiFileStreamingDiff(previews: PerFileDiffPreview[], uiTheme: T const parts: string[] = []; for (const preview of previews) { if (!preview.diff && !preview.error) continue; - const header = uiTheme.fg("dim", `\n\n\u2500\u2500 ${shortenPath(preview.path)} \u2500\u2500`); + const header = uiTheme.fg("dim", `\n\n── ${shortenPath(preview.path)} ──`); if (preview.error) { parts.push(`${header}\n${uiTheme.fg("error", replaceTabs(preview.error, preview.path))}`); continue; @@ -306,6 +308,9 @@ function getCallPreview( if (args.newText || args.patch) { return renderPlainTextPreview(args.newText ?? args.patch ?? "", uiTheme, rawPath); } + if (renderContext?.editStreamingFallback) { + return renderContext.editStreamingFallback; + } return ""; } diff --git a/packages/coding-agent/src/edit/streaming.ts b/packages/coding-agent/src/edit/streaming.ts index 7285a2956..a8ee0f6dc 100644 --- a/packages/coding-agent/src/edit/streaming.ts +++ b/packages/coding-agent/src/edit/streaming.ts @@ -13,14 +13,19 @@ * the injected `editMode` rather than probing argument shape. */ +import { sanitizeText } from "@oh-my-pi/pi-natives"; import { + ABORT_MARKER, + BEGIN_PATCH_MARKER, computeHashlineDiff, computeHashlineSectionDiff, containsRecognizableHashlineOperations, + END_PATCH_MARKER, type HashlineInputSection, splitHashlineInputs, } from "../hashline"; import type { Theme } from "../modes/theme/theme"; +import { replaceTabs, truncateToWidth } from "../tools/render-utils"; import { type EditMode, resolveEditMode } from "../utils/edit-mode"; import { computeEditDiff, type DiffError, type DiffResult } from "./diff"; import { type ApplyPatchEntry, expandApplyPatchToEntries, expandApplyPatchToPreviewEntries } from "./modes/apply-patch"; @@ -61,6 +66,52 @@ export interface EditStreamingStrategy<Args = unknown> { renderStreamingFallback(args: Args, uiTheme: Theme): string; } +const STREAMING_FALLBACK_LINES = 12; +const STREAMING_FALLBACK_WIDTH = 80; + +function isHashlineHeaderLine(line: string): boolean { + const trimmed = line.trimEnd(); + return trimmed.startsWith("@") && trimmed.length > 1; +} + +function isHashlineEnvelopeMarkerLine(line: string): boolean { + const trimmed = line.trimEnd(); + return trimmed === BEGIN_PATCH_MARKER || trimmed === END_PATCH_MARKER || trimmed === ABORT_MARKER; +} + +function trimHashlineStreamingSyntax(lines: string[]): string[] { + let index = lines.findIndex(line => line.trim().length > 0); + if (index === -1) return []; + + if (lines[index].trimEnd() === BEGIN_PATCH_MARKER) { + index++; + while (index < lines.length && lines[index].trim().length === 0) index++; + } + if (index < lines.length && isHashlineHeaderLine(lines[index])) { + index++; + } + + return lines.slice(index).filter(line => !isHashlineEnvelopeMarkerLine(line)); +} + +function renderHashlineInputFallback(input: string, uiTheme: Theme): string { + const lines = trimHashlineStreamingSyntax(sanitizeText(input).split("\n")); + if (!lines.some(line => line.trim().length > 0)) return ""; + + const displayLines = lines.slice(-STREAMING_FALLBACK_LINES); + const hidden = lines.length - displayLines.length; + let text = "\n\n"; + text += displayLines + .map(line => uiTheme.fg("toolOutput", truncateToWidth(replaceTabs(line), STREAMING_FALLBACK_WIDTH))) + .join("\n"); + if (hidden > 0) { + text += uiTheme.fg("dim", `\n… (streaming +${hidden} lines)`); + } else { + text += uiTheme.fg("dim", "\n(streaming)"); + } + return text; +} + // ----------------------------------------------------------------------------- // Partial-JSON handling // ----------------------------------------------------------------------------- @@ -273,8 +324,8 @@ const hashlineStrategy: EditStreamingStrategy<HashlineArgs> = { } return previews.length > 0 ? previews : null; }, - renderStreamingFallback() { - return ""; + renderStreamingFallback(args, uiTheme) { + return typeof args.input === "string" ? renderHashlineInputFallback(args.input, uiTheme) : ""; }, }; diff --git a/packages/coding-agent/src/eval/js/context-manager.ts b/packages/coding-agent/src/eval/js/context-manager.ts index d865924f0..4987d1be8 100644 --- a/packages/coding-agent/src/eval/js/context-manager.ts +++ b/packages/coding-agent/src/eval/js/context-manager.ts @@ -32,9 +32,6 @@ interface VmHelperOptions { reverse?: boolean; unique?: boolean; count?: boolean; - cwd?: string; - timeoutMs?: number; - timeout?: number; } interface VmContextState { @@ -303,41 +300,6 @@ async function createHelpers(state: VmContextState) { emitStatus(state, { op: "tree", path: root, entries: entryCount, preview: result.slice(0, 1000) }); return result; }, - run: async ( - command: string, - options: VmHelperOptions = {}, - ): Promise<{ stdout: string; stderr: string; exit_code: number }> => { - const cwd = options.cwd ? resolvePath(state, options.cwd) : state.cwd; - const timeoutMs = - typeof options.timeoutMs === "number" - ? options.timeoutMs - : typeof options.timeout === "number" - ? options.timeout * 1000 - : undefined; - const timeoutSignal = - typeof timeoutMs === "number" && Number.isFinite(timeoutMs) && timeoutMs > 0 - ? AbortSignal.timeout(timeoutMs) - : undefined; - const signal = - state.currentRun?.signal && timeoutSignal - ? AbortSignal.any([state.currentRun.signal, timeoutSignal]) - : (state.currentRun?.signal ?? timeoutSignal); - const child = Bun.spawn(["bash", "-lc", command], { - cwd, - env: getMergedEnv(state), - stdout: "pipe", - stderr: "pipe", - signal, - }); - const [stdout, stderr, exit_code] = await Promise.all([ - new Response(child.stdout as ReadableStream<Uint8Array>).text(), - new Response(child.stderr as ReadableStream<Uint8Array>).text(), - child.exited, - ]); - const output = `${stdout}${stderr}`.slice(0, 500); - emitStatus(state, { op: "run", cmd: command.slice(0, 120), code: exit_code, output }); - return { stdout, stderr, exit_code }; - }, env: (key?: string, value?: string): string | Record<string, string> | undefined => { if (!key) { const env = Object.fromEntries(Object.entries(getMergedEnv(state)).sort(([a], [b]) => a.localeCompare(b))); @@ -419,6 +381,7 @@ async function createVmState( atob, btoa, Buffer, + Bun, process: createProcessSubset(cwd), require: buildRequire(cwd), createRequire, diff --git a/packages/coding-agent/src/eval/js/prelude.txt b/packages/coding-agent/src/eval/js/prelude.txt index b4f6ae701..0fd503eab 100644 --- a/packages/coding-agent/src/eval/js/prelude.txt +++ b/packages/coding-agent/src/eval/js/prelude.txt @@ -12,7 +12,6 @@ if (!globalThis.__omp_js_prelude_loaded__) { const counter = (items, opts = {}) => callHelper("counter", items, toOptions(opts)); const diff = (a, b) => callHelper("diff", a, b); const tree = (path = ".", opts = {}) => callHelper("tree", path, toOptions(opts)); - const run = (cmd, opts = {}) => callHelper("run", cmd, toOptions(opts)); const env = (key, value) => callHelper("env", key, value); const tool = new Proxy( @@ -67,6 +66,5 @@ if (!globalThis.__omp_js_prelude_loaded__) { globalThis.counter = counter; globalThis.diff = diff; globalThis.tree = tree; - globalThis.run = run; globalThis.env = env; } diff --git a/packages/coding-agent/src/eval/py/executor.ts b/packages/coding-agent/src/eval/py/executor.ts index 7aedb1de6..123bb417f 100644 --- a/packages/coding-agent/src/eval/py/executor.ts +++ b/packages/coding-agent/src/eval/py/executor.ts @@ -39,6 +39,13 @@ export interface PythonExecutorOptions { useSharedGateway?: boolean; /** Session file path for accessing task outputs */ sessionFile?: string; + /** + * Effective artifacts directory for the current session. Subagents share + * the parent's directory, so this can differ from `sessionFile`'s sibling + * dir. When present, exported to the kernel as `PI_ARTIFACTS_DIR` and + * preferred over `PI_SESSION_FILE`-derived paths. + */ + artifactsDir?: string; /** Artifact path/id for full output storage */ artifactPath?: string; artifactId?: string; @@ -102,6 +109,7 @@ let cleanupTimer: NodeJS.Timeout | null = null; interface KernelSessionExecutionOptions { useSharedGateway?: boolean; sessionFile?: string; + artifactsDir?: string; signal?: AbortSignal; deadlineMs?: number; kernelOwnerId?: string; @@ -123,6 +131,19 @@ function getExecutionDeadlineMs(options?: Pick<PythonExecutorOptions, "deadlineM return Date.now() + options.timeoutMs; } +/** + * Build the env block exposed to the Python kernel. Includes the session file + * (for things that need the raw session path) and the effective artifacts + * directory (preferred by the prelude when resolving output IDs, so subagents + * see the parent's flat dir instead of a non-existent sibling). + */ +function buildKernelEnv(options: { sessionFile?: string; artifactsDir?: string }): Record<string, string> | undefined { + const env: Record<string, string> = {}; + if (options.sessionFile) env.PI_SESSION_FILE = options.sessionFile; + if (options.artifactsDir) env.PI_ARTIFACTS_DIR = options.artifactsDir; + return Object.keys(env).length > 0 ? env : undefined; +} + function getRemainingTimeoutMs(deadlineMs?: number): number | undefined { if (deadlineMs === undefined) return undefined; return deadlineMs - Date.now(); @@ -523,9 +544,7 @@ async function createKernelSession( isRetry?: boolean, ): Promise<KernelSession> { requireRemainingTimeoutMs(options.deadlineMs); - const env: Record<string, string> | undefined = options.sessionFile - ? { PI_SESSION_FILE: options.sessionFile } - : undefined; + const env = buildKernelEnv(options); const startOptions = buildKernelStartOptions(cwd, env, options); let kernel: PythonKernel; @@ -586,9 +605,7 @@ async function restartKernelSession( }); } } - const env: Record<string, string> | undefined = options.sessionFile - ? { PI_SESSION_FILE: options.sessionFile } - : undefined; + const env = buildKernelEnv(options); const startOptions = buildKernelStartOptions(cwd, env, options); const kernel = await PythonKernel.start(startOptions); session.kernel = kernel; @@ -936,10 +953,9 @@ export async function executePython(code: string, options?: PythonExecutorOption await ensureKernelAvailable(cwd); const kernelMode = executionOptions.kernelMode ?? "session"; - const sessionFile = executionOptions.sessionFile; if (kernelMode === "per-call") { - const env: Record<string, string> | undefined = sessionFile ? { PI_SESSION_FILE: sessionFile } : undefined; + const env = buildKernelEnv(executionOptions); requireRemainingTimeoutMs(deadlineMs); const startOptions = buildKernelStartOptions(cwd, env, executionOptions); const kernel = await PythonKernel.start(startOptions); diff --git a/packages/coding-agent/src/eval/py/index.ts b/packages/coding-agent/src/eval/py/index.ts index 8eec1a9a1..8fd63f8c7 100644 --- a/packages/coding-agent/src/eval/py/index.ts +++ b/packages/coding-agent/src/eval/py/index.ts @@ -35,6 +35,7 @@ export default { kernelMode, useSharedGateway, sessionFile: opts.sessionFile, + artifactsDir: opts.session.getArtifactsDir?.() ?? undefined, kernelOwnerId: opts.kernelOwnerId, reset: opts.reset, artifactPath: opts.artifactPath, diff --git a/packages/coding-agent/src/eval/py/prelude.py b/packages/coding-agent/src/eval/py/prelude.py index e9203610b..0c13619fe 100644 --- a/packages/coding-agent/src/eval/py/prelude.py +++ b/packages/coding-agent/src/eval/py/prelude.py @@ -3,7 +3,7 @@ from __future__ import annotations if "__omp_prelude_loaded__" not in globals(): __omp_prelude_loaded__ = True from pathlib import Path - import os, json, shutil, subprocess + import os, json from IPython.display import display as _ipy_display, JSON _PRESENTABLE_REPRS = ( @@ -79,79 +79,6 @@ if "__omp_prelude_loaded__" not in globals(): f.write(content) _emit_status("append", path=str(p), chars=len(content)) return p - class ShellResult: - """Result from shell command execution.""" - __slots__ = ("args", "stdout", "stderr", "returncode") - def __init__(self, args: str, stdout: str, stderr: str, returncode: int): - self.args = args - self.stdout = stdout - self.stderr = stderr - self.returncode = returncode - - @property - def code(self) -> int: - return self.returncode - - @property - def exit_code(self) -> int: - return self.returncode - - def check_returncode(self) -> None: - if self.returncode != 0: - raise subprocess.CalledProcessError( - self.returncode, self.args, output=self.stdout, stderr=self.stderr - ) - - def __repr__(self): - if self.returncode == 0: - return "" - return f"exit code {self.returncode}" - - def __bool__(self): - return self.returncode == 0 - - def _make_shell_result(proc: subprocess.CompletedProcess[str], cmd: str) -> ShellResult: - """Create ShellResult and emit status.""" - output = proc.stdout + proc.stderr if proc.stderr else proc.stdout - _emit_status("sh", cmd=cmd[:80], code=proc.returncode, output=output[:500]) - return ShellResult(cmd, proc.stdout, proc.stderr, proc.returncode) - - import signal as _signal - - def _run_with_interrupt(args: list[str], cwd: str | None, timeout: int | None, cmd: str) -> ShellResult: - """Run subprocess with proper interrupt handling.""" - proc = subprocess.Popen( - args, - cwd=cwd, - stdout=subprocess.PIPE, - stderr=subprocess.PIPE, - text=True, - start_new_session=True, - ) - try: - stdout, stderr = proc.communicate(timeout=timeout) - except KeyboardInterrupt: - os.killpg(proc.pid, _signal.SIGINT) - try: - stdout, stderr = proc.communicate(timeout=2) - except subprocess.TimeoutExpired: - os.killpg(proc.pid, _signal.SIGKILL) - stdout, stderr = proc.communicate() - result = subprocess.CompletedProcess(args, -_signal.SIGINT, stdout, stderr) - return _make_shell_result(result, cmd) - except subprocess.TimeoutExpired: - os.killpg(proc.pid, _signal.SIGKILL) - stdout, stderr = proc.communicate() - result = subprocess.CompletedProcess(args, -_signal.SIGKILL, stdout, stderr) - return _make_shell_result(result, cmd) - result = subprocess.CompletedProcess(args, proc.returncode, stdout, stderr) - return _make_shell_result(result, cmd) - - def run(cmd: str, *, cwd: str | Path | None = None, timeout: int | None = None) -> ShellResult: - """Run a shell command. Returns ShellResult with stdout/stderr and returncode/exit_code fields.""" - shell_path = shutil.which("bash") or shutil.which("sh") or "/bin/sh" - args = [shell_path, "-c", cmd] - return _run_with_interrupt(args, str(cwd) if cwd else None, timeout, cmd) def sort(text: str, *, reverse: bool = False, unique: bool = False) -> str: """Sort lines of text.""" @@ -268,12 +195,16 @@ if "__omp_prelude_loaded__" not in globals(): output('explore_0', offset=10, limit=20) # Lines 10-29 output('explore_0', 'reviewer_1') # Read multiple outputs """ - session_file = os.environ.get("PI_SESSION_FILE") - if not session_file: - _emit_status("output", error="No session file available") - raise RuntimeError("No session - output artifacts unavailable") - - artifacts_dir = session_file.rsplit(".", 1)[0] # Strip .jsonl extension + # Prefer PI_ARTIFACTS_DIR so subagents resolve through the parent's + # shared artifacts dir; fall back to deriving from PI_SESSION_FILE + # for legacy callers / top-level sessions where the two coincide. + artifacts_dir = os.environ.get("PI_ARTIFACTS_DIR") + if not artifacts_dir: + session_file = os.environ.get("PI_SESSION_FILE") + if not session_file: + _emit_status("output", error="No session file available") + raise RuntimeError("No session - output artifacts unavailable") + artifacts_dir = session_file.rsplit(".", 1)[0] # Strip .jsonl extension if not Path(artifacts_dir).exists(): _emit_status("output", error="Artifacts directory not found", path=artifacts_dir) raise RuntimeError(f"No artifacts directory found: {artifacts_dir}") diff --git a/packages/coding-agent/src/export/html/template.css b/packages/coding-agent/src/export/html/template.css index 63decc2cb..de869c410 100644 --- a/packages/coding-agent/src/export/html/template.css +++ b/packages/coding-agent/src/export/html/template.css @@ -179,6 +179,10 @@ color: var(--accent); } + .tree-role-developer { + color: var(--dim); + } + .tree-role-assistant { color: var(--success); } @@ -316,6 +320,14 @@ position: relative; } + .user-message.developer-message { + opacity: 0.7; + } + + .user-message.developer-message .markdown-content { + color: var(--dim); + } + .assistant-message { padding: 0; position: relative; diff --git a/packages/coding-agent/src/export/html/template.generated.ts b/packages/coding-agent/src/export/html/template.generated.ts index 40edaa6b3..cbb1aa425 100644 --- a/packages/coding-agent/src/export/html/template.generated.ts +++ b/packages/coding-agent/src/export/html/template.generated.ts @@ -1,2 +1,2 @@ // Auto-generated by scripts/generate-template.ts - DO NOT EDIT -export const TEMPLATE = "<!DOCTYPE html>\n<html lang=\"en\">\n<head>\n <meta charset=\"UTF-8\">\n <meta name=\"viewport\" content=\"width=device-width, initial-scale=1.0\">\n <title>Session Export\n \n \n\n\n \n
\n
\n \n
\n
\n
\n
\n
\n
\n \"\"\n
\n
\n\n \n \n \n \n\n\n"; +export const TEMPLATE = "\n\n\n \n \n Session Export\n \n \n\n\n \n
\n
\n \n
\n
\n
\n
\n
\n
\n \"\"\n
\n
\n\n \n \n \n \n\n\n"; diff --git a/packages/coding-agent/src/export/html/template.js b/packages/coding-agent/src/export/html/template.js index d093252e5..a61fcd217 100644 --- a/packages/coding-agent/src/export/html/template.js +++ b/packages/coding-agent/src/export/html/template.js @@ -456,6 +456,10 @@ const content = truncate(normalize(extractContent(msg.content))); return labelHtml + `user: ${escapeHtml(content)}`; } + if (msg.role === 'developer') { + const content = truncate(normalize(extractContent(msg.content))); + return labelHtml + `developer: ${escapeHtml(content)}`; + } if (msg.role === 'assistant') { const textContent = truncate(normalize(extractContent(msg.content))); if (textContent) { @@ -1648,6 +1652,18 @@ return html; } + if (msg.role === 'developer') { + let html = `
${copyBtnHtml}${tsHtml}`; + const content = msg.content; + const text = typeof content === 'string' ? content : + content.filter(c => c.type === 'text').map(c => c.text).join('\n'); + if (text.trim()) { + html += `
${safeMarkedParse(text)}
`; + } + html += '
'; + return html; + } + if (msg.role === 'assistant') { let html = `
${copyBtnHtml}${tsHtml}`; @@ -1750,7 +1766,7 @@ // ============================================================ function computeStats(entryList) { - let userMessages = 0, assistantMessages = 0, toolResults = 0; + let userMessages = 0, developerMessages = 0, assistantMessages = 0, toolResults = 0; let customMessages = 0, compactions = 0, branchSummaries = 0, toolCalls = 0; const tokens = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; const cost = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; @@ -1760,6 +1776,7 @@ if (entry.type === 'message') { const msg = entry.message; if (msg.role === 'user') userMessages++; + if (msg.role === 'developer') developerMessages++; if (msg.role === 'assistant') { assistantMessages++; if (msg.model) models.add(msg.provider ? `${msg.provider}/${msg.model}` : msg.model); @@ -1787,7 +1804,7 @@ } } - return { userMessages, assistantMessages, toolResults, customMessages, compactions, branchSummaries, toolCalls, tokens, cost, models: Array.from(models) }; + return { userMessages, developerMessages, assistantMessages, toolResults, customMessages, compactions, branchSummaries, toolCalls, tokens, cost, models: Array.from(models) }; } const globalStats = computeStats(entries); @@ -1803,6 +1820,7 @@ const msgParts = []; if (globalStats.userMessages) msgParts.push(`${globalStats.userMessages} user`); + if (globalStats.developerMessages) msgParts.push(`${globalStats.developerMessages} developer`); if (globalStats.assistantMessages) msgParts.push(`${globalStats.assistantMessages} assistant`); if (globalStats.toolResults) msgParts.push(`${globalStats.toolResults} tool results`); if (globalStats.customMessages) msgParts.push(`${globalStats.customMessages} custom`); diff --git a/packages/coding-agent/src/extensibility/skills.ts b/packages/coding-agent/src/extensibility/skills.ts index b78cdfa9e..ee048738f 100644 --- a/packages/coding-agent/src/extensibility/skills.ts +++ b/packages/coding-agent/src/extensibility/skills.ts @@ -28,6 +28,26 @@ export interface LoadSkillsResult { warnings: SkillWarning[]; } +let activeSkills: readonly Skill[] = []; + +/** + * Process-global snapshot of skills the active session loaded. + * Read by internal URL protocol handlers (skill://). + */ +export function getActiveSkills(): readonly Skill[] { + return activeSkills; +} + +/** Replace the active skill snapshot. Called once per top-level session. */ +export function setActiveSkills(value: readonly Skill[]): void { + activeSkills = value; +} + +/** Reset the active skill snapshot. Test-only. */ +export function resetActiveSkillsForTests(): void { + activeSkills = []; +} + export interface LoadSkillsFromDirOptions { /** Directory to scan for skills */ dir: string; diff --git a/packages/coding-agent/src/internal-urls/agent-protocol.ts b/packages/coding-agent/src/internal-urls/agent-protocol.ts index 69c8e7078..bd3f2df0b 100644 --- a/packages/coding-agent/src/internal-urls/agent-protocol.ts +++ b/packages/coding-agent/src/internal-urls/agent-protocol.ts @@ -1,7 +1,10 @@ /** * Protocol handler for agent:// URLs. * - * Resolves agent output IDs to artifact files in the session directory. + * Resolves agent output IDs against the artifacts directories of every active + * session. Parents and subagents share outputs via this registry: a subagent + * can read its parent's output IDs because both sessions are registered in + * the shared context. * * URL forms: * - agent:// - Full output content @@ -11,27 +14,27 @@ import * as fs from "node:fs/promises"; import * as path from "node:path"; import { isEnoent } from "@oh-my-pi/pi-utils"; +import { AgentRegistry } from "../registry/agent-registry"; import { applyQuery, pathToQuery } from "./json-query"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; -export interface AgentProtocolOptions { - /** - * Returns the artifacts directory path, or null if no session. - * Artifacts directory is the session file path without .jsonl extension. - */ - getArtifactsDir: () => string | null; -} - /** - * List available output IDs in artifacts directory. + * Snapshot of artifacts dirs for every registered session, deduped. + * + * Prefers `sessionManager.getArtifactsDir()` because subagents adopt the + * parent's manager and report the parent's dir there; dedup then collapses + * the whole agent tree to one entry. Falls back to the raw session file + * when no live session reference is attached. */ -async function listAvailableOutputs(artifactsDir: string): Promise { - try { - const files = await fs.readdir(artifactsDir); - return files.filter(f => f.endsWith(".md")).map(f => f.replace(".md", "")); - } catch { - return []; +function artifactsDirsFromRegistry(): string[] { + const dirs: string[] = []; + for (const ref of AgentRegistry.global().list()) { + const dir = + ref.session?.sessionManager.getArtifactsDir() ?? (ref.sessionFile ? ref.sessionFile.slice(0, -6) : null); + if (!dir) continue; + if (!dirs.includes(dir)) dirs.push(dir); } + return dirs; } /** @@ -44,30 +47,12 @@ export class AgentProtocolHandler implements ProtocolHandler { readonly scheme = "agent"; readonly immutable = true; - constructor(private readonly options: AgentProtocolOptions) {} - async resolve(url: InternalUrl): Promise { - const artifactsDir = this.options.getArtifactsDir(); - if (!artifactsDir) { - throw new Error("No session - agent outputs unavailable"); - } - - try { - await fs.stat(artifactsDir); - } catch (err) { - if (isEnoent(err)) { - throw new Error("No artifacts directory found"); - } - throw err; - } - - // Extract output ID from host const outputId = url.rawHost || url.hostname; if (!outputId) { throw new Error("agent:// URL requires an output ID: agent://"); } - // Check for conflicting extraction methods const urlPath = url.pathname; const queryParam = url.searchParams.get("q"); const hasPathExtraction = urlPath && urlPath !== "/" && urlPath !== ""; @@ -77,28 +62,57 @@ export class AgentProtocolHandler implements ProtocolHandler { throw new Error("agent:// URL cannot combine path extraction with ?q="); } - // Load the output file - const outputPath = path.join(artifactsDir, `${outputId}.md`); - try { - await fs.stat(outputPath); - } catch (err) { - if (isEnoent(err)) { - const available = await listAvailableOutputs(artifactsDir); - const availableStr = available.length > 0 ? available.join(", ") : "none"; - throw new Error(`Not found: ${outputId}\nAvailable: ${availableStr}`); - } - throw err; + const dirs = artifactsDirsFromRegistry(); + + if (dirs.length === 0) { + throw new Error("No session - agent outputs unavailable"); } - const rawContent = await Bun.file(outputPath).text(); - const notes: string[] = []; + let foundPath: string | undefined; + let anyDirExists = false; + const availableIds = new Set(); - // Handle extraction + for (const dir of dirs) { + try { + await fs.stat(dir); + anyDirExists = true; + } catch (err) { + if (isEnoent(err)) continue; + throw err; + } + const candidate = path.join(dir, `${outputId}.md`); + try { + await fs.stat(candidate); + foundPath = candidate; + break; + } catch (err) { + if (!isEnoent(err)) throw err; + try { + const files = await fs.readdir(dir); + for (const f of files) { + if (f.endsWith(".md")) availableIds.add(f.replace(/\.md$/, "")); + } + } catch { + // Listing failures are non-fatal; continue searching. + } + } + } + + if (!anyDirExists) { + throw new Error("No artifacts directory found"); + } + + if (!foundPath) { + const availableStr = availableIds.size > 0 ? [...availableIds].join(", ") : "none"; + throw new Error(`Not found: ${outputId}\nAvailable: ${availableStr}`); + } + + const rawContent = await Bun.file(foundPath).text(); + const notes: string[] = []; let content = rawContent; let contentType: InternalResource["contentType"] = "text/markdown"; if (hasPathExtraction || hasQueryExtraction) { - // Parse JSON let jsonValue: unknown; try { jsonValue = JSON.parse(rawContent); @@ -107,9 +121,7 @@ export class AgentProtocolHandler implements ProtocolHandler { throw new Error(`Output ${outputId} is not valid JSON: ${message}`); } - // Convert path to query if needed const query = hasPathExtraction ? pathToQuery(urlPath) : queryParam!; - if (query) { const extracted = applyQuery(jsonValue, query); try { @@ -119,7 +131,6 @@ export class AgentProtocolHandler implements ProtocolHandler { } notes.push(`Extracted: ${query}`); } else { - // Empty path/query means return full JSON content = JSON.stringify(jsonValue, null, 2); } contentType = "application/json"; @@ -130,7 +141,7 @@ export class AgentProtocolHandler implements ProtocolHandler { content, contentType, size: Buffer.byteLength(content, "utf-8"), - sourcePath: outputPath, + sourcePath: foundPath, notes, }; } diff --git a/packages/coding-agent/src/internal-urls/artifact-protocol.ts b/packages/coding-agent/src/internal-urls/artifact-protocol.ts index 9d2e70fef..844387b25 100644 --- a/packages/coding-agent/src/internal-urls/artifact-protocol.ts +++ b/packages/coding-agent/src/internal-urls/artifact-protocol.ts @@ -1,8 +1,8 @@ /** * Protocol handler for artifact:// URLs. * - * Resolves artifact IDs to files in the session artifacts directory. - * Unlike agent://, artifacts are raw text with no JSON extraction. + * Resolves artifact IDs against the artifacts directories of every active + * session. Unlike agent://, artifacts are raw text with no JSON extraction. * * URL form: * - artifact:// - Full artifact content @@ -12,87 +12,87 @@ import * as fs from "node:fs/promises"; import * as path from "node:path"; import { isEnoent } from "@oh-my-pi/pi-utils"; +import { AgentRegistry } from "../registry/agent-registry"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; -export interface ArtifactProtocolOptions { - /** - * Returns the artifacts directory path, or null if no session. - */ - getArtifactsDir: () => string | null; -} - /** - * List available artifact IDs in the directory. - */ -async function listAvailableArtifacts(artifactsDir: string): Promise { - try { - const files = await fs.readdir(artifactsDir); - return files - .filter(f => /^\d+\./.test(f)) - .map(f => f.split(".")[0]) - .sort((a, b) => Number(a) - Number(b)); - } catch { - return []; - } -} - -/** - * Handler for artifact:// URLs. + * Snapshot of artifacts dirs across all registered sessions, deduped. * - * Resolves numeric artifact IDs to their text content. - * Artifacts are created by tools when output is truncated. + * Subagents adopt their parent's `ArtifactManager`, so their + * `sessionManager.getArtifactsDir()` returns the parent's dir; dedup + * collapses parent + N subagents to a single entry. */ +function artifactsDirsFromRegistry(): string[] { + const dirs: string[] = []; + for (const ref of AgentRegistry.global().list()) { + const dir = + ref.session?.sessionManager.getArtifactsDir() ?? (ref.sessionFile ? ref.sessionFile.slice(0, -6) : null); + if (!dir) continue; + if (!dirs.includes(dir)) dirs.push(dir); + } + return dirs; +} + export class ArtifactProtocolHandler implements ProtocolHandler { readonly scheme = "artifact"; readonly immutable = true; - constructor(private readonly options: ArtifactProtocolOptions) {} - async resolve(url: InternalUrl): Promise { - const artifactsDir = this.options.getArtifactsDir(); - if (!artifactsDir) { - throw new Error("No session - artifacts unavailable"); - } - - // Extract artifact ID from host const id = url.rawHost || url.hostname; if (!id) { throw new Error("artifact:// URL requires a numeric ID: artifact://0"); } - - // Validate ID is numeric if (!/^\d+$/.test(id)) { throw new Error(`artifact:// ID must be numeric, got: ${id}`); } - // Check directory exists and find file matching ID prefix - let files: string[]; - try { - files = await fs.readdir(artifactsDir); - } catch (err) { - if (isEnoent(err)) { - throw new Error("No artifacts directory found"); - } - throw err; + const dirs = artifactsDirsFromRegistry(); + + if (dirs.length === 0) { + throw new Error("No session - artifacts unavailable"); } - const match = files.find(f => f.startsWith(`${id}.`)); + let foundPath: string | undefined; + let anyDirExists = false; + const availableIds = new Set(); - if (!match) { - const available = await listAvailableArtifacts(artifactsDir); - const availableStr = available.length > 0 ? available.join(", ") : "none"; + for (const dir of dirs) { + let files: string[]; + try { + files = await fs.readdir(dir); + anyDirExists = true; + } catch (err) { + if (isEnoent(err)) continue; + throw err; + } + const match = files.find(f => f.startsWith(`${id}.`)); + if (match) { + foundPath = path.join(dir, match); + break; + } + for (const f of files) { + const m = f.match(/^(\d+)\./); + if (m) availableIds.add(m[1]); + } + } + + if (!anyDirExists) { + throw new Error("No artifacts directory found"); + } + + if (!foundPath) { + const sorted = [...availableIds].sort((a, b) => Number(a) - Number(b)); + const availableStr = sorted.length > 0 ? sorted.join(", ") : "none"; throw new Error(`Artifact ${id} not found. Available: ${availableStr}`); } - const filePath = path.join(artifactsDir, match); - const content = await Bun.file(filePath).text(); - + const content = await Bun.file(foundPath).text(); return { url: url.href, content, contentType: "text/plain", size: Buffer.byteLength(content, "utf-8"), - sourcePath: filePath, + sourcePath: foundPath, }; } } diff --git a/packages/coding-agent/src/internal-urls/index.ts b/packages/coding-agent/src/internal-urls/index.ts index f20354538..b4c22f912 100644 --- a/packages/coding-agent/src/internal-urls/index.ts +++ b/packages/coding-agent/src/internal-urls/index.ts @@ -1,28 +1,15 @@ /** - * Internal URL routing system for internal protocols like agent://, memory://, skill://, mcp://, and local://. + * Internal URL routing system for internal protocols like agent://, memory://, + * skill://, mcp://, and local://. * - * This module provides a unified way to resolve internal URLs without - * exposing filesystem paths to the agent. - * - * @example - * ```ts - * import { InternalUrlRouter, AgentProtocolHandler, MemoryProtocolHandler, SkillProtocolHandler } from './internal-urls'; - * - * const router = new InternalUrlRouter(); - * router.register(new AgentProtocolHandler({ getArtifactsDir: () => sessionDir })); - * router.register(new MemoryProtocolHandler({ getMemoryRoot: () => memoryRoot })); - * router.register(new SkillProtocolHandler({ getSkills: () => skills })); - * - * if (router.canHandle('agent://reviewer_0')) { - * const resource = await router.resolve('agent://reviewer_0'); - * console.log(resource.content); - * } - * ``` + * One process-global `InternalUrlRouter` is shared across sessions. Handlers + * are stateless; they pull whatever they need (active skills/rules, active + * MCP/async managers, AgentRegistry-listed sessions) from the owning module + * on each resolve call. */ export * from "./agent-protocol"; export * from "./artifact-protocol"; -export * from "./jobs-protocol"; export * from "./json-query"; export * from "./local-protocol"; export * from "./mcp-protocol"; diff --git a/packages/coding-agent/src/internal-urls/jobs-protocol.ts b/packages/coding-agent/src/internal-urls/jobs-protocol.ts deleted file mode 100644 index f45b02608..000000000 --- a/packages/coding-agent/src/internal-urls/jobs-protocol.ts +++ /dev/null @@ -1,120 +0,0 @@ -import type { AsyncJobManager } from "../async"; -import { formatDuration } from "../tools/render-utils"; -import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; - -export interface JobsProtocolOptions { - getAsyncJobManager: () => AsyncJobManager | undefined; -} - -function formatJobTime(startTime: number): string { - return new Date(startTime).toISOString(); -} - -function formatJobDuration(startTime: number): string { - return formatDuration(Math.max(0, Date.now() - startTime)); -} - -function normalizeJobId(url: InternalUrl): string { - const host = url.rawHost || url.hostname; - const pathname = (url.rawPathname ?? url.pathname).replace(/^\/+/, "").trim(); - if (host && pathname) return `${host}/${pathname}`; - if (host) return host; - return pathname; -} - -export class JobsProtocolHandler implements ProtocolHandler { - readonly scheme = "jobs"; - readonly immutable = true; - - constructor(private readonly options: JobsProtocolOptions) {} - - async resolve(url: InternalUrl): Promise { - const manager = this.options.getAsyncJobManager(); - if (!manager) { - const content = - "# Jobs\n\nBackground job support is disabled. Enable `async.enabled` or `bash.autoBackground.enabled` to use jobs://."; - return { - url: url.href, - content, - contentType: "text/markdown", - size: Buffer.byteLength(content, "utf-8"), - }; - } - - const jobId = normalizeJobId(url); - if (!jobId) { - return this.#listJobs(url, manager); - } - - return this.#getJob(url, manager, jobId); - } - - #listJobs(url: InternalUrl, manager: AsyncJobManager): InternalResource { - const jobs = manager.getAllJobs(); - const running = jobs.filter(job => job.status === "running").sort((a, b) => a.startTime - b.startTime); - const done = jobs.filter(job => job.status !== "running").sort((a, b) => b.startTime - a.startTime); - const ordered = [...running, ...done]; - - if (ordered.length === 0) { - const content = "# Jobs\n\nNo background jobs found."; - return { - url: url.href, - content, - contentType: "text/markdown", - size: Buffer.byteLength(content, "utf-8"), - }; - } - - const lines = ordered.map(job => { - return `- \`${job.id}\` [${job.type}] **${job.status}** — ${job.label} \n started: ${formatJobTime(job.startTime)} · duration: ${formatJobDuration(job.startTime)}`; - }); - const content = `# Jobs\n\n${ordered.length} job${ordered.length === 1 ? "" : "s"}\n\n${lines.join("\n")}`; - return { - url: url.href, - content, - contentType: "text/markdown", - size: Buffer.byteLength(content, "utf-8"), - }; - } - - #getJob(url: InternalUrl, manager: AsyncJobManager, jobId: string): InternalResource { - const job = manager.getJob(jobId); - if (!job) { - const content = `# Job Not Found\n\n404: No async job found with id \`${jobId}\`.`; - return { - url: url.href, - content, - contentType: "text/markdown", - size: Buffer.byteLength(content, "utf-8"), - }; - } - - const sections = [ - `# Job ${job.id}`, - "", - `- type: ${job.type}`, - `- status: ${job.status}`, - `- label: ${job.label}`, - `- start: ${formatJobTime(job.startTime)}`, - `- duration: ${formatJobDuration(job.startTime)}`, - ]; - - if (job.status === "completed" && job.resultText) { - sections.push("", "## Result", "", "```", job.resultText, "```"); - } - if (job.status === "failed" && job.errorText) { - sections.push("", "## Error", "", "```", job.errorText, "```"); - } - if (job.status === "cancelled" && job.errorText) { - sections.push("", "## Cancellation", "", "```", job.errorText, "```"); - } - - const content = sections.join("\n"); - return { - url: url.href, - content, - contentType: "text/markdown", - size: Buffer.byteLength(content, "utf-8"), - }; - } -} diff --git a/packages/coding-agent/src/internal-urls/local-protocol.ts b/packages/coding-agent/src/internal-urls/local-protocol.ts index 972956508..12c9cb94c 100644 --- a/packages/coding-agent/src/internal-urls/local-protocol.ts +++ b/packages/coding-agent/src/internal-urls/local-protocol.ts @@ -2,6 +2,7 @@ import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; import { isEnoent } from "@oh-my-pi/pi-utils"; +import { AgentRegistry } from "../registry/agent-registry"; import { parseInternalUrl } from "./parse"; import { validateRelativePath } from "./skill-protocol"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; @@ -136,17 +137,60 @@ export function resolveLocalUrlToPath(input: string | InternalUrl, options: Loca * Protocol handler for local:// URLs. * * URL forms: - * - local:// - Lists all session local files - * - local:// - Reads a file under session local root + * - local:// - Lists files at the session local root + * - local:// - Reads a file under the session local root */ export class LocalProtocolHandler implements ProtocolHandler { readonly scheme = "local"; readonly immutable = false; - constructor(private readonly options: LocalProtocolOptions) {} + static #override: LocalProtocolOptions | undefined; + + /** + * Install a process-global override that wins over the AgentRegistry-based + * derivation. Used by SDK consumers that wire `localProtocolOptions` on + * `createAgentSession` and by subagents that share their parent's root. + */ + static setOverride(value: LocalProtocolOptions | undefined): void { + LocalProtocolHandler.#override = value; + } + + /** Reset the process-global override. Test-only. */ + static resetOverrideForTests(): void { + LocalProtocolHandler.#override = undefined; + } + + /** + * Returns the active local-protocol options. + * + * Resolution order: + * 1. Explicit override installed via {@link setOverride} (used by subagents + * that share their parent's root and by SDK consumers with a custom + * artifacts/session id mapping). + * 2. The main session in `AgentRegistry.global()`. Its `SessionManager` + * supplies both `getArtifactsDir` and `getSessionId`. + */ + static resolveOptions(): LocalProtocolOptions | undefined { + const override = LocalProtocolHandler.#override; + if (override) return override; + const main = AgentRegistry.global() + .list() + .find(ref => ref.kind === "main"); + const sessionManager = main?.session?.sessionManager; + if (!sessionManager) return undefined; + return { + getArtifactsDir: () => sessionManager.getArtifactsDir(), + getSessionId: () => sessionManager.getSessionId(), + }; + } async resolve(url: InternalUrl): Promise { - const localRoot = path.resolve(resolveLocalRoot(this.options)); + const opts = LocalProtocolHandler.resolveOptions(); + if (!opts) { + throw new Error("No session - local:// unavailable"); + } + + const localRoot = path.resolve(resolveLocalRoot(opts)); await fs.mkdir(localRoot, { recursive: true }); let resolvedRoot: string; @@ -172,9 +216,7 @@ export class LocalProtocolHandler implements ProtocolHandler { const realParent = await fs.realpath(parentDir); ensureWithinRoot(realParent, resolvedRoot); } catch (error) { - if (!isEnoent(error)) { - throw error; - } + if (!isEnoent(error)) throw error; } let realTargetPath: string; diff --git a/packages/coding-agent/src/internal-urls/mcp-protocol.ts b/packages/coding-agent/src/internal-urls/mcp-protocol.ts index fd498415d..f1db2e591 100644 --- a/packages/coding-agent/src/internal-urls/mcp-protocol.ts +++ b/packages/coding-agent/src/internal-urls/mcp-protocol.ts @@ -1,11 +1,7 @@ -import type { MCPManager } from "../mcp/manager"; +import { MCPManager } from "../mcp/manager"; import type { MCPResourceReadResult } from "../mcp/types"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; -export interface McpProtocolOptions { - getMcpManager: () => MCPManager | undefined; -} - function escapeRegex(text: string): string { return text.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); } @@ -108,10 +104,8 @@ export class McpProtocolHandler implements ProtocolHandler { readonly scheme = "mcp"; readonly immutable = true; - constructor(private readonly options: McpProtocolOptions) {} - async resolve(url: InternalUrl): Promise { - const mcpManager = this.options.getMcpManager(); + const mcpManager = MCPManager.instance(); if (!mcpManager) { throw new Error("No MCP manager available. MCP servers may not be configured."); } diff --git a/packages/coding-agent/src/internal-urls/memory-protocol.ts b/packages/coding-agent/src/internal-urls/memory-protocol.ts index 94e7de0b2..88fd13195 100644 --- a/packages/coding-agent/src/internal-urls/memory-protocol.ts +++ b/packages/coding-agent/src/internal-urls/memory-protocol.ts @@ -1,6 +1,8 @@ import * as fs from "node:fs/promises"; import * as path from "node:path"; -import { isEnoent } from "@oh-my-pi/pi-utils"; +import { getAgentDir, isEnoent } from "@oh-my-pi/pi-utils"; +import { getMemoryRoot } from "../memories"; +import { AgentRegistry } from "../registry/agent-registry"; import { validateRelativePath } from "./skill-protocol"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; @@ -8,13 +10,20 @@ const DEFAULT_MEMORY_FILE = "memory_summary.md"; const MEMORY_NAMESPACE = "root"; /** - * Options for the memory:// URL protocol. + * Snapshot of memory roots for every registered session, deduped. + * Each session has its own cwd (possibly a worktree), so subagents and main + * may see different roots. */ -export interface MemoryProtocolOptions { - /** - * Returns the absolute path to the current project's memory root. - */ - getMemoryRoot: () => string; +function memoryRootsFromRegistry(): string[] { + const agentDir = getAgentDir(); + const roots: string[] = []; + for (const ref of AgentRegistry.global().list()) { + const sm = ref.session?.sessionManager; + if (!sm) continue; + const root = getMemoryRoot(agentDir, sm.getCwd()); + if (root && !roots.includes(root)) roots.push(root); + } + return roots; } function ensureWithinRoot(targetPath: string, rootPath: string): void { @@ -61,74 +70,95 @@ export function resolveMemoryUrlToPath(url: InternalUrl, memoryRoot: string): st return path.resolve(memoryRoot, relativePath); } +async function tryResolveInRoot(url: InternalUrl, memoryRoot: string): Promise { + const resolved = path.resolve(memoryRoot); + let resolvedRoot: string; + try { + resolvedRoot = await fs.realpath(resolved); + } catch (error) { + if (isEnoent(error)) return undefined; + throw error; + } + + const targetPath = resolveMemoryUrlToPath(url, resolvedRoot); + ensureWithinRoot(targetPath, resolvedRoot); + + const parentDir = path.dirname(targetPath); + try { + const realParent = await fs.realpath(parentDir); + ensureWithinRoot(realParent, resolvedRoot); + } catch (error) { + if (!isEnoent(error)) throw error; + } + + let realTargetPath: string; + try { + realTargetPath = await fs.realpath(targetPath); + } catch (error) { + if (isEnoent(error)) return undefined; + throw error; + } + + ensureWithinRoot(realTargetPath, resolvedRoot); + + const stat = await fs.stat(realTargetPath); + if (!stat.isFile()) { + throw new Error(`memory:// URL must resolve to a file: ${url.href}`); + } + + const content = await Bun.file(realTargetPath).text(); + const ext = path.extname(realTargetPath).toLowerCase(); + const contentType: InternalResource["contentType"] = ext === ".md" ? "text/markdown" : "text/plain"; + + return { + url: url.href, + content, + contentType, + size: Buffer.byteLength(content, "utf-8"), + sourcePath: realTargetPath, + notes: [], + }; +} + /** * Protocol handler for memory:// URLs. * - * URL forms: - * - memory://root - Reads memory_summary.md - * - memory://root/ - Reads a relative file under memory root + * Walks every active session's memory root. Worktree-based subagents have + * their own root; first one containing the file wins. Parent and subagent + * sharing a cwd see the same file regardless of order. */ export class MemoryProtocolHandler implements ProtocolHandler { readonly scheme = "memory"; readonly immutable = true; - constructor(private readonly options: MemoryProtocolOptions) {} - async resolve(url: InternalUrl): Promise { - const memoryRoot = path.resolve(this.options.getMemoryRoot()); - let resolvedRoot: string; - try { - resolvedRoot = await fs.realpath(memoryRoot); - } catch (error) { - if (isEnoent(error)) { - throw new Error( - "Memory artifacts are not available for this project yet. Run a session with memories enabled first.", - ); - } - throw error; + const roots = memoryRootsFromRegistry(); + + if (roots.length === 0) { + throw new Error( + "Memory artifacts are not available for this project yet. Run a session with memories enabled first.", + ); } - const targetPath = resolveMemoryUrlToPath(url, resolvedRoot); - ensureWithinRoot(targetPath, resolvedRoot); - - const parentDir = path.dirname(targetPath); - try { - const realParent = await fs.realpath(parentDir); - ensureWithinRoot(realParent, resolvedRoot); - } catch (error) { - if (!isEnoent(error)) { + let anyExists = false; + for (const root of roots) { + try { + await fs.stat(root); + anyExists = true; + } catch (error) { + if (isEnoent(error)) continue; throw error; } + const result = await tryResolveInRoot(url, root); + if (result) return result; } - let realTargetPath: string; - try { - realTargetPath = await fs.realpath(targetPath); - } catch (error) { - if (isEnoent(error)) { - throw new Error(`Memory file not found: ${url.href}`); - } - throw error; + if (!anyExists) { + throw new Error( + "Memory artifacts are not available for this project yet. Run a session with memories enabled first.", + ); } - ensureWithinRoot(realTargetPath, resolvedRoot); - - const stat = await fs.stat(realTargetPath); - if (!stat.isFile()) { - throw new Error(`memory:// URL must resolve to a file: ${url.href}`); - } - - const content = await Bun.file(realTargetPath).text(); - const ext = path.extname(realTargetPath).toLowerCase(); - const contentType: InternalResource["contentType"] = ext === ".md" ? "text/markdown" : "text/plain"; - - return { - url: url.href, - content, - contentType, - size: Buffer.byteLength(content, "utf-8"), - sourcePath: realTargetPath, - notes: [], - }; + throw new Error(`Memory file not found: ${url.href}`); } } diff --git a/packages/coding-agent/src/internal-urls/router.ts b/packages/coding-agent/src/internal-urls/router.ts index 717742b57..62c4eac96 100644 --- a/packages/coding-agent/src/internal-urls/router.ts +++ b/packages/coding-agent/src/internal-urls/router.ts @@ -1,42 +1,58 @@ /** * Internal URL router for internal protocols (agent://, artifact://, memory://, skill://, rule://, mcp://, pi://, local://). + * + * One process-global router with one handler per scheme. Access via + * `InternalUrlRouter.instance()`. Handlers are stateless; per-session and + * shared state lives in `./state.ts`. */ +import { AgentProtocolHandler } from "./agent-protocol"; +import { ArtifactProtocolHandler } from "./artifact-protocol"; +import { LocalProtocolHandler } from "./local-protocol"; +import { McpProtocolHandler } from "./mcp-protocol"; +import { MemoryProtocolHandler } from "./memory-protocol"; import { parseInternalUrl } from "./parse"; +import { PiProtocolHandler } from "./pi-protocol"; +import { RuleProtocolHandler } from "./rule-protocol"; +import { SkillProtocolHandler } from "./skill-protocol"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; -/** - * Router for internal URL schemes. - * - * Dispatches URLs like `agent://output_id` or `memory://root/memory_summary.md` to - * registered protocol handlers. - */ export class InternalUrlRouter { + static #instance: InternalUrlRouter | undefined; + #handlers = new Map(); - /** - * Register a protocol handler. - * @param handler Handler to register (uses handler.scheme as key) - */ - register(handler: ProtocolHandler): void { - this.#handlers.set(handler.scheme, handler); + constructor() { + this.register(new PiProtocolHandler()); + this.register(new AgentProtocolHandler()); + this.register(new ArtifactProtocolHandler()); + this.register(new MemoryProtocolHandler()); + this.register(new LocalProtocolHandler()); + this.register(new SkillProtocolHandler()); + this.register(new RuleProtocolHandler()); + this.register(new McpProtocolHandler()); + } + + /** Process-global router instance. */ + static instance(): InternalUrlRouter { + InternalUrlRouter.#instance ??= new InternalUrlRouter(); + return InternalUrlRouter.#instance; + } + + /** Reset the global instance in tests. */ + static resetForTests(): void { + InternalUrlRouter.#instance = undefined; + } + + register(handler: ProtocolHandler): void { + this.#handlers.set(handler.scheme.toLowerCase(), handler); } - /** - * Check if the router can handle a URL. - * @param input URL string to check - */ canHandle(input: string): boolean { const match = input.match(/^([a-z][a-z0-9+.-]*):\/\//i); if (!match) return false; - const scheme = match[1].toLowerCase(); - return this.#handlers.has(scheme); + return this.#handlers.has(match[1].toLowerCase()); } - /** - * Resolve an internal URL to its content. - * @param input URL string (e.g., "agent://reviewer_0", "skill://notion-pages") - * @throws Error if scheme is not registered or resolution fails - */ async resolve(input: string): Promise { const parsed = parseInternalUrl(input); const scheme = parsed.protocol.replace(/:$/, "").toLowerCase(); diff --git a/packages/coding-agent/src/internal-urls/rule-protocol.ts b/packages/coding-agent/src/internal-urls/rule-protocol.ts index f8aea56e1..92f97bb9c 100644 --- a/packages/coding-agent/src/internal-urls/rule-protocol.ts +++ b/packages/coding-agent/src/internal-urls/rule-protocol.ts @@ -1,42 +1,24 @@ /** * Protocol handler for rule:// URLs. * - * Resolves rule names to their content files. - * * URL forms: * - rule:// - Reads rule content */ -import type { Rule } from "../capability/rule"; +import { getActiveRules } from "../capability/rule"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; -export interface RuleProtocolOptions { - /** - * Returns the currently loaded rules. - */ - getRules: () => readonly Rule[]; -} - -/** - * Handler for rule:// URLs. - * - * Resolves rule names to their content. - */ export class RuleProtocolHandler implements ProtocolHandler { readonly scheme = "rule"; readonly immutable = true; - constructor(private readonly options: RuleProtocolOptions) {} - async resolve(url: InternalUrl): Promise { - const rules = this.options.getRules(); + const rules = getActiveRules(); - // Extract rule name from host const ruleName = url.rawHost || url.hostname; if (!ruleName) { throw new Error("rule:// URL requires a rule name: rule://"); } - // Find the rule const rule = rules.find(r => r.name === ruleName); if (!rule) { const available = rules.map(r => r.name); diff --git a/packages/coding-agent/src/internal-urls/skill-protocol.ts b/packages/coding-agent/src/internal-urls/skill-protocol.ts index e506489ad..8bbcdbb72 100644 --- a/packages/coding-agent/src/internal-urls/skill-protocol.ts +++ b/packages/coding-agent/src/internal-urls/skill-protocol.ts @@ -8,19 +8,9 @@ * - skill:/// - Reads relative path within skill's baseDir */ import * as path from "node:path"; -import type { Skill } from "../extensibility/skills"; +import { getActiveSkills } from "../extensibility/skills"; import type { InternalResource, InternalUrl, ProtocolHandler } from "./types"; -export interface SkillProtocolOptions { - /** - * Returns the currently loaded skills. - */ - getSkills: () => readonly Skill[]; -} - -/** - * Get content type based on file extension. - */ function getContentType(filePath: string): InternalResource["contentType"] { const ext = path.extname(filePath).toLowerCase(); if (ext === ".md") return "text/markdown"; @@ -43,25 +33,19 @@ export function validateRelativePath(relativePath: string): void { /** * Handler for skill:// URLs. - * - * Resolves skill names to their content files. */ export class SkillProtocolHandler implements ProtocolHandler { readonly scheme = "skill"; readonly immutable = true; - constructor(private readonly options: SkillProtocolOptions) {} - async resolve(url: InternalUrl): Promise { - const skills = this.options.getSkills(); + const skills = getActiveSkills(); - // Extract skill name from host const skillName = url.rawHost || url.hostname; if (!skillName) { throw new Error("skill:// URL requires a skill name: skill://"); } - // Find the skill const skill = skills.find(s => s.name === skillName); if (!skill) { const available = skills.map(s => s.name); @@ -69,41 +53,34 @@ export class SkillProtocolHandler implements ProtocolHandler { throw new Error(`Unknown skill: ${skillName}\nAvailable: ${availableStr}`); } - // Determine the file to read let targetPath: string; const urlPath = url.pathname; const hasRelativePath = urlPath && urlPath !== "/" && urlPath !== ""; if (hasRelativePath) { - // Read relative path within skill's baseDir - const relativePath = decodeURIComponent(urlPath.slice(1)); // Remove leading / + const relativePath = decodeURIComponent(urlPath.slice(1)); validateRelativePath(relativePath); targetPath = path.join(skill.baseDir, relativePath); - // Verify the resolved path is still within baseDir const resolvedPath = path.resolve(targetPath); const resolvedBaseDir = path.resolve(skill.baseDir); if (!resolvedPath.startsWith(resolvedBaseDir + path.sep) && resolvedPath !== resolvedBaseDir) { throw new Error("Path traversal is not allowed"); } } else { - // Read SKILL.md targetPath = skill.filePath; } - // Read the file const file = Bun.file(targetPath); if (!(await file.exists())) { throw new Error(`File not found: ${targetPath}`); } const content = await file.text(); - const contentType = getContentType(targetPath); - return { url: url.href, content, - contentType, + contentType: getContentType(targetPath), size: Buffer.byteLength(content, "utf-8"), sourcePath: targetPath, notes: [], diff --git a/packages/coding-agent/src/main.ts b/packages/coding-agent/src/main.ts index 80b603ff6..66ccdb598 100644 --- a/packages/coding-agent/src/main.ts +++ b/packages/coding-agent/src/main.ts @@ -765,7 +765,7 @@ export async function runRootCommand(parsed: Args, rawArgs: string[]): Promise(); #tools: CustomTool[] = []; #pendingConnections = new Map>(); diff --git a/packages/coding-agent/src/modes/components/session-observer-overlay.ts b/packages/coding-agent/src/modes/components/session-observer-overlay.ts index e80ed59ae..cf37498d2 100644 --- a/packages/coding-agent/src/modes/components/session-observer-overlay.ts +++ b/packages/coding-agent/src/modes/components/session-observer-overlay.ts @@ -192,7 +192,7 @@ export class SessionObserverOverlayComponent extends Container { const statsLine = this.#buildStatsLine(session); if (statsLine) this.#viewerFooterLines.push(statsLine); this.#viewerFooterLines.push( - theme.fg("dim", "j/k:scroll Enter:expand [/]/\u2190\u2192:cycle agents Esc/Ctrl+S:close g/G:top/bottom"), + theme.fg("dim", "j/k:scroll Enter:expand [/]/←→:cycle agents Esc/Ctrl+S:close g/G:top/bottom"), ); // Auto-scroll to bottom if we were at bottom @@ -452,7 +452,7 @@ export class SessionObserverOverlayComponent extends Container { // Tool call header const intentStr = call.intent ? theme.fg("dim", ` ${sanitizeLine(call.intent, TRUNCATE_LENGTHS.SHORT)}`) : ""; - lines.push(`${cursor} ${theme.fg("accent", "\u25B8")} ${theme.bold(theme.fg("muted", call.name))}${intentStr}`); + lines.push(`${cursor} ${theme.fg("accent", "▸")} ${theme.bold(theme.fg("muted", call.name))}${intentStr}`); // Key arguments const argSummary = this.#formatToolArgs(call.name, call.arguments); diff --git a/packages/coding-agent/src/modes/components/tool-execution.ts b/packages/coding-agent/src/modes/components/tool-execution.ts index 65b954955..515e7f2d8 100644 --- a/packages/coding-agent/src/modes/components/tool-execution.ts +++ b/packages/coding-agent/src/modes/components/tool-execution.ts @@ -665,6 +665,12 @@ export class ToolExecutionComponent extends Container { context.perFileDiffPreview = previews; } } + if (!previews?.some(preview => preview.diff)) { + const editMode = this.#editMode; + const strategy = editMode ? EDIT_MODE_STRATEGIES[editMode] : undefined; + const fallback = strategy?.renderStreamingFallback(this.#args, theme); + if (fallback) context.editStreamingFallback = fallback; + } context.renderDiff = renderDiff; } diff --git a/packages/coding-agent/src/modes/components/tree-selector.ts b/packages/coding-agent/src/modes/components/tree-selector.ts index 9a803749b..2d9693dea 100644 --- a/packages/coding-agent/src/modes/components/tree-selector.ts +++ b/packages/coding-agent/src/modes/components/tree-selector.ts @@ -539,6 +539,10 @@ class TreeList implements Component { const msgWithContent = msg as { content?: unknown }; const content = normalize(this.#extractContent(msgWithContent.content)); result = theme.fg("accent", "user: ") + content; + } else if (role === "developer") { + const msgWithContent = msg as { content?: unknown }; + const content = normalize(this.#extractContent(msgWithContent.content)); + result = theme.fg("dim", "developer: ") + theme.fg("muted", content); } else if (role === "assistant") { const msgWithContent = msg as { content?: unknown; stopReason?: string; errorMessage?: string }; const textContent = normalize(this.#extractContent(msgWithContent.content)); diff --git a/packages/coding-agent/src/modes/controllers/event-controller.ts b/packages/coding-agent/src/modes/controllers/event-controller.ts index a1df5ccec..ab0f4fa2a 100644 --- a/packages/coding-agent/src/modes/controllers/event-controller.ts +++ b/packages/coding-agent/src/modes/controllers/event-controller.ts @@ -1,6 +1,6 @@ import { INTENT_FIELD } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, ImageContent } from "@oh-my-pi/pi-ai"; -import { Loader, TERMINAL, Text } from "@oh-my-pi/pi-tui"; +import { type Component, Loader, TERMINAL, Text } from "@oh-my-pi/pi-tui"; import { settings } from "../../config/settings"; import { AssistantMessageComponent } from "../../modes/components/assistant-message"; import { ReadToolGroupComponent } from "../../modes/components/read-tool-group"; @@ -15,6 +15,8 @@ import type { ExitPlanModeDetails } from "../../tools"; type AgentSessionEventKind = AgentSessionEvent["type"]; +const IRC_MESSAGE_VISIBLE_TTL_MS = 10_000; + type AgentSessionEventHandlers = { [E in AgentSessionEventKind]: (event: Extract) => Promise; }; @@ -29,6 +31,7 @@ export class EventController { #readToolCallAssistantComponents = new Map(); #lastAssistantComponent: AssistantMessageComponent | undefined = undefined; #idleCompactionTimer?: NodeJS.Timeout; + #ircExpiryTimers = new Map(); #handlers: AgentSessionEventHandlers; constructor(private ctx: InteractiveModeContext) { @@ -59,6 +62,10 @@ export class EventController { dispose(): void { this.#cancelIdleCompaction(); + for (const timer of this.#ircExpiryTimers.values()) { + clearTimeout(timer); + } + this.#ircExpiryTimers.clear(); } #resetReadGroup(): void { @@ -222,10 +229,24 @@ export class EventController { } this.#renderedCustomMessages.add(signature); this.#resetReadGroup(); - this.ctx.addMessageToChat(event.message); + const components = this.ctx.addMessageToChat(event.message); + this.#scheduleIrcExpiry(signature, components); this.ctx.ui.requestRender(); } + #scheduleIrcExpiry(signature: string, components: Component[]): void { + if (components.length === 0 || this.#ircExpiryTimers.has(signature)) return; + const timer = setTimeout(() => { + this.#ircExpiryTimers.delete(signature); + for (const component of components) { + this.ctx.chatContainer.removeChild(component); + } + this.ctx.ui.requestRender(); + }, IRC_MESSAGE_VISIBLE_TTL_MS); + timer.unref?.(); + this.#ircExpiryTimers.set(signature, timer); + } + async #handleNotice(event: Extract): Promise { const message = event.source ? `${event.source}: ${event.message}` : event.message; if (event.level === "error") { diff --git a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts index 0896825cc..dc28c7248 100644 --- a/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts +++ b/packages/coding-agent/src/modes/controllers/mcp-command-controller.ts @@ -1238,12 +1238,12 @@ export class MCPCommandController { ? theme.fg("muted", "Connecting") : theme.fg("warning", "Not connected yet"); this.#showMessage( - ["", theme.fg("success", `\u2713 Enabled "${name}"`), "", ` Status: ${status}`, ""].join("\n"), + ["", theme.fg("success", `✓ Enabled "${name}"`), "", ` Status: ${status}`, ""].join("\n"), ); } else { await this.ctx.mcpManager?.disconnectServer(name); await this.ctx.session.refreshMCPTools(this.ctx.mcpManager?.getTools() ?? []); - this.#showMessage(["", theme.fg("success", `\u2713 Disabled "${name}"`), ""].join("\n")); + this.#showMessage(["", theme.fg("success", `✓ Disabled "${name}"`), ""].join("\n")); } return; } @@ -1429,12 +1429,9 @@ export class MCPCommandController { await this.ctx.session.refreshMCPTools(this.ctx.mcpManager.getTools()); const serverTools = this.ctx.mcpManager.getTools().filter(t => t.mcpServerName === name); this.#showMessage( - [ + ["\n", theme.fg("success", `✓ Reconnected to "${name}"`), ` Tools: ${serverTools.length}`, "\n"].join( "\n", - theme.fg("success", `\u2713 Reconnected to "${name}"`), - ` Tools: ${serverTools.length}`, - "\n", - ].join("\n"), + ), ); } else { this.ctx.showError(`Failed to reconnect to "${name}". Check server status and logs.`); @@ -1589,8 +1586,8 @@ export class MCPCommandController { hasAny = true; lines.push(`${theme.fg("accent", name)}:`); - const check = theme.fg("success", "\u2713"); - const cross = theme.fg("dim", "\u2717"); + const check = theme.fg("success", "✓"); + const cross = theme.fg("dim", "✗"); if (supportsToolsChanged) lines.push(` ${check} tools/list_changed`); if (supportsResourcesChanged) lines.push(` ${check} resources/list_changed`); if (supportsPromptsChanged) lines.push(` ${check} prompts/list_changed`); @@ -1607,7 +1604,7 @@ export class MCPCommandController { lines.push(` ${check} resources/subscribe ${subStatus}`); if (enabled && subscribedUris && subscribedUris.size > 0) { for (const uri of subscribedUris) { - lines.push(` ${theme.fg("success", "\u2713")} ${theme.fg("dim", uri)}`); + lines.push(` ${theme.fg("success", "✓")} ${theme.fg("dim", uri)}`); } } } else if (supportsResources) { diff --git a/packages/coding-agent/src/modes/interactive-mode.ts b/packages/coding-agent/src/modes/interactive-mode.ts index c24854ba2..14062cfa1 100644 --- a/packages/coding-agent/src/modes/interactive-mode.ts +++ b/packages/coding-agent/src/modes/interactive-mode.ts @@ -1502,8 +1502,8 @@ export class InteractiveMode implements InteractiveModeContext { return this.#uiHelpers.isKnownSlashCommand(text); } - addMessageToChat(message: AgentMessage, options?: { populateHistory?: boolean }): void { - this.#uiHelpers.addMessageToChat(message, options); + addMessageToChat(message: AgentMessage, options?: { populateHistory?: boolean }): Component[] { + return this.#uiHelpers.addMessageToChat(message, options); } renderSessionContext( diff --git a/packages/coding-agent/src/modes/theme/theme.ts b/packages/coding-agent/src/modes/theme/theme.ts index 490e99df2..9fdebddb6 100644 --- a/packages/coding-agent/src/modes/theme/theme.ts +++ b/packages/coding-agent/src/modes/theme/theme.ts @@ -387,51 +387,51 @@ const NERD_SYMBOLS: SymbolMap = { "nav.back": "\uf060", // Tree Connectors (same as unicode) // pick: ├─ | alt: ├╴ ├╌ ╠═ ┣━ - "tree.branch": "\u251c\u2500", + "tree.branch": "├─", // pick: └─ | alt: └╴ └╌ ╚═ ┗━ - "tree.last": "\u2514\u2500", + "tree.last": "└─", // pick: │ | alt: ┃ ║ ▏ ▕ - "tree.vertical": "\u2502", + "tree.vertical": "│", // pick: ─ | alt: ━ ═ ╌ ┄ - "tree.horizontal": "\u2500", + "tree.horizontal": "─", // pick: └ | alt: ╰ ⎿ ↳ - "tree.hook": "\u2514", + "tree.hook": "└", // Box Drawing - Rounded (same as unicode) // pick: ╭ | alt: ┌ ┏ ╔ - "boxRound.topLeft": "\u256d", + "boxRound.topLeft": "╭", // pick: ╮ | alt: ┐ ┓ ╗ - "boxRound.topRight": "\u256e", + "boxRound.topRight": "╮", // pick: ╰ | alt: └ ┗ ╚ - "boxRound.bottomLeft": "\u2570", + "boxRound.bottomLeft": "╰", // pick: ╯ | alt: ┘ ┛ ╝ - "boxRound.bottomRight": "\u256f", + "boxRound.bottomRight": "╯", // pick: ─ | alt: ━ ═ ╌ - "boxRound.horizontal": "\u2500", + "boxRound.horizontal": "─", // pick: │ | alt: ┃ ║ ▏ - "boxRound.vertical": "\u2502", + "boxRound.vertical": "│", // Box Drawing - Sharp (same as unicode) // pick: ┌ | alt: ┏ ╭ ╔ - "boxSharp.topLeft": "\u250c", + "boxSharp.topLeft": "┌", // pick: ┐ | alt: ┓ ╮ ╗ - "boxSharp.topRight": "\u2510", + "boxSharp.topRight": "┐", // pick: └ | alt: ┗ ╰ ╚ - "boxSharp.bottomLeft": "\u2514", + "boxSharp.bottomLeft": "└", // pick: ┘ | alt: ┛ ╯ ╝ - "boxSharp.bottomRight": "\u2518", + "boxSharp.bottomRight": "┘", // pick: ─ | alt: ━ ═ ╌ - "boxSharp.horizontal": "\u2500", + "boxSharp.horizontal": "─", // pick: │ | alt: ┃ ║ ▏ - "boxSharp.vertical": "\u2502", + "boxSharp.vertical": "│", // pick: ┼ | alt: ╋ ╬ ┿ - "boxSharp.cross": "\u253c", + "boxSharp.cross": "┼", // pick: ┬ | alt: ╦ ┯ ┳ - "boxSharp.teeDown": "\u252c", + "boxSharp.teeDown": "┬", // pick: ┴ | alt: ╩ ┷ ┻ - "boxSharp.teeUp": "\u2534", + "boxSharp.teeUp": "┴", // pick: ├ | alt: ╠ ┝ ┣ - "boxSharp.teeRight": "\u251c", + "boxSharp.teeRight": "├", // pick: ┤ | alt: ╣ ┥ ┫ - "boxSharp.teeLeft": "\u2524", + "boxSharp.teeLeft": "┤", // Separators - Nerd Font specific // pick:  | alt:    "sep.powerline": "\ue0b0", @@ -446,7 +446,7 @@ const NERD_SYMBOLS: SymbolMap = { // pick:  | alt:  "sep.powerlineThinRight": "\ue0b3", // pick: █ | alt: ▓ ▒ ░ ▉ ▌ - "sep.block": "\u2588", + "sep.block": "█", // pick: space | alt: ␠ · "sep.space": " ", // pick: > | alt: › » ▸ @@ -454,7 +454,7 @@ const NERD_SYMBOLS: SymbolMap = { // pick: < | alt: ‹ « ◂ "sep.asciiRight": "<", // pick: · | alt: • ⋅ - "sep.dot": " \u00b7 ", + "sep.dot": " · ", // pick:  | alt: / ∕ ⁄ "sep.slash": "\ue0bb", // pick:  | alt: │ ┃ | @@ -545,16 +545,16 @@ const NERD_SYMBOLS: SymbolMap = { // pick:  | alt:   • "format.bullet": "\uf111", // pick: – | alt: — ― - - "format.dash": "\u2013", + "format.dash": "–", // pick: ⟨ | alt: [ ⟦ "format.bracketLeft": "⟨", // pick: ⟩ | alt: ] ⟧ "format.bracketRight": "⟩", // Markdown-specific // pick: │ | alt: ┃ ║ - "md.quoteBorder": "\u2502", + "md.quoteBorder": "│", // pick: ─ | alt: ━ ═ - "md.hrChar": "\u2500", + "md.hrChar": "─", // pick:  | alt:  • "md.bullet": "\uf111", // Language icons (nerd font devicons) diff --git a/packages/coding-agent/src/modes/types.ts b/packages/coding-agent/src/modes/types.ts index 39c8c386d..bc5ecb759 100644 --- a/packages/coding-agent/src/modes/types.ts +++ b/packages/coding-agent/src/modes/types.ts @@ -170,7 +170,7 @@ export interface InteractiveModeContext { */ withLocalSubmission(text: string, fn: () => Promise, options?: { imageCount?: number }): Promise; isKnownSlashCommand(text: string): boolean; - addMessageToChat(message: AgentMessage, options?: { populateHistory?: boolean }): void; + addMessageToChat(message: AgentMessage, options?: { populateHistory?: boolean }): Component[]; renderSessionContext( sessionContext: SessionContext, options?: { updateFooter?: boolean; populateHistory?: boolean }, diff --git a/packages/coding-agent/src/modes/utils/ui-helpers.ts b/packages/coding-agent/src/modes/utils/ui-helpers.ts index bc6a748b6..bdcc5972f 100644 --- a/packages/coding-agent/src/modes/utils/ui-helpers.ts +++ b/packages/coding-agent/src/modes/utils/ui-helpers.ts @@ -1,6 +1,6 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; import type { AssistantMessage, ImageContent, Message } from "@oh-my-pi/pi-ai"; -import { Spacer, Text, TruncatedText } from "@oh-my-pi/pi-tui"; +import { type Component, Spacer, Text, TruncatedText } from "@oh-my-pi/pi-tui"; import { settings } from "../../config/settings"; import { AssistantMessageComponent } from "../../modes/components/assistant-message"; import { BashExecutionComponent } from "../../modes/components/bash-execution"; @@ -70,7 +70,7 @@ export class UiHelpers { this.ctx.ui.requestRender(); } - addMessageToChat(message: AgentMessage, options?: { populateHistory?: boolean }): void { + addMessageToChat(message: AgentMessage, options?: { populateHistory?: boolean }): Component[] { switch (message.role) { case "bashExecution": { const component = new BashExecutionComponent(message.command, this.ctx.ui, message.excludeFromContext); @@ -147,26 +147,30 @@ export class UiHelpers { if (message.customType === "irc:incoming") { const peer = details?.from ?? "?"; body = details?.message ?? ""; - arrow = `\u21e6 ${peer}`; + arrow = `⇦ ${peer}`; } else if (message.customType === "irc:autoreply") { const peer = details?.to ?? "?"; body = details?.reply ?? ""; - arrow = `\u21e8 ${peer} (auto)`; + arrow = `⇨ ${peer}`; } else { const from = details?.from ?? "?"; const to = details?.to ?? "?"; body = details?.body ?? ""; - const suffix = details?.kind === "reply" ? " (auto)" : ""; - arrow = `${from} \u21e8 ${to}${suffix}`; + arrow = `${from} ⇨ ${to}`; } + const components: Component[] = []; const header = `${theme.fg("accent", `[IRC] ${arrow}`)}`; - this.ctx.chatContainer.addChild(new Text(header, 1, 0)); + const headerComponent = new Text(header, 1, 0); + this.ctx.chatContainer.addChild(headerComponent); + components.push(headerComponent); if (body) { for (const line of body.split("\n")) { - this.ctx.chatContainer.addChild(new Text(theme.fg("muted", ` ${line}`), 0, 0)); + const lineComponent = new Text(theme.fg("muted", ` ${line}`), 0, 0); + this.ctx.chatContainer.addChild(lineComponent); + components.push(lineComponent); } } - break; + return components; } const renderer = this.ctx.session.extensionRunner?.getMessageRenderer(message.customType); // Both HookMessage and CustomMessage have the same structure, cast for compatibility @@ -240,6 +244,7 @@ export class UiHelpers { const _exhaustive: never = message; } } + return []; } /** diff --git a/packages/coding-agent/src/prompts/commands/orchestrate.md b/packages/coding-agent/src/prompts/commands/orchestrate.md index 28631e3a7..20092bbbd 100644 --- a/packages/coding-agent/src/prompts/commands/orchestrate.md +++ b/packages/coding-agent/src/prompts/commands/orchestrate.md @@ -26,6 +26,7 @@ You decompose, dispatch, verify, and iterate. You do **not** edit code. Every fi 6. **Commit policy.** If the task asks for commits or the repo workflow expects them, commit after each green phase with a focused message. Never commit a red tree. Never commit work the user did not ask to commit. 7. **Respawn, do not absorb.** If a subagent returns incomplete or wrong work, spawn a corrective subagent with the specific gap — do not silently fix it yourself. 8. **No scope creep, no scope shrink.** Do not add work the user did not ask for. Do not relabel unfinished items as "follow-up", "v1", or "MVP" to imply completion. +9. **Subagents do not verify, lint, or format.** Every `task` assignment **MUST** instruct the subagent to skip all gates and formatters. Their job is the edit only. You — the orchestrator — run verification and formatting **once** at the end of the phase across the union of changed files. Avoids redundant runs and racing formatter passes. diff --git a/packages/coding-agent/src/prompts/system/now-prompt.md b/packages/coding-agent/src/prompts/system/now-prompt.md deleted file mode 100644 index 52028be64..000000000 --- a/packages/coding-agent/src/prompts/system/now-prompt.md +++ /dev/null @@ -1,7 +0,0 @@ -Today is {{date}}, and the current working directory is '{{cwd}}'. - - -- Each response **MUST** advance the task. There is no stopping condition other than completion. -- You **MUST** default to informed action; do not ask for confirmation when tools or repo context can answer. -- You **MUST** verify the effect of significant behavioral changes before yielding: run the specific test, command, or scenario that covers your change. - diff --git a/packages/coding-agent/src/prompts/system/project-prompt.md b/packages/coding-agent/src/prompts/system/project-prompt.md index 2390737d5..cbea49461 100644 --- a/packages/coding-agent/src/prompts/system/project-prompt.md +++ b/packages/coding-agent/src/prompts/system/project-prompt.md @@ -32,6 +32,14 @@ Working directory layout (sorted by mtime, recent first; depth ≤ 3): {{/if}} +Today is {{date}}, and the current working directory is '{{cwd}}'. + + +- Each response **MUST** advance the task. There is no stopping condition other than completion. +- You **MUST** default to informed action; do not ask for confirmation when tools or repo context can answer. +- You **MUST** verify the effect of significant behavioral changes before yielding: run the specific test, command, or scenario that covers your change. + + {{#if appendPrompt}} {{appendPrompt}} {{/if}} diff --git a/packages/coding-agent/src/prompts/system/system-prompt.md b/packages/coding-agent/src/prompts/system/system-prompt.md index 5237704c9..01d072f4b 100644 --- a/packages/coding-agent/src/prompts/system/system-prompt.md +++ b/packages/coding-agent/src/prompts/system/system-prompt.md @@ -37,6 +37,13 @@ Assumptions you didn't validate: incidents to debug. - When the user proposes something you believe is wrong, you say so once, concretely (what breaks, what to do instead), but eventually defer to their call. You **MUST NOT** relitigate. + +- You **MUST NOT** narrate about or even consider, session limits, token/tool budgets, effort estimates, or how much of the task you think you can finish. These are not your concern: + - Even if it was true, start, as if it was not. It's the only way to make progress. + - Execute the work or delegate it. +- You **MUST NOT** speculate about scope inflation ("this is actually a multi-week effort"). You have no comprehension of time, so stop pretending. + + [ENV] You operate within the Oh My Pi coding harness. - Given a task, you **MUST** complete it using the tools available to you. @@ -53,7 +60,6 @@ With most FS/bash-like tools, static references to them will automatically resol - `/`: JSON field extraction - `artifact://`: Artifact content - `local://.md`: Plan artifacts and shared content with subagents -- `jobs://`: Job status and result - `mcp://`: MCP resource - `pi://`: Harness documentation; do **NOT** read unless user mentions the harness itself diff --git a/packages/coding-agent/src/prompts/tools/ask.md b/packages/coding-agent/src/prompts/tools/ask.md index 8cf5bbbd6..9620922bb 100644 --- a/packages/coding-agent/src/prompts/tools/ask.md +++ b/packages/coding-agent/src/prompts/tools/ask.md @@ -8,7 +8,6 @@ Asks user when you need clarification or input during task execution. - Use `recommended: ` to mark default (0-indexed); " (Recommended)" added automatically - Use `questions` for multiple related questions instead of asking one at a time - Set `multi: true` on question to allow multiple selections -- `ask.timeout` only applies while choosing options; once the user selects "Other (type your own)", there is no timeout diff --git a/packages/coding-agent/src/prompts/tools/bash.md b/packages/coding-agent/src/prompts/tools/bash.md index f30f50662..bcdece577 100644 --- a/packages/coding-agent/src/prompts/tools/bash.md +++ b/packages/coding-agent/src/prompts/tools/bash.md @@ -10,16 +10,6 @@ Executes bash command in shell session for terminal operations like git, bun, ca {{#if asyncEnabled}} - Use `async: true` for long-running commands when you don't need immediate output; the call returns a background job ID and the result is delivered automatically as a follow-up. {{/if}} -{{#if autoBackgroundEnabled}} -- Long-running non-PTY commands may auto-background after ~{{autoBackgroundThresholdSeconds}}s and continue as background jobs. -{{/if}} -{{#if asyncEnabled}} -- Inspect background jobs with `read jobs://` (`read jobs://` for detail). To wait for results, call `job` (with `poll`) — do NOT poll `read jobs://` in a loop or yield and hope for delivery. -{{else}} -{{#if autoBackgroundEnabled}} -- For auto-backgrounded jobs, inspect with `read jobs://` and call `job` (with `poll`) to wait — do NOT poll in a loop. -{{/if}} -{{/if}} diff --git a/packages/coding-agent/src/prompts/tools/eval.md b/packages/coding-agent/src/prompts/tools/eval.md index 973a1dc97..b9c291124 100644 --- a/packages/coding-agent/src/prompts/tools/eval.md +++ b/packages/coding-agent/src/prompts/tools/eval.md @@ -46,8 +46,6 @@ tree(path?=".", max_depth?=3, show_hidden?=False) → str Render a directory tree. diff(a, b) → str Unified diff between two files. -run(cmd, cwd?=None, timeout?=None) → {stdout, stderr, exit_code} - Run a shell command. env(key?=None, value?=None) → str | None | dict No args → full environment as dict. One arg → value of `key`. Two args → set `key=value` and return value. output(*ids, format?="raw", query?=None, offset?=None, limit?=None) → str | dict | list[dict] @@ -63,7 +61,7 @@ Cells render like a Jupyter notebook. `display(value)` renders non-presentable d - In session mode, use `*** Reset` on a cell to wipe its language's kernel before running.{{#ifAll py js}} Reset is per-language: a python cell's `*** Reset` does not touch the JavaScript kernel and vice versa.{{/ifAll}} -{{#if js}}- **js**: the VM exposes a selective `process` subset, Web APIs, `Buffer`, `fs/promises`. +{{#if js}}- **js**: the VM exposes a selective `process` subset, Web APIs, `Buffer`, `fs/promises`, and the `Bun` global. {{/if}} diff --git a/packages/coding-agent/src/prompts/tools/github.md b/packages/coding-agent/src/prompts/tools/github.md index cebdb8c00..575d3c399 100644 --- a/packages/coding-agent/src/prompts/tools/github.md +++ b/packages/coding-agent/src/prompts/tools/github.md @@ -9,11 +9,12 @@ Pick the operation via `op`. Each op uses a subset of the parameters: - `pr_diff` — Read one or more pull request diffs. Optional `pr` (single identifier or array for batch). Optional `repo`. Set `nameOnly: true` for changed file names. Use `exclude` to drop generated paths from the diff. - `pr_checkout` — Check one or more pull requests out into dedicated git worktrees. Optional `pr` (number, URL, branch, or array of any of those — pass an array to batch-check-out multiple PRs in one call), `repo`, `force` (reset existing local branch). - `pr_push` — Push a checked-out PR branch back to its source branch. Requires the branch to have been checked out via `op: pr_checkout` (carries push metadata). Optional `branch`; defaults to the current checked-out git branch. Optional `forceWithLease`. -- `search_issues` — Search issues using normal GitHub issue search syntax. Required `query`. Optional `repo`, `limit`. -- `search_prs` — Search pull requests using normal GitHub PR search syntax. Required `query`. Optional `repo`, `limit`. -- `search_code` — Search code with GitHub code search syntax. Required `query`. Optional `repo`, `limit`. Returns matching paths with surrounding fragments. -- `search_commits` — Search commits across GitHub. Required `query`. Optional `repo`, `limit`. Returns short SHA, author, and the first line of each commit message. -- `search_repos` — Search repositories across GitHub. Required `query`. Optional `limit` (use query qualifiers like `org:`, `language:` instead of `repo`). +- `search_issues` — Search issues using normal GitHub issue search syntax. Optional `query` (required unless `since`/`until` is set), `repo`, `limit`, `since`, `until`, `dateField`. +- `search_prs` — Search pull requests using normal GitHub PR search syntax. Optional `query` (required unless `since`/`until` is set), `repo`, `limit`, `since`, `until`, `dateField`. +- `search_code` — Search code with GitHub code search syntax. Required `query`. Optional `repo`, `limit`. Returns matching paths with surrounding fragments. Date filtering (`since`/`until`) is **not** supported by GitHub code search. +- `search_commits` — Search commits across GitHub. Optional `query` (required unless `since`/`until` is set), `repo`, `limit`, `since`, `until`. `dateField` is ignored — always uses `committer-date`. +- `search_repos` — Search repositories across GitHub. Optional `query` (required unless `since`/`until` is set), `limit`, `since`, `until`, `dateField` (use query qualifiers like `org:`, `language:` instead of `repo`). +- Date filter format for `since` / `until`: relative duration `` (`m`/`h`/`d`/`w`/`mo`/`y`, e.g. `3d`, `12h`, `2w`), an ISO date `YYYY-MM-DD`, or an ISO datetime. Translated to a single GitHub-search qualifier (`created:≥…`, `created:≤…`, or `created:since..until`). `dateField: "updated"` maps to `updated:` for issues/prs and `pushed:` for repos. When you only want a date filter and no keywords, omit `query` entirely. - `run_watch` — Watch a GitHub Actions workflow run. Optional `run` (id or URL). Omitting `run` watches all workflow runs for the current HEAD commit; `branch` falls back to the current branch. Optional `tail` (log lines per failed job). Streams snapshots, fast-fails on the first detected job failure (with a brief grace period to capture concurrent failures), then fetches tailed logs for the failed jobs. The full failed-job logs are saved as a session artifact for on-demand reads. diff --git a/packages/coding-agent/src/prompts/tools/hashline.md b/packages/coding-agent/src/prompts/tools/hashline.md index db3fcabc4..62068c0f6 100644 --- a/packages/coding-agent/src/prompts/tools/hashline.md +++ b/packages/coding-agent/src/prompts/tools/hashline.md @@ -17,6 +17,7 @@ Purely textual format. The tool has NO awareness of language, indentation, brack - Every line of inserted/replacement content **MUST** be emitted as a payload line starting with `{{hsep}}`. - `{{hsep}}` is syntax, not content. The inserted text begins after the first `{{hsep}}`; use a bare `{{hsep}}` to insert a blank line. +- Payload is verbatim — don't escape unicode (write `—`, not `\u2014`). - `< A` inserts before line A; `+ A` inserts after line A. `< BOF` / `+ BOF` both prepend; `< EOF` / `+ EOF` both append. - `= A..B` replaces the inclusive range with the following payload lines. `= A..B` with no payload blanks the range to a single empty line. - `- A..B` deletes the inclusive range; `A..A` for one line. diff --git a/packages/coding-agent/src/prompts/tools/job.md b/packages/coding-agent/src/prompts/tools/job.md index 561e13094..6c5bcef07 100644 --- a/packages/coding-agent/src/prompts/tools/job.md +++ b/packages/coding-agent/src/prompts/tools/job.md @@ -1,11 +1,19 @@ -Manages background jobs: poll to wait for completion, cancel to stop running jobs. +Inspects, waits, or cancels async jobs. -You **MUST** use the `job` tool (in a loop, if necessary) instead of manually reading in a loop or issuing sleep commands. +Background job results are delivered automatically when complete. Reach for this tool only when you need to intervene. -Pass `poll` to wait for one or more background jobs to finalize. If the timeout elapses before any job changes state, it returns the current snapshot (still-running jobs and any already-completed deliveries) without erroring — call `job` again to keep waiting. Calling with no `poll` and no `cancel` waits on every running background job. +# Operations -You **MUST NOT** poll the same job repeatedly without evidence of progress. Between calls, inspect `read jobs://` to confirm new output or activity. If a job is stalled, has hung, or is producing nothing useful, cancel it via `cancel` and try a different approach instead of waiting indefinitely. +## `list: true` +Use to inspect what's running. -Pass `cancel` to stop one or more running background jobs (started via async tool execution or bash auto-backgrounding). You **SHOULD** cancel jobs that are no longer needed or stuck. You **MAY** inspect jobs first with `read jobs://` or `read jobs://`. +## `poll: [id, …]` +Block until the specified jobs finish or the wait window elapses. +- Use when you are genuinely blocked on a result and have no other work to do. +- Returns the current snapshot when the timer elapses; running jobs remain running. +- Completed jobs include their final output in the returned snapshot. -`poll` and `cancel` may be combined in a single call: cancellations apply first, then polling waits on the remaining ids. When only `cancel` is provided the call returns immediately without waiting. +## `cancel: [id, …]` +Stop running jobs. +- Use when a job is stalled, hung, or no longer needed. +- Returns immediately after cancelling. diff --git a/packages/coding-agent/src/prompts/tools/task.md b/packages/coding-agent/src/prompts/tools/task.md index da6eeb251..944a14609 100644 --- a/packages/coding-agent/src/prompts/tools/task.md +++ b/packages/coding-agent/src/prompts/tools/task.md @@ -1,11 +1,16 @@ Launches subagents to parallelize workflows. {{#if asyncEnabled}} -- `read jobs://` for state, `read jobs://` for detail. -- Use `job` (with `poll`) to wait. **MUST NOT** poll `read jobs://` in a loop. +- Results are delivered automatically when complete. +- If genuinely blocked on task completion, wait with `job` using `poll`; otherwise continue with another task when possible. +- Call `job` with `list: true` to snapshot manager state; pass `poll: [id]` to wait or `cancel: [id]` to stop \u2014 only when inspection or intervention is useful. {{/if}} -Subagents have no conversation history. Every fact, file path, and decision they need **MUST** be explicit in {{#if contextEnabled}}`context` or `assignment`{{else}}each `assignment`{{/if}}. +{{#if ircEnabled}} +Subagents have no conversation history, but they can reach you and their siblings live via the `irc` tool. Front-load every fact, file path, and direction they need in {{#if contextEnabled}}`context` or `assignment`{{else}}each `assignment`{{/if}}. +{{else}} +Subagents have no conversation history. Every fact, file path, and direction they need **MUST** be explicit in {{#if contextEnabled}}`context` or `assignment`{{else}}each `assignment`{{/if}}. +{{/if}} - `agent`: agent type for all tasks @@ -20,16 +25,28 @@ Subagents have no conversation history. Every fact, file path, and decision they - **MUST NOT** assign tasks to run project-wide build/test/lint. Caller verifies after the batch. +- **Subagents do not verify, lint, or format.** Every assignment **MUST** instruct the subagent to skip all gates and formatters. You run them once at the end across the union of changed files — avoids redundant runs and racing formatter passes. +{{#if ircEnabled}} +- Each task: ≤3–5 explicit files. Overlapping file sets are tolerable when peers can coordinate via `irc`, but still fan out to a cluster when the scopes are cleanly separable. +- No globs, no "update all", no package-wide scope. +{{else}} - Each task: ≤3–5 explicit files. No globs, no "update all", no package-wide scope. Fan out to a cluster instead. +{{/if}} - Pass large payloads via `local://` URIs, not inline. {{#if contextEnabled}}- Put shared constraints in `context` once; do not duplicate across assignments.{{/if}} - Prefer agents that investigate **and** edit in one pass; only spin a read-only discovery step when affected files are genuinely unknown. +{{#if ircEnabled}} +Test: can task B run correctly without seeing A's output? If no, sequence A → B — **unless** B can reasonably ask A for the missing piece over `irc`. Live coordination beats a serial waterfall when the contract is small and easy to describe in a DM. +Still sequence when one task produces a large, evolving contract (generated types, schema migration, core module API) the other consumes wholesale — IRC round-trips do not replace a finished artifact. +Parallel when tasks touch disjoint files, are independent refactors/tests, or only need occasional clarification that can be resolved peer-to-peer. +{{else}} Test: can task B run correctly without seeing A's output? If no, sequence A → B. Sequential when one task produces a contract (types, API, schema, core module) the other consumes. Parallel when tasks touch disjoint files or are independent refactors/tests. +{{/if}} {{#if contextEnabled}} diff --git a/packages/coding-agent/src/sdk.ts b/packages/coding-agent/src/sdk.ts index e1724f75f..3a356e246 100644 --- a/packages/coding-agent/src/sdk.ts +++ b/packages/coding-agent/src/sdk.ts @@ -27,7 +27,7 @@ import chalk from "chalk"; import { AsyncJobManager, isBackgroundJobSupportEnabled } from "./async"; import { createAutoresearchExtension } from "./autoresearch"; import { loadCapability } from "./capability"; -import { type Rule, ruleCapability } from "./capability/rule"; +import { type Rule, ruleCapability, setActiveRules } from "./capability/rule"; import { ModelRegistry } from "./config/model-registry"; import { formatModelString, parseModelPattern, parseModelString, resolveModelRoleValue } from "./config/model-resolver"; import { loadPromptTemplates as loadPromptTemplatesInternal, type PromptTemplate } from "./config/prompt-templates"; @@ -59,30 +59,22 @@ import { type ToolDefinition, wrapRegisteredTools, } from "./extensibility/extensions"; -import { loadSkills as loadSkillsInternal, type Skill, type SkillWarning } from "./extensibility/skills"; +import { + loadSkills as loadSkillsInternal, + type Skill, + type SkillWarning, + setActiveSkills, +} from "./extensibility/skills"; import { type FileSlashCommand, loadSlashCommands as loadSlashCommandsInternal } from "./extensibility/slash-commands"; import type { HindsightSessionState } from "./hindsight/state"; -import { - AgentProtocolHandler, - ArtifactProtocolHandler, - InternalUrlRouter, - JobsProtocolHandler, - LocalProtocolHandler, - type LocalProtocolOptions, - McpProtocolHandler, - MemoryProtocolHandler, - PiProtocolHandler, - RuleProtocolHandler, - SkillProtocolHandler, -} from "./internal-urls"; +import { LocalProtocolHandler, type LocalProtocolOptions } from "./internal-urls"; import { LSP_STARTUP_EVENT_CHANNEL, type LspStartupEvent } from "./lsp/startup-events"; -import { discoverAndLoadMCPTools, type MCPManager, type MCPToolsLoadResult } from "./mcp"; +import { discoverAndLoadMCPTools, MCPManager, type MCPToolsLoadResult } from "./mcp"; import { collectDiscoverableMCPTools, formatDiscoverableMCPToolServerSummary, selectDiscoverableMCPToolNamesByServer, } from "./mcp/discoverable-tool-metadata"; -import { getMemoryRoot } from "./memories"; import { resolveMemoryBackend } from "./memory-backend"; import asyncResultTemplate from "./prompts/tools/async-result.md" with { type: "text" }; import { AgentRegistry, MAIN_AGENT_ID } from "./registry/agent-registry"; @@ -943,34 +935,39 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} return preview; }; - const asyncJobManager = backgroundJobsEnabled - ? new AsyncJobManager({ - maxRunningJobs: asyncMaxJobs, - onJobComplete: async (jobId, result, job) => { - if (!session || asyncJobManager!.isDeliverySuppressed(jobId)) return; - const formattedResult = await formatAsyncResultForFollowUp(result); - if (asyncJobManager!.isDeliverySuppressed(jobId)) return; + // Only top-level sessions own an AsyncJobManager. Subagents reach the + // parent's manager via `AsyncJobManager.instance()` (set below), so creating + // a second instance here just to leave it orphaned wastes a constructor and + // risks accidental disposal of the parent's manager on subagent teardown. + const asyncJobManager = + backgroundJobsEnabled && !options.parentTaskPrefix + ? new AsyncJobManager({ + maxRunningJobs: asyncMaxJobs, + onJobComplete: async (jobId, result, job) => { + if (!session || asyncJobManager!.isDeliverySuppressed(jobId)) return; + const formattedResult = await formatAsyncResultForFollowUp(result); + if (asyncJobManager!.isDeliverySuppressed(jobId)) return; - const message = prompt.render(asyncResultTemplate, { jobId, result: formattedResult }); - const durationMs = job ? Math.max(0, Date.now() - job.startTime) : undefined; - await session.sendCustomMessage( - { - customType: "async-result", - content: message, - display: true, - attribution: "agent", - details: { - jobId, - type: job?.type, - label: job?.label, - durationMs, + const message = prompt.render(asyncResultTemplate, { jobId, result: formattedResult }); + const durationMs = job ? Math.max(0, Date.now() - job.startTime) : undefined; + await session.sendCustomMessage( + { + customType: "async-result", + content: message, + display: true, + attribution: "agent", + details: { + jobId, + type: job?.type, + label: job?.label, + durationMs, + }, }, - }, - { deliverAs: "followUp", triggerTurn: true }, - ); - }, - }) - : undefined; + { deliverAs: "followUp", triggerTurn: true }, + ); + }, + }) + : undefined; const agentRegistry = options.agentRegistry ?? AgentRegistry.global(); const resolvedAgentId = options.agentId ?? options.parentTaskPrefix ?? MAIN_AGENT_ID; @@ -1056,44 +1053,27 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} return {}; } }, + getArtifactManager: () => sessionManager.getArtifactManager(), settings, authStorage, modelRegistry, - asyncJobManager, }; - // Initialize internal URL router for internal protocols (agent://, artifact://, memory://, skill://, rule://, mcp://, local://) - const internalRouter = new InternalUrlRouter(); + // Wire process-wide internal URL singletons owned by their real classes. + // Top-level sessions install the active snapshots; subagents inherit them. + // Artifact and agent-output URLs resolve via `AgentRegistry.global()` — + // the protocol handlers walk each ref's `sessionManager.getArtifactsDir()`, + // which collapses to the parent's dir for subagents (they adopt the + // parent's ArtifactManager) so one lookup hits everything. const getArtifactsDir = () => sessionManager.getArtifactsDir(); - internalRouter.register(new AgentProtocolHandler({ getArtifactsDir })); - internalRouter.register(new ArtifactProtocolHandler({ getArtifactsDir })); - internalRouter.register( - new MemoryProtocolHandler({ - getMemoryRoot: () => getMemoryRoot(agentDir, settings.getCwd()), - }), - ); - internalRouter.register( - new LocalProtocolHandler( - options.localProtocolOptions ?? { - getArtifactsDir, - getSessionId: () => sessionManager.getSessionId(), - }, - ), - ); - internalRouter.register( - new SkillProtocolHandler({ - getSkills: () => skills, - }), - ); - internalRouter.register( - new RuleProtocolHandler({ - getRules: () => [...rulebookRules, ...alwaysApplyRules], - }), - ); - internalRouter.register(new PiProtocolHandler()); - internalRouter.register(new JobsProtocolHandler({ getAsyncJobManager: () => asyncJobManager })); - internalRouter.register(new McpProtocolHandler({ getMcpManager: () => mcpManager })); - toolSession.internalRouter = internalRouter; + if (!options.parentTaskPrefix) { + setActiveSkills(skills); + setActiveRules([...rulebookRules, ...alwaysApplyRules]); + if (asyncJobManager) AsyncJobManager.setInstance(asyncJobManager); + } + if (options.localProtocolOptions) { + LocalProtocolHandler.setOverride(options.localProtocolOptions); + } toolSession.getArtifactsDir = getArtifactsDir; toolSession.agentOutputManager = new AgentOutputManager( getArtifactsDir, @@ -1142,7 +1122,11 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} customTools.push(...mcpResult.tools.map(loaded => loaded.tool)); } } - toolSession.mcpManager = mcpManager; + // Only top-level sessions own the global MCPManager. Subagents already + // receive the parent's manager via `options.mcpManager`, and reassigning + // the singleton to the same value is a no-op \u2014 keep the gate explicit + // to mirror the AsyncJobManager ownership rule. + if (mcpManager && !options.parentTaskPrefix) MCPManager.setInstance(mcpManager); // Add image tools when the active model or configured image providers can generate images. const imageGenTools = await logger.time("getImageGenTools", () => getImageGenTools(modelRegistry, model)); @@ -1724,6 +1708,11 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} sessionManager, settings, evalKernelOwnerId, + // Defined only for top-level sessions (creation is gated above). + // AgentSession uses this to decide whether it may dispose the global + // AsyncJobManager on teardown; subagents inherit the parent's and + // **MUST NOT** tear it down. + ownedAsyncJobManager: asyncJobManager, scopedModels: options.scopedModels, promptTemplates, slashCommands, @@ -1760,7 +1749,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {} defaultSelectedMCPServerNames: [...discoveryDefaultServers], ttsrManager, obfuscator, - asyncJobManager, agentId: resolvedAgentId, agentRegistry, providerSessionId: options.providerSessionId, diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index bc1595ce5..a17f4c6f3 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -55,7 +55,7 @@ import { } from "@oh-my-pi/pi-ai"; import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives"; import { abortableSleep, getAgentDbPath, isEnoent, logger, prompt, Snowflake } from "@oh-my-pi/pi-utils"; -import type { AsyncJob, AsyncJobManager } from "../async"; +import { type AsyncJob, AsyncJobManager } from "../async"; import type { Rule } from "../capability/rule"; import { MODEL_ROLE_IDS, type ModelRegistry } from "../config/model-registry"; import { @@ -225,8 +225,6 @@ export interface AgentSessionConfig { agent: Agent; sessionManager: SessionManager; settings: Settings; - /** Async background jobs launched by tools */ - asyncJobManager?: AsyncJobManager; /** Models to cycle through with Ctrl+P (from --models flag) */ scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; /** Initial session thinking selector. */ @@ -285,6 +283,12 @@ export interface AgentSessionConfig { obfuscator?: SecretObfuscator; /** Logical owner for retained Python kernels created by this session. */ evalKernelOwnerId?: string; + /** + * AsyncJobManager that this session installed as the process-global instance. + * Only set for top-level sessions; subagents inherit the parent's manager and + * **MUST NOT** dispose it on their own teardown. + */ + ownedAsyncJobManager?: AsyncJobManager; /** Agent identity (registry id like "0-Main" or "3-Alice") used for IRC routing. */ agentId?: string; /** Shared agent registry (for forwarding IRC observations to the main session UI). */ @@ -507,7 +511,6 @@ export class AgentSession { readonly configWarnings: string[] = []; - #asyncJobManager: AsyncJobManager | undefined = undefined; #scopedModels: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>; #thinkingLevel: ThinkingLevel | undefined; #promptTemplates: PromptTemplate[]; @@ -558,6 +561,11 @@ export class AgentSession { // Python execution state #evalAbortControllers = new Set(); #evalKernelOwnerId: string; + /** + * AsyncJobManager owned by this session (top-level only). Subagents leave + * this undefined and **MUST NOT** dispose the global instance on teardown. + */ + readonly #ownedAsyncJobManager: AsyncJobManager | undefined; #pendingPythonMessages: PythonExecutionMessage[] = []; #activeEvalExecutions = new Set>(); #evalExecutionDisposing = false; @@ -704,8 +712,8 @@ export class AgentSession { this.sessionManager = config.sessionManager; this.settings = config.settings; // Power assertions are taken per turn (see #beginInFlight); nothing acquired here. - this.#asyncJobManager = config.asyncJobManager; this.#evalKernelOwnerId = config.evalKernelOwnerId ?? `agent-session:${Snowflake.next()}`; + this.#ownedAsyncJobManager = config.ownedAsyncJobManager; this.#scopedModels = config.scopedModels ?? []; this.#thinkingLevel = config.thinkingLevel; this.#promptTemplates = config.promptTemplates ?? []; @@ -848,15 +856,16 @@ export class AgentSession { } getAsyncJobSnapshot(options?: { recentLimit?: number }): AsyncJobSnapshot | null { - if (!this.#asyncJobManager) return null; - const running = this.#asyncJobManager.getRunningJobs().map(job => ({ + const manager = AsyncJobManager.instance(); + if (!manager) return null; + const running = manager.getRunningJobs().map(job => ({ id: job.id, type: job.type, status: job.status, label: job.label, startTime: job.startTime, })); - const recent = this.#asyncJobManager.getRecentJobs(options?.recentLimit ?? 5).map(job => ({ + const recent = manager.getRecentJobs(options?.recentLimit ?? 5).map(job => ({ id: job.id, type: job.type, status: job.status, @@ -866,6 +875,17 @@ export class AgentSession { return { running, recent }; } + /** + * Cancel async jobs registered by *this* agent only. Used by lifecycle + * transitions (newSession, switchSession, handoff, dispose) so a subagent + * cleans up its own background work without touching its parent's jobs. + * No-op when no manager is installed or this session has no agent id. + */ + #cancelOwnAsyncJobs(): void { + if (!this.#agentId) return; + AsyncJobManager.instance()?.cancelAll({ ownerId: this.#agentId }); + } + // ========================================================================= // Event Subscription // ========================================================================= @@ -1739,7 +1759,6 @@ export class AgentSession { } #preCacheStreamingEditFile(event: AgentEvent): void { - if (!this.settings.get("edit.streamingAbort")) return; if (this.#streamingEditAbortTriggered) return; if (event.type !== "message_update") return; @@ -1755,6 +1774,9 @@ export class AgentSession { const streamingEdit = this.#getStreamingEditToolCall(event); if (!streamingEdit) return; + // The auto-generated guard runs unconditionally: editing a generated file + // is never the user's intent, and the cost of a false-positive abort is one + // wasted turn vs. silently corrupting a regenerated source. const shouldCheckAutoGenerated = !streamingEdit.toolCall.id || !this.#streamingEditPrecheckedToolCallIds.has(streamingEdit.toolCall.id); if (shouldCheckAutoGenerated) { @@ -1768,7 +1790,12 @@ export class AgentSession { ); } - this.#ensureFileCache(streamingEdit.resolvedPath); + // File-cache priming feeds #maybeAbortStreamingEdit's removed-lines check, + // which is the optional patch-preview verification gated by + // edit.streamingAbort. Skip the read when the setting is off. + if (this.settings.get("edit.streamingAbort")) { + this.#ensureFileCache(streamingEdit.resolvedPath); + } } #ensureFileCache(resolvedPath: string): void { @@ -2149,10 +2176,21 @@ export class AgentSession { } await this.#cancelPostPromptTasks(); this.#clearTodoClearTimers(); - const drained = await this.#asyncJobManager?.dispose({ timeoutMs: 3_000 }); - const deliveryState = this.#asyncJobManager?.getDeliveryState(); - if (drained === false && deliveryState) { - logger.warn("Async job completion deliveries still pending during dispose", { ...deliveryState }); + // Cancel jobs this agent registered so a subagent's teardown doesn't + // leak its background bash/task work into the parent's manager. Only + // the session that owns the manager goes on to dispose it (which itself + // nukes any leftover jobs and pending deliveries). + this.#cancelOwnAsyncJobs(); + const ownedAsyncManager = this.#ownedAsyncJobManager; + if (ownedAsyncManager) { + const drained = await ownedAsyncManager.dispose({ timeoutMs: 3_000 }); + const deliveryState = ownedAsyncManager.getDeliveryState(); + if (drained === false && deliveryState) { + logger.warn("Async job completion deliveries still pending during dispose", { ...deliveryState }); + } + if (AsyncJobManager.instance() === ownedAsyncManager) { + AsyncJobManager.setInstance(undefined); + } } const pythonExecutionsSettled = await this.#prepareEvalExecutionsForDispose(); if (!pythonExecutionsSettled) { @@ -3948,7 +3986,7 @@ export class AgentSession { this.#disconnectFromAgent(); await this.abort(); - this.#asyncJobManager?.cancelAll(); + this.#cancelOwnAsyncJobs(); this.#closeAllProviderSessions("new session"); this.agent.reset(); if (options?.drop && previousSessionFile) { @@ -4756,7 +4794,7 @@ export class AgentSession { // Start a new session const previousSessionFile = this.sessionFile; await this.sessionManager.flush(); - this.#asyncJobManager?.cancelAll(); + this.#cancelOwnAsyncJobs(); await this.sessionManager.newSession(previousSessionFile ? { parentSession: previousSessionFile } : undefined); this.agent.reset(); this.#syncAgentSessionId(); @@ -6675,7 +6713,7 @@ export class AgentSession { const incomingRecord: CustomMessage = { role: "custom", customType: "irc:incoming", - content: `[IRC \`${args.from}\` \u2192 you]\n\n${args.message}`, + content: `[IRC \`${args.from}\` → you]\n\n${args.message}`, display: true, details: { from: args.from, message: args.message }, attribution: "agent", @@ -6707,7 +6745,7 @@ export class AgentSession { const replyRecord: CustomMessage = { role: "custom", customType: "irc:autoreply", - content: `[IRC you \u2192 \`${args.from}\` (auto)]\n\n${replyText}`, + content: `[IRC you → \`${args.from}\` (auto)]\n\n${replyText}`, display: true, details: { to: args.from, reply: replyText }, attribution: "agent", @@ -6746,7 +6784,7 @@ export class AgentSession { const mainRef = registry.get(MAIN_AGENT_ID); const mainSession = mainRef?.session; if (!mainSession || mainSession === this) return; - const arrow = args.kind === "reply" ? "\u2192 (auto)" : "\u2192"; + const arrow = args.kind === "reply" ? "→ (auto)" : "→"; const relayRecord: CustomMessage = { role: "custom", customType: "irc:relay", @@ -7149,7 +7187,7 @@ export class AgentSession { // Flush pending writes before branching await this.sessionManager.flush(); - this.#asyncJobManager?.cancelAll(); + this.#cancelOwnAsyncJobs(); if (!selectedEntry.parentId) { await this.sessionManager.newSession({ parentSession: previousSessionFile }); diff --git a/packages/coding-agent/src/session/artifacts.ts b/packages/coding-agent/src/session/artifacts.ts index 4dcb487c7..bd7676d32 100644 --- a/packages/coding-agent/src/session/artifacts.ts +++ b/packages/coding-agent/src/session/artifacts.ts @@ -12,6 +12,10 @@ import * as path from "node:path"; * * Artifacts are stored with sequential IDs in the session's artifact directory. * The directory is created lazily on first write. + * + * Subagents do not own their own `ArtifactManager`. The parent's instance is + * adopted via `SessionManager.adoptArtifactManager`, so the whole parent + + * subagent tree shares one ID space and one directory. */ export class ArtifactManager { #nextId = 0; @@ -20,11 +24,10 @@ export class ArtifactManager { #initialized = false; /** - * @param sessionFile Path to the session .jsonl file + * @param dir Directory that will hold artifact files. Created lazily on first save. */ - constructor(sessionFile: string) { - // Artifact directory is session file path without .jsonl extension - this.#dir = sessionFile.slice(0, -6); + constructor(dir: string) { + this.#dir = dir; } /** diff --git a/packages/coding-agent/src/session/session-manager.ts b/packages/coding-agent/src/session/session-manager.ts index b08d539ac..56555016c 100644 --- a/packages/coding-agent/src/session/session-manager.ts +++ b/packages/coding-agent/src/session/session-manager.ts @@ -275,6 +275,7 @@ export type ReadonlySessionManager = Pick< | "getSessionFile" | "getSessionName" | "getArtifactsDir" + | "getArtifactManager" | "allocateArtifactPath" | "saveArtifact" | "getArtifactPath" @@ -1622,6 +1623,10 @@ export class SessionManager { #persistErrorReported = false; #artifactManager: ArtifactManager | null = null; #artifactManagerSessionFile: string | null = null; + // When set, take precedence over the lazily-derived per-session manager. + // Subagents adopt the parent's manager so artifact IDs are unique across the + // whole agent tree and all files land in the parent's artifacts dir. + #adoptedArtifactManager: ArtifactManager | null = null; // In-memory artifact fallback for non-persistent sessions (persist=false). // Keyed by sequential numeric ID string; mirrors the file-based ArtifactManager ID scheme. #inMemoryArtifacts: Map | null = null; @@ -1675,6 +1680,7 @@ export class SessionManager { this.#persistErrorReported = false; this.#artifactManager = null; this.#artifactManagerSessionFile = null; + this.#adoptedArtifactManager = null; this.#buildIndex(); if (this.#sessionFile) { writeTerminalBreadcrumb(this.cwd, this.#sessionFile); @@ -2120,17 +2126,40 @@ export class SessionManager { /** * Returns the session artifacts directory path (session file path without .jsonl). * Returns null when the session is not persisted to a file. + * When this session has adopted an external ArtifactManager (subagent case), + * returns that manager's directory so reads/writes land in the shared parent + * dir instead of a private (non-existent) subdir. */ getArtifactsDir(): string | null { + if (this.#adoptedArtifactManager) return this.#adoptedArtifactManager.dir; const sessionFile = this.#sessionFile; return sessionFile ? sessionFile.slice(0, -6) : null; } + /** + * Adopt an externally-owned ArtifactManager. Used by subagents to share + * the parent session's artifact directory and ID counter. + */ + adoptArtifactManager(manager: ArtifactManager): void { + this.#adoptedArtifactManager = manager; + } + + /** + * Returns the ArtifactManager this session writes through. Lazily creates + * one bound to the current session file unless an external manager was + * adopted via `adoptArtifactManager`. Returns null only for non-persistent + * sessions with no adopted manager. + */ + getArtifactManager(): ArtifactManager | null { + return this.#getOrCreateArtifactManager(); + } + /** * Returns an artifact manager bound to the current session file. * Recreates the manager when the active session file changes. */ #getOrCreateArtifactManager(): ArtifactManager | null { + if (this.#adoptedArtifactManager) return this.#adoptedArtifactManager; const sessionFile = this.#sessionFile; if (!sessionFile) { this.#artifactManager = null; @@ -2142,7 +2171,7 @@ export class SessionManager { return this.#artifactManager; } - const manager = new ArtifactManager(sessionFile); + const manager = new ArtifactManager(sessionFile.slice(0, -6)); this.#artifactManager = manager; this.#artifactManagerSessionFile = sessionFile; return manager; diff --git a/packages/coding-agent/src/ssh/connection-manager.ts b/packages/coding-agent/src/ssh/connection-manager.ts index ded5f0c78..4cf0ad211 100644 --- a/packages/coding-agent/src/ssh/connection-manager.ts +++ b/packages/coding-agent/src/ssh/connection-manager.ts @@ -15,6 +15,11 @@ export interface SSHConnectionTarget { export type SSHHostOs = "windows" | "linux" | "macos" | "unknown"; export type SSHHostShell = "cmd" | "powershell" | "bash" | "zsh" | "sh" | "unknown"; +export type SshPlatform = typeof process.platform; + +export function supportsSshControlMaster(platform: SshPlatform = process.platform): boolean { + return platform !== "win32"; +} export interface SSHHostInfo { version: number; @@ -33,6 +38,10 @@ const activeHosts = new Map(); const pendingConnections = new Map>(); const hostInfoCache = new Map(); +interface SSHArgsOptions { + platform?: SshPlatform; +} + function ensureControlDir() { fs.mkdirSync(CONTROL_DIR, { recursive: true, mode: 0o700 }); try { @@ -66,20 +75,14 @@ async function validateKeyPermissions(keyPath?: string): Promise { } } -function buildCommonArgs(host: SSHConnectionTarget): string[] { - const args = [ - "-n", - "-o", - "ControlMaster=auto", - "-o", - `ControlPath=${CONTROL_PATH}`, - "-o", - "ControlPersist=3600", - "-o", - "BatchMode=yes", - "-o", - "StrictHostKeyChecking=accept-new", - ]; +function buildCommonArgs(host: SSHConnectionTarget, options?: SSHArgsOptions): string[] { + const args = ["-n"]; + + if (supportsSshControlMaster(options?.platform)) { + args.push("-o", "ControlMaster=auto", "-o", `ControlPath=${CONTROL_PATH}`, "-o", "ControlPersist=3600"); + } + + args.push("-o", "BatchMode=yes", "-o", "StrictHostKeyChecking=accept-new"); if (host.port) { args.push("-p", String(host.port)); @@ -357,9 +360,13 @@ export async function ensureHostInfo(host: SSHConnectionTarget): Promise { +export async function buildRemoteCommand( + host: SSHConnectionTarget, + command: string, + options?: SSHArgsOptions, +): Promise { await validateKeyPermissions(host.keyPath); - return [...buildCommonArgs(host), buildSshTarget(host.username, host.host), command]; + return [...buildCommonArgs(host, options), buildSshTarget(host.username, host.host), command]; } let registered = false; @@ -385,6 +392,14 @@ export async function ensureConnection(host: SSHConnectionTarget): Promise } const target = buildSshTarget(host.username, host.host); + if (!supportsSshControlMaster()) { + activeHosts.set(key, host); + if (!hostInfoCache.has(key) && !(await loadHostInfoFromDisk(host))) { + await probeHostInfo(host); + } + return; + } + const check = await runSshSync(["-O", "check", ...buildCommonArgs(host), target]); if (check.exitCode === 0) { activeHosts.set(key, host); @@ -415,6 +430,7 @@ export async function ensureConnection(host: SSHConnectionTarget): Promise } async function closeConnectionInternal(host: SSHConnectionTarget): Promise { + if (!supportsSshControlMaster()) return; const target = buildSshTarget(host.username, host.host); await runSshSync(["-O", "exit", ...buildCommonArgs(host), target]); } diff --git a/packages/coding-agent/src/ssh/sshfs-mount.ts b/packages/coding-agent/src/ssh/sshfs-mount.ts index 402c287e7..efc1e2351 100644 --- a/packages/coding-agent/src/ssh/sshfs-mount.ts +++ b/packages/coding-agent/src/ssh/sshfs-mount.ts @@ -2,7 +2,12 @@ import * as fs from "node:fs"; import * as path from "node:path"; import { $which, getRemoteDir, postmortem } from "@oh-my-pi/pi-utils"; import { $ } from "bun"; -import { getControlDir, getControlPathTemplate, type SSHConnectionTarget } from "./connection-manager"; +import { + getControlDir, + getControlPathTemplate, + type SSHConnectionTarget, + supportsSshControlMaster, +} from "./connection-manager"; import { buildSshTarget, sanitizeHostName } from "./utils"; const REMOTE_DIR = getRemoteDir(); @@ -40,14 +45,12 @@ function buildSshfsArgs(host: SSHConnectionTarget): string[] { "BatchMode=yes", "-o", "StrictHostKeyChecking=accept-new", - "-o", - "ControlMaster=auto", - "-o", - `ControlPath=${CONTROL_PATH}`, - "-o", - "ControlPersist=3600", ]; + if (supportsSshControlMaster()) { + args.push("-o", "ControlMaster=auto", "-o", `ControlPath=${CONTROL_PATH}`, "-o", "ControlPersist=3600"); + } + if (host.port) { args.push("-p", String(host.port)); } diff --git a/packages/coding-agent/src/system-prompt.ts b/packages/coding-agent/src/system-prompt.ts index 7666ead75..25d32f3f6 100644 --- a/packages/coding-agent/src/system-prompt.ts +++ b/packages/coding-agent/src/system-prompt.ts @@ -12,7 +12,6 @@ import type { SkillsSettings } from "./config/settings"; import { type ContextFile, loadCapability, type SystemPrompt as SystemPromptFile } from "./discovery"; import { loadSkills, type Skill } from "./extensibility/skills"; import customSystemPromptTemplate from "./prompts/system/custom-system-prompt.md" with { type: "text" }; -import nowPromptTemplate from "./prompts/system/now-prompt.md" with { type: "text" }; import projectPromptTemplate from "./prompts/system/project-prompt.md" with { type: "text" }; import systemPromptTemplate from "./prompts/system/system-prompt.md" with { type: "text" }; import { shortenPath } from "./tools/render-utils"; @@ -575,10 +574,6 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}): if (projectPrompt) { systemPrompt.push(projectPrompt); } - const nowPrompt = prompt.render(nowPromptTemplate, data).trim(); - if (nowPrompt) { - systemPrompt.push(nowPrompt); - } return { systemPrompt }; } diff --git a/packages/coding-agent/src/task/executor.ts b/packages/coding-agent/src/task/executor.ts index 73157a33b..a46e4c688 100644 --- a/packages/coding-agent/src/task/executor.ts +++ b/packages/coding-agent/src/task/executor.ts @@ -26,9 +26,11 @@ import submitReminderTemplate from "../prompts/system/subagent-yield-reminder.md import { AgentRegistry } from "../registry/agent-registry"; import { createAgentSession, discoverAuthStorage } from "../sdk"; import type { AgentSession, AgentSessionEvent } from "../session/agent-session"; +import type { ArtifactManager } from "../session/artifacts"; import type { AuthStorage } from "../session/auth-storage"; import { SessionManager } from "../session/session-manager"; -import { type ContextFileEntry, truncateTail } from "../tools"; +import { truncateTail } from "../session/streaming-output"; +import type { ContextFileEntry } from "../tools"; import { jtdToJsonSchema, normalizeSchema } from "../tools/jtd-to-json-schema"; import { ToolAbortError } from "../tools/tool-errors"; import type { EventBus } from "../utils/event-bus"; @@ -172,6 +174,12 @@ export interface ExecutorOptions { settings?: Settings; /** Override local:// protocol options so subagent shares parent's local:// root */ localProtocolOptions?: LocalProtocolOptions; + /** + * Parent session's ArtifactManager. Subagent adopts it so artifact IDs are + * unique across the whole agent tree and all artifacts land in the parent's + * artifacts directory (no per-subagent subdir). + */ + parentArtifactManager?: ArtifactManager; parentHindsightSessionState?: HindsightSessionState; } @@ -563,6 +571,7 @@ export async function runSubprocess(options: ExecutorOptions): Promise 0 ? agents.filter(a => !disabledAgents.includes(a.name)) : agents; const { contextEnabled, customSchemaEnabled } = getTaskSimpleModeCapabilities(simpleMode); @@ -151,6 +154,7 @@ function renderDescription( asyncEnabled, contextEnabled, customSchemaEnabled, + ircEnabled, defaultMode: simpleMode === "default", schemaFreeMode: simpleMode === "schema-free", independentMode: simpleMode === "independent", @@ -229,6 +233,7 @@ export class TaskTool implements AgentTool { this.session.settings.get("async.enabled"), disabledAgents, this.#getTaskSimpleMode(), + this.session.settings.get("irc.enabled") === true, ); } private constructor( @@ -270,7 +275,7 @@ export class TaskTool implements AgentTool { return this.#executeSync(_toolCallId, params, signal, onUpdate); } - const manager = this.session.asyncJobManager; + const manager = AsyncJobManager.instance(); if (!manager) { return { content: [{ type: "text", text: "Async execution is enabled but no async job manager is available." }], @@ -444,6 +449,7 @@ export class TaskTool implements AgentTool { }, { id: label, + ownerId: this.session.getAgentId?.() ?? undefined, onProgress: (text, details) => { const progressDetails = (details as TaskToolDetails | undefined) ?? @@ -729,6 +735,10 @@ export class TaskTool implements AgentTool { getSessionId: this.session.getSessionId ?? (() => null), }; + // Subagents adopt the parent's ArtifactManager so artifact IDs are unique + // across the whole tree and outputs land flat in the parent's dir. + const parentArtifactManager = this.session.getArtifactManager?.() ?? undefined; + // Initialize progress tracking const progressMap = new Map(); @@ -785,9 +795,11 @@ export class TaskTool implements AgentTool { }; } - // Write parent conversation context for subagents + // Write parent conversation context for subagents. When IRC is available, + // subagents should ask live peers instead of reading a stale markdown dump. await fs.mkdir(effectiveArtifactsDir, { recursive: true }); - const compactContext = this.session.getCompactContext?.(); + const shouldWriteConversationContext = this.session.settings.get("irc.enabled") !== true; + const compactContext = shouldWriteConversationContext ? this.session.getCompactContext?.() : undefined; let contextFilePath: string | undefined; if (compactContext) { contextFilePath = path.join(effectiveArtifactsDir, "context.md"); @@ -867,12 +879,13 @@ export class TaskTool implements AgentTool { authStorage: this.session.authStorage, modelRegistry: this.session.modelRegistry, settings: this.session.settings, - mcpManager: this.session.mcpManager, + mcpManager: MCPManager.instance(), contextFiles, skills: availableSkills, workspaceTree: this.session.workspaceTree, promptTemplates, localProtocolOptions, + parentArtifactManager, parentHindsightSessionState: this.session.getHindsightSessionState?.(), }); } @@ -925,12 +938,13 @@ export class TaskTool implements AgentTool { authStorage: this.session.authStorage, modelRegistry: this.session.modelRegistry, settings: this.session.settings, - mcpManager: this.session.mcpManager, + mcpManager: MCPManager.instance(), contextFiles, skills: availableSkills, workspaceTree: this.session.workspaceTree, promptTemplates, localProtocolOptions, + parentArtifactManager, parentHindsightSessionState: this.session.getHindsightSessionState?.(), }); if (mergeMode === "branch" && result.exitCode === 0) { diff --git a/packages/coding-agent/src/tool-discovery/tool-index.ts b/packages/coding-agent/src/tool-discovery/tool-index.ts index b2ffdc5f8..71662ddb5 100644 --- a/packages/coding-agent/src/tool-discovery/tool-index.ts +++ b/packages/coding-agent/src/tool-discovery/tool-index.ts @@ -89,6 +89,7 @@ export interface DiscoverableMCPSearchResult { const BM25_K1 = 1.2; const BM25_B = 0.75; +const BM25_DELTA = 1.0; const FIELD_WEIGHTS = { name: 6, label: 4, @@ -112,13 +113,24 @@ function getSchemaPropertyKeys(parameters: unknown): string[] { } function tokenize(value: string): string[] { - return value - .replace(/([a-z0-9])([A-Z])/g, "$1 $2") - .replace(/[^a-zA-Z0-9]+/g, " ") - .toLowerCase() - .trim() - .split(/\s+/) - .filter(token => token.length > 0); + return ( + value + .normalize("NFKD") + // Drop combining marks (accents) so "café" → "cafe". + .replace(/\p{M}+/gu, "") + // Split ACRONYMBoundary: "MCPTool" → "MCP Tool". + .replace(/(\p{Lu}+)(\p{Lu}\p{Ll})/gu, "$1 $2") + // Split camelCase / digit→letter: "fooBar" → "foo Bar", "v2Beta" → "v2 Beta". + .replace(/(\p{Ll}|\p{N})(\p{Lu})/gu, "$1 $2") + // Everything that isn't a letter or digit becomes a separator. This subsumes markdown + // punctuation (`|*_`#-~>[]()`), box-drawing glyphs (─│┌), em/en dashes, smart quotes, + // zero-width spaces, NBSPs, etc. + .replace(/[^\p{L}\p{N}]+/gu, " ") + .toLowerCase() + .trim() + .split(/\s+/) + .filter(token => token.length > 0) + ); } function addWeightedTokens(termFrequencies: Map, value: string | undefined, weight: number): void { @@ -274,7 +286,8 @@ export function searchDiscoverableTools( const documentFrequency = index.documentFrequencies.get(token) ?? 0; const idf = Math.log(1 + (index.documents.length - documentFrequency + 0.5) / (documentFrequency + 0.5)); const normalization = BM25_K1 * (1 - BM25_B + BM25_B * (document.length / index.averageLength)); - score += queryTermCount * idf * ((termFrequency * (BM25_K1 + 1)) / (termFrequency + normalization)); + score += + queryTermCount * idf * ((termFrequency * (BM25_K1 + 1)) / (termFrequency + normalization) + BM25_DELTA); } return { tool: document.tool, score }; }) diff --git a/packages/coding-agent/src/tools/ast-edit.ts b/packages/coding-agent/src/tools/ast-edit.ts index ed07f0e0c..60d44a309 100644 --- a/packages/coding-agent/src/tools/ast-edit.ts +++ b/packages/coding-agent/src/tools/ast-edit.ts @@ -7,6 +7,7 @@ import { $envpos, prompt, untilAborted } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { computeLineHash, HL_BODY_SEP } from "../hashline/hash"; +import { InternalUrlRouter } from "../internal-urls"; import type { Theme } from "../modes/theme/theme"; import astEditDescription from "../prompts/tools/ast-edit.md" with { type: "text" }; import { Ellipsis, Hasher, type RenderCache, renderStatusLine, renderTreeList, truncateToWidth } from "../tui"; @@ -213,10 +214,10 @@ export class AstEditTool implements AgentTool rawPath.length === 0)) { throw new ToolError("`paths` must contain non-empty paths or globs"); } - const internalRouter = this.session.internalRouter; + const internalRouter = InternalUrlRouter.instance(); const resolvedPathInputs: string[] = []; for (const rawPath of rawPaths) { - if (!internalRouter?.canHandle(rawPath)) { + if (!internalRouter.canHandle(rawPath)) { resolvedPathInputs.push(rawPath); continue; } diff --git a/packages/coding-agent/src/tools/ast-grep.ts b/packages/coding-agent/src/tools/ast-grep.ts index 582f1e0cc..12fcd56b1 100644 --- a/packages/coding-agent/src/tools/ast-grep.ts +++ b/packages/coding-agent/src/tools/ast-grep.ts @@ -6,6 +6,7 @@ import { Text } from "@oh-my-pi/pi-tui"; import { prompt, untilAborted } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; +import { InternalUrlRouter } from "../internal-urls"; import type { Theme } from "../modes/theme/theme"; import astGrepDescription from "../prompts/tools/ast-grep.md" with { type: "text" }; import { Ellipsis, Hasher, type RenderCache, renderStatusLine, renderTreeList, truncateToWidth } from "../tui"; @@ -158,10 +159,10 @@ export class AstGrepTool implements AgentTool rawPath.length === 0)) { throw new ToolError("`paths` must contain non-empty paths or globs"); } - const internalRouter = this.session.internalRouter; + const internalRouter = InternalUrlRouter.instance(); const resolvedPathInputs: string[] = []; for (const rawPath of rawPaths) { - if (!internalRouter?.canHandle(rawPath)) { + if (!internalRouter.canHandle(rawPath)) { resolvedPathInputs.push(rawPath); continue; } diff --git a/packages/coding-agent/src/tools/bash.ts b/packages/coding-agent/src/tools/bash.ts index 39726d7cc..31f47fc35 100644 --- a/packages/coding-agent/src/tools/bash.ts +++ b/packages/coding-agent/src/tools/bash.ts @@ -4,8 +4,10 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { ImageProtocol, TERMINAL, Text } from "@oh-my-pi/pi-tui"; import { $env, getProjectDir, isEnoent, prompt } from "@oh-my-pi/pi-utils"; import { Type } from "@sinclair/typebox"; +import { AsyncJobManager } from "../async"; import { type BashResult, executeBash } from "../exec/bash-executor"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; +import { InternalUrlRouter } from "../internal-urls"; import { truncateToVisualLines } from "../modes/components/visual-truncate"; import type { Theme } from "../modes/theme/theme"; import bashDescription from "../prompts/tools/bash.md" with { type: "text" }; @@ -326,7 +328,9 @@ export class BashTool implements AgentTool { } lines.push(`Background job ${jobId} started: ${label}`); lines.push("Result will be delivered automatically when complete."); - lines.push(`Use \`job\` (with \`poll\` or \`cancel\`) or \`read jobs://${jobId}\` if needed.`); + lines.push( + `You can use \`job\` to poll until complete, but prefer to continue with another task in the meanwhile if it's not blocking.`, + ); return { content: [{ type: "text", text: lines.join("\n") }], details, @@ -349,7 +353,7 @@ export class BashTool implements AgentTool { onUpdate?: AgentToolUpdateCallback; startBackgrounded: boolean; }): ManagedBashJobHandle { - const manager = this.session.asyncJobManager; + const manager = AsyncJobManager.instance(); if (!manager) { throw new ToolError("Background job manager unavailable for this session."); } @@ -399,6 +403,7 @@ export class BashTool implements AgentTool { } }, { + ownerId: this.session.getAgentId?.() ?? undefined, onProgress: async (text, details) => { latestText = text; await options.onUpdate?.({ @@ -501,7 +506,7 @@ export class BashTool implements AgentTool { const internalUrlOptions: InternalUrlExpansionOptions = { skills: this.session.skills ?? [], - internalRouter: this.session.internalRouter, + internalRouter: InternalUrlRouter.instance(), localOptions: { getArtifactsDir: this.session.getArtifactsDir, getSessionId: this.session.getSessionId, @@ -549,7 +554,7 @@ export class BashTool implements AgentTool { const timeoutClampNotice = formatTimeoutClampNotice(requestedTimeoutSec, timeoutSec); if (asyncRequested) { - if (!this.session.asyncJobManager) { + if (!AsyncJobManager.instance()) { throw new ToolError("Async job manager unavailable for this session."); } const job = this.#startManagedBashJob({ @@ -570,7 +575,8 @@ export class BashTool implements AgentTool { }); } - if (this.#autoBackgroundEnabled && !pty && this.session.asyncJobManager) { + const autoBgManager = AsyncJobManager.instance(); + if (this.#autoBackgroundEnabled && !pty && autoBgManager) { const autoBackgroundWaitMs = this.#resolveAutoBackgroundWaitMs(timeoutMs); const startBackgrounded = autoBackgroundWaitMs === 0; const job = this.#startManagedBashJob({ @@ -593,16 +599,16 @@ export class BashTool implements AgentTool { } const waitResult = await this.#waitForManagedBashJob(job, autoBackgroundWaitMs, signal); if (waitResult.kind === "completed") { - this.session.asyncJobManager.acknowledgeDeliveries([job.jobId]); + autoBgManager.acknowledgeDeliveries([job.jobId]); return waitResult.result; } if (waitResult.kind === "failed") { - this.session.asyncJobManager.acknowledgeDeliveries([job.jobId]); + autoBgManager.acknowledgeDeliveries([job.jobId]); throw waitResult.error; } if (waitResult.kind === "aborted") { - this.session.asyncJobManager.cancel(job.jobId); - this.session.asyncJobManager.acknowledgeDeliveries([job.jobId]); + autoBgManager.cancel(job.jobId); + autoBgManager.acknowledgeDeliveries([job.jobId]); throw new ToolAbortError(job.getLatestText() || "Command aborted"); } job.setBackgrounded(true); diff --git a/packages/coding-agent/src/tools/browser/tab-supervisor.ts b/packages/coding-agent/src/tools/browser/tab-supervisor.ts index 96bad2ae6..6afd56630 100644 --- a/packages/coding-agent/src/tools/browser/tab-supervisor.ts +++ b/packages/coding-agent/src/tools/browser/tab-supervisor.ts @@ -16,6 +16,13 @@ import type { WorkerInitPayload, WorkerOutbound, } from "./tab-protocol"; +// Imported with `type: "file"` so Bun's bundler statically discovers the +// worker entry and embeds it inside `bun build --compile` single-file +// binaries. Without this attribute the bundler cannot reach the entry through +// a `new URL(..., import.meta.url)` literal stored in a local variable, and +// the prebuilt binary surfaces `Timed out initializing browser tab worker` +// (issue #1011) because `/$bunfs/root/tab-worker-entry.ts` is missing. +import tabWorkerEntryUrl from "./tab-worker-entry.ts" with { type: "file" }; interface WorkerHandle { send(msg: WorkerInbound, transferList?: Transferable[]): void; @@ -364,8 +371,7 @@ async function raceWithTimeout( async function spawnTabWorker(): Promise { try { - const url = new URL("./tab-worker-entry.ts", import.meta.url); - const worker = new Worker(url.href, { type: "module" }); + const worker = new Worker(tabWorkerEntryUrl, { type: "module" }); return wrapBunWorker(worker); } catch (err) { logger.warn("Bun Worker spawn failed; using inline tab worker (no sync-loop guard)", { diff --git a/packages/coding-agent/src/tools/eval.ts b/packages/coding-agent/src/tools/eval.ts index ce3b9b7ea..ccd40a877 100644 --- a/packages/coding-agent/src/tools/eval.ts +++ b/packages/coding-agent/src/tools/eval.ts @@ -8,7 +8,7 @@ import { jsBackend, parseEvalInput, pythonBackend, sniffEvalLanguage } from "../ import type { ExecutorBackend } from "../eval/backend"; import evalGrammar from "../eval/eval.lark" with { type: "text" }; import { ABORT_WARNING, type ParsedEvalCell } from "../eval/parse"; -import type { EvalCellResult, EvalLanguage, EvalStatusEvent, EvalToolDetails } from "../eval/types"; +import type { EvalCellResult, EvalDisplayOutput, EvalLanguage, EvalStatusEvent, EvalToolDetails } from "../eval/types"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { truncateToVisualLines } from "../modes/components/visual-truncate"; import { getMarkdownTheme, type Theme } from "../modes/theme/theme"; @@ -47,6 +47,38 @@ function formatJsonScalar(value: unknown): string { return "[object]"; } +/** Cap per `display()` value sent back to the model. */ +const MAX_DISPLAY_TEXT_BYTES = 8000; + +function formatDisplayJsonForText(value: unknown): string { + let text: string; + try { + text = JSON.stringify(value, null, 2) ?? String(value); + } catch { + text = String(value); + } + if (text.length > MAX_DISPLAY_TEXT_BYTES) { + text = `${text.slice(0, MAX_DISPLAY_TEXT_BYTES)}\n… (${text.length - MAX_DISPLAY_TEXT_BYTES} chars truncated)`; + } + return text; +} + +/** + * Format display() JSON values into text the model can see. Images are surfaced + * separately as ImageContent so the model can actually inspect them; this helper + * intentionally does not touch images. + */ +function formatDisplayOutputsForText(outputs: EvalDisplayOutput[]): string { + const chunks: string[] = []; + let displayIndex = 0; + for (const output of outputs) { + if (output.type !== "json") continue; + displayIndex++; + chunks.push(`display[${displayIndex}]:\n${formatDisplayJsonForText(output.data)}`); + } + return chunks.join("\n\n"); +} + function renderJsonTree(value: unknown, theme: Theme, expanded: boolean, maxDepth = expanded ? 6 : 2): string[] { const maxItems = expanded ? 20 : 5; @@ -370,13 +402,16 @@ export class EvalTool implements AgentTool { const durationMs = Date.now() - startTime; const cellStatusEvents: EvalStatusEvent[] = []; + const cellDisplayOutputs: EvalDisplayOutput[] = []; let cellHasMarkdown = false; for (const output of result.displayOutputs) { if (output.type === "json") { jsonOutputs.push(output.data); + cellDisplayOutputs.push(output); } if (output.type === "image") { images.push({ type: "image", data: output.data, mimeType: output.mimeType }); + cellDisplayOutputs.push(output); } if (output.type === "status") { statusEvents.push(output.event); @@ -387,7 +422,10 @@ export class EvalTool implements AgentTool { } } - const cellOutput = result.output.trim(); + const stdoutTrimmed = result.output.trim(); + const displayText = formatDisplayOutputsForText(cellDisplayOutputs); + const cellOutput = + stdoutTrimmed && displayText ? `${stdoutTrimmed}\n\n${displayText}` : stdoutTrimmed || displayText; cellResult.output = cellOutput; cellResult.exitCode = result.exitCode; cellResult.durationMs = durationMs; @@ -431,14 +469,13 @@ export class EvalTool implements AgentTool { languages, cells: cellResults, jsonOutputs: jsonOutputs.length > 0 ? jsonOutputs : undefined, - images: images.length > 0 ? images : undefined, statusEvents: statusEvents.length > 0 ? statusEvents : undefined, isError: true, }; if (notice) details.notice = notice; return toolResult(details) - .text(outputText) + .content([{ type: "text", text: outputText }, ...images]) .truncationFromSummary(summaryForMeta, { direction: "tail" }) .done(); } @@ -461,14 +498,13 @@ export class EvalTool implements AgentTool { languages, cells: cellResults, jsonOutputs: jsonOutputs.length > 0 ? jsonOutputs : undefined, - images: images.length > 0 ? images : undefined, statusEvents: statusEvents.length > 0 ? statusEvents : undefined, isError: true, }; if (notice) details.notice = notice; return toolResult(details) - .text(outputText) + .content([{ type: "text", text: outputText }, ...images]) .truncationFromSummary(summaryForMeta, { direction: "tail" }) .done(); } @@ -479,9 +515,12 @@ export class EvalTool implements AgentTool { const combinedOutput = cellOutputs.join("\n\n"); const abortSuffix = parsedInput.aborted ? `\n\n${ABORT_WARNING}` : ""; + const hasImages = images.length > 0; const outputText = - (combinedOutput || (jsonOutputs.length > 0 || images.length > 0 ? "(no text output)" : "(no output)")) + - abortSuffix; + (combinedOutput || + (hasImages + ? `(displayed ${images.length} image${images.length === 1 ? "" : "s"}; no text output)` + : "(no output)")) + abortSuffix; const summaryForMeta = await summarizeFinal(combinedOutput, finalizeOutput); const details: EvalToolDetails = { @@ -489,13 +528,12 @@ export class EvalTool implements AgentTool { languages, cells: cellResults, jsonOutputs: jsonOutputs.length > 0 ? jsonOutputs : undefined, - images: images.length > 0 ? images : undefined, statusEvents: statusEvents.length > 0 ? statusEvents : undefined, }; if (notice) details.notice = notice; return toolResult(details) - .text(outputText) + .content([{ type: "text", text: outputText }, ...images]) .truncationFromSummary(summaryForMeta, { direction: "tail" }) .done(); } finally { diff --git a/packages/coding-agent/src/tools/fetch.ts b/packages/coding-agent/src/tools/fetch.ts index 33db792c2..06cdb5d98 100644 --- a/packages/coding-agent/src/tools/fetch.ts +++ b/packages/coding-agent/src/tools/fetch.ts @@ -1352,7 +1352,7 @@ export function renderReadUrlCall( ): Component { const url = args.path ?? args.url ?? ""; const domain = getDomain(url); - const path = truncate(url.replace(/^https?:\/\/[^/]+/, ""), 50, "\u2026"); + const path = truncate(url.replace(/^https?:\/\/[^/]+/, ""), 50, "…"); const description = `${domain}${path ? ` ${path}` : ""}`.trim(); const meta: string[] = []; if (args.raw) meta.push("raw"); diff --git a/packages/coding-agent/src/tools/gh.ts b/packages/coding-agent/src/tools/gh.ts index b7fe1a21e..e86895d2f 100644 --- a/packages/coding-agent/src/tools/gh.ts +++ b/packages/coding-agent/src/tools/gh.ts @@ -260,6 +260,27 @@ const githubSchema = Type.Object({ examples: ["is:open label:bug"], }), ), + since: Type.Optional( + Type.String({ + description: + "lower-bound date for search_issues/search_prs/search_commits/search_repos. Accepts a relative duration (`` with unit `m`/`h`/`d`/`w`/`mo`/`y`, e.g. `3d`, `12h`, `2w`) or an ISO date (`YYYY-MM-DD`) / datetime. Translated to a `created:>=…` (or `committer-date:`/`pushed:`) qualifier; not supported by search_code.", + examples: ["3d", "2w", "2026-05-01"], + }), + ), + until: Type.Optional( + Type.String({ + description: + "upper-bound date in the same format as `since`. With both, builds a `field:since..until` range qualifier.", + examples: ["1d", "2026-05-09"], + }), + ), + dateField: Type.Optional( + StringEnum(["created", "updated"], { + description: + "date field used by `since`/`until`. issues/prs: `created` (default) or `updated`. repos: `created` (default) or `updated` (mapped to GitHub's `pushed:`). commits: ignored — always uses `committer-date`.", + default: "created", + }), + ), limit: Type.Optional( Type.Number({ description: "max results (search_issues, search_prs, search_code, search_commits, search_repos)", @@ -686,6 +707,110 @@ const SEARCH_FIELDS_BY_COMMAND: Record<"issues" | "prs" | "code" | "commits" | " repos: GH_SEARCH_REPOS_FIELDS, }; +const RELATIVE_DURATION_PATTERN = /^(\d+)\s*(m|h|d|w|mo|y)$/i; +const ISO_DATE_PATTERN = /^\d{4}-\d{2}-\d{2}$/; +const FIXED_UNIT_MS: Record = { + m: 60_000, + h: 3_600_000, + d: 86_400_000, + w: 7 * 86_400_000, +}; + +/** + * Resolve a search date bound to a GitHub-search-compatible literal. Returns + * either a `YYYY-MM-DD` date (relative durations and date-only inputs) or a + * full ISO 8601 datetime string (datetime inputs), so the caller can drop it + * straight into a qualifier like `created:>=`. + */ +export function parseSearchDateBound(raw: string, now: Date = new Date()): string { + const trimmed = raw.trim(); + if (!trimmed) { + throw new ToolError("date bound must not be empty"); + } + + const relMatch = trimmed.match(RELATIVE_DURATION_PATTERN); + if (relMatch) { + const count = Number(relMatch[1]); + const unit = relMatch[2].toLowerCase(); + const fixedMs = FIXED_UNIT_MS[unit]; + let bound: Date; + if (fixedMs !== undefined) { + bound = new Date(now.getTime() - count * fixedMs); + } else { + bound = new Date(now); + if (unit === "mo") { + bound.setUTCMonth(bound.getUTCMonth() - count); + } else { + bound.setUTCFullYear(bound.getUTCFullYear() - count); + } + } + return bound.toISOString().slice(0, 10); + } + + if (ISO_DATE_PATTERN.test(trimmed)) { + return trimmed; + } + + const parsedMs = Date.parse(trimmed); + if (!Number.isNaN(parsedMs)) { + return new Date(parsedMs).toISOString(); + } + + throw new ToolError( + `invalid date bound: ${raw}. Expected a relative duration like "3d", "12h", "2w", an ISO date "YYYY-MM-DD", or an ISO datetime.`, + ); +} + +/** + * Build the GitHub-search qualifier (e.g. `created:>=2026-05-09`) for the + * provided bounds, or `undefined` if neither bound is set. + */ +export function buildSearchDateQualifier( + field: string, + since: string | undefined, + until: string | undefined, + now?: Date, +): string | undefined { + const sinceVal = since ? parseSearchDateBound(since, now) : undefined; + const untilVal = until ? parseSearchDateBound(until, now) : undefined; + if (sinceVal && untilVal) { + return `${field}:${sinceVal}..${untilVal}`; + } + if (sinceVal) { + return `${field}:>=${sinceVal}`; + } + if (untilVal) { + return `${field}:<=${untilVal}`; + } + return undefined; +} + +function resolveSearchDateField( + command: "issues" | "prs" | "commits" | "repos", + requested: "created" | "updated" | undefined, +): string { + if (command === "commits") { + return "committer-date"; + } + const dateField = requested ?? "created"; + if (command === "repos" && dateField === "updated") { + return "pushed"; + } + return dateField; +} + +function composeSearchQuery(parts: ReadonlyArray): string { + const cleaned: string[] = []; + for (const part of parts) { + const trimmed = part?.trim(); + if (trimmed) cleaned.push(trimmed); + } + if (cleaned.length === 0) { + throw new ToolError("query is required (or pass since/until to filter by date)"); + } + return cleaned.join(" "); +} + function buildGhSearchArgs( command: "issues" | "prs" | "code" | "commits" | "repos", query: string, @@ -2636,9 +2761,11 @@ async function executeSearchIssues( params: GithubInput, signal: AbortSignal | undefined, ): Promise> { - const query = requireNonEmpty(params.query, "query"); const repo = normalizeOptionalString(params.repo); const limit = resolveSearchLimit(params.limit); + const dateField = resolveSearchDateField("issues", params.dateField); + const dateQualifier = buildSearchDateQualifier(dateField, params.since, params.until); + const query = composeSearchQuery([params.query, dateQualifier]); const args = buildGhSearchArgs("issues", query, limit, repo); const items = await git.github.json(session.cwd, args, signal, { @@ -2652,9 +2779,11 @@ async function executeSearchPrs( params: GithubInput, signal: AbortSignal | undefined, ): Promise> { - const query = requireNonEmpty(params.query, "query"); const repo = normalizeOptionalString(params.repo); const limit = resolveSearchLimit(params.limit); + const dateField = resolveSearchDateField("prs", params.dateField); + const dateQualifier = buildSearchDateQualifier(dateField, params.since, params.until); + const query = composeSearchQuery([params.query, dateQualifier]); const args = buildGhSearchArgs("prs", query, limit, repo); const items = await git.github.json(session.cwd, args, signal, { @@ -2669,6 +2798,9 @@ async function executeSearchCode( signal: AbortSignal | undefined, ): Promise> { const query = requireNonEmpty(params.query, "query"); + if (params.since !== undefined || params.until !== undefined) { + throw new ToolError("search_code does not support since/until; GitHub code search has no date qualifier."); + } const repo = normalizeOptionalString(params.repo); const limit = resolveSearchLimit(params.limit); const args = buildGhSearchArgs("code", query, limit, repo); @@ -2684,9 +2816,11 @@ async function executeSearchCommits( params: GithubInput, signal: AbortSignal | undefined, ): Promise> { - const query = requireNonEmpty(params.query, "query"); const repo = normalizeOptionalString(params.repo); const limit = resolveSearchLimit(params.limit); + const dateField = resolveSearchDateField("commits", params.dateField); + const dateQualifier = buildSearchDateQualifier(dateField, params.since, params.until); + const query = composeSearchQuery([params.query, dateQualifier]); const args = buildGhSearchArgs("commits", query, limit, repo); const items = await git.github.json(session.cwd, args, signal, { @@ -2700,8 +2834,10 @@ async function executeSearchRepos( params: GithubInput, signal: AbortSignal | undefined, ): Promise> { - const query = requireNonEmpty(params.query, "query"); const limit = resolveSearchLimit(params.limit); + const dateField = resolveSearchDateField("repos", params.dateField); + const dateQualifier = buildSearchDateQualifier(dateField, params.since, params.until); + const query = composeSearchQuery([params.query, dateQualifier]); const args = buildGhSearchArgs("repos", query, limit, undefined); const items = await git.github.json(session.cwd, args, signal); diff --git a/packages/coding-agent/src/tools/index.ts b/packages/coding-agent/src/tools/index.ts index a96477dac..c25cccb23 100644 --- a/packages/coding-agent/src/tools/index.ts +++ b/packages/coding-agent/src/tools/index.ts @@ -1,17 +1,16 @@ import type { AgentTool } from "@oh-my-pi/pi-agent-core"; import type { ToolChoice } from "@oh-my-pi/pi-ai"; import { $env, $flag, logger } from "@oh-my-pi/pi-utils"; -import type { AsyncJobManager } from "../async"; import type { PromptTemplate } from "../config/prompt-templates"; import type { Settings } from "../config/settings"; import { EditTool } from "../edit"; import { checkPythonKernelAvailability } from "../eval/py/kernel"; import type { Skill } from "../extensibility/skills"; import type { HindsightSessionState } from "../hindsight/state"; -import type { InternalUrlRouter } from "../internal-urls"; import { LspTool } from "../lsp"; import type { PlanModeState } from "../plan-mode/state"; import { type AgentRegistry, MAIN_AGENT_ID } from "../registry/agent-registry"; +import type { ArtifactManager } from "../session/artifacts"; import type { CustomMessage } from "../session/messages"; import type { ToolChoiceQueue } from "../session/tool-choice-queue"; import { TaskTool } from "../task"; @@ -159,6 +158,8 @@ export interface ToolSession { agentRegistry?: AgentRegistry; /** Get artifacts directory for artifact:// URLs */ getArtifactsDir?: () => string | null; + /** Get the ArtifactManager backing this session (shared across parent + subagents). */ + getArtifactManager?: () => ArtifactManager | null; /** Allocate a new artifact path and ID for session-scoped truncated output. */ allocateOutputArtifact?: (toolType: string) => Promise<{ id?: string; path?: string }>; /** Get session spawns */ @@ -171,14 +172,8 @@ export interface ToolSession { authStorage?: import("../session/auth-storage").AuthStorage; /** Model registry for passing to subagents (avoids re-discovery) */ modelRegistry?: import("../config/model-registry").ModelRegistry; - /** MCP manager for proxying MCP calls through parent */ - mcpManager?: import("../mcp/manager").MCPManager; - /** Internal URL router for protocols like agent://, skill://, and mcp:// */ - internalRouter?: InternalUrlRouter; /** Agent output manager for unique agent:// IDs across task invocations */ agentOutputManager?: AgentOutputManager; - /** Async background job manager for bash/task async execution */ - asyncJobManager?: AsyncJobManager; /** Settings instance for passing to subagents */ settings: Settings; /** Plan mode state (if active) */ @@ -282,7 +277,7 @@ export const BUILTIN_TOOLS: Record = { browser: s => new BrowserTool(s), checkpoint: CheckpointTool.createIf, rewind: RewindTool.createIf, - task: TaskTool.create, + task: s => TaskTool.create(s), job: JobTool.createIf, recipe: RecipeTool.createIf, irc: IrcTool.createIf, diff --git a/packages/coding-agent/src/tools/job.ts b/packages/coding-agent/src/tools/job.ts index d40c00ebf..05ed772e6 100644 --- a/packages/coding-agent/src/tools/job.ts +++ b/packages/coding-agent/src/tools/job.ts @@ -3,7 +3,7 @@ import type { Component } from "@oh-my-pi/pi-tui"; import { Text } from "@oh-my-pi/pi-tui"; import { prompt } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; -import { isBackgroundJobSupportEnabled } from "../async"; +import { type AsyncJob, AsyncJobManager, isBackgroundJobSupportEnabled } from "../async"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import type { Theme } from "../modes/theme/theme"; import jobDescription from "../prompts/tools/job.md" with { type: "text" }; @@ -20,6 +20,7 @@ import { type ToolUIColor, type ToolUIStatus, } from "./render-utils"; +import { ToolError } from "./tool-errors"; const jobSchema = Type.Object({ poll: Type.Optional( @@ -34,6 +35,12 @@ const jobSchema = Type.Object({ examples: [["job-1234"]], }), ), + list: Type.Optional( + Type.Boolean({ + description: + "Return an immediate snapshot of every job spawned by this agent (running + completed within retention). Read-only \u2014 cannot be combined with `poll` or `cancel`.", + }), + ), }); type JobParams = Static; @@ -97,7 +104,7 @@ export class JobTool implements AgentTool { onUpdate?: AgentToolUpdateCallback, _context?: AgentToolContext, ): Promise> { - const manager = this.session.asyncJobManager; + const manager = AsyncJobManager.instance(); if (!manager) { return { content: [{ type: "text", text: "Async execution is disabled; no background jobs are available." }], @@ -105,11 +112,24 @@ export class JobTool implements AgentTool { }; } + // Scope every visible operation to the calling agent. Tests / SDK + // consumers without an agent id see everything (legacy behavior). + const ownerId = this.session.getAgentId?.() ?? undefined; + const ownerFilter = ownerId ? { ownerId } : undefined; + + // `list` is a read-only snapshot mode. Replaces the legacy `jobs://` URL. + if (params.list) { + if (params.cancel?.length || params.poll?.length) { + throw new ToolError("`list` cannot be combined with `poll` or `cancel`."); + } + return this.#buildResult(manager, manager.getAllJobs(ownerFilter), []); + } + const cancelIds = params.cancel ?? []; const cancelOutcomes: CancelOutcome[] = []; for (const id of cancelIds) { const existing = manager.getJob(id); - if (!existing) { + if (!existing || (ownerId && existing.ownerId !== ownerId)) { cancelOutcomes.push({ id, status: "not_found", message: `Background job not found: ${id}` }); continue; } @@ -121,7 +141,7 @@ export class JobTool implements AgentTool { }); continue; } - const cancelled = manager.cancel(id); + const cancelled = manager.cancel(id, ownerFilter); cancelOutcomes.push( cancelled ? { id, status: "cancelled", message: `Cancelled background job ${id}.` } @@ -130,11 +150,11 @@ export class JobTool implements AgentTool { } const requestedPollIds = params.poll; - // If only `cancel` was provided (no `poll`), don't wait — return immediately. + // If only `cancel` was provided (no `poll`), don't wait \u2014 return immediately. const shouldPoll = requestedPollIds !== undefined || cancelIds.length === 0; if (!shouldPoll) { - const cancelledJobs = cancelIds.map(id => manager.getJob(id)).filter(j => j != null); + const cancelledJobs = this.#visibleJobs(manager, cancelIds, ownerId); return this.#buildResult(manager, cancelledJobs, cancelOutcomes); } @@ -142,12 +162,12 @@ export class JobTool implements AgentTool { // - If `poll` was passed explicitly, watch exactly those (filtered to existing). // - If `poll` was omitted (and so was `cancel`), default to all running jobs. const jobsToWatch = requestedPollIds - ? requestedPollIds.map(id => manager.getJob(id)).filter(j => j != null) - : manager.getRunningJobs(); + ? this.#visibleJobs(manager, requestedPollIds, ownerId) + : manager.getRunningJobs(ownerFilter); if (jobsToWatch.length === 0) { if (cancelOutcomes.length > 0) { - const cancelledJobs = cancelIds.map(id => manager.getJob(id)).filter(j => j != null); + const cancelledJobs = this.#visibleJobs(manager, cancelIds, ownerId); return this.#buildResult(manager, cancelledJobs, cancelOutcomes); } const message = requestedPollIds?.length @@ -176,7 +196,7 @@ export class JobTool implements AgentTool { const watchedJobIds = runningJobs.map(job => job.id); manager.watchJobs(watchedJobIds); - const cancelledJobs = cancelIds.map(id => manager.getJob(id)).filter(j => j != null); + const cancelledJobs = this.#visibleJobs(manager, cancelIds, ownerId); const allTrackedJobs = [...cancelledJobs, ...jobsToWatch]; const PROGRESS_INTERVAL_MS = 500; @@ -219,6 +239,22 @@ export class JobTool implements AgentTool { return this.#buildResult(manager, allTrackedJobs, cancelOutcomes); } + /** + * Resolve a list of job ids to job records visible to the calling agent. + * Drops missing ids and ids owned by other agents, so cross-agent inspection + * via the `job` tool is impossible. + */ + #visibleJobs(manager: AsyncJobManager, ids: string[], ownerId: string | undefined): AsyncJob[] { + const out: AsyncJob[] = []; + for (const id of ids) { + const job = manager.getJob(id); + if (!job) continue; + if (ownerId && job.ownerId !== ownerId) continue; + out.push(job); + } + return out; + } + #snapshotJobs( jobs: { id: string; @@ -232,7 +268,7 @@ export class JobTool implements AgentTool { ): JobSnapshot[] { const now = Date.now(); return jobs.map(j => { - const current = this.session.asyncJobManager?.getJob(j.id); + const current = AsyncJobManager.instance()?.getJob(j.id); const latest = current ?? j; return { id: latest.id, @@ -247,7 +283,7 @@ export class JobTool implements AgentTool { } #buildResult( - manager: NonNullable, + manager: AsyncJobManager, jobs: { id: string; type: "bash" | "task"; diff --git a/packages/coding-agent/src/tools/read.ts b/packages/coding-agent/src/tools/read.ts index a18e1ca44..b984159b4 100644 --- a/packages/coding-agent/src/tools/read.ts +++ b/packages/coding-agent/src/tools/read.ts @@ -12,6 +12,7 @@ import { getFileReadCache } from "../edit/file-read-cache"; import { isNotebookPath, readEditableNotebookText } from "../edit/notebook"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; import { formatHashLine, formatHashLines, formatLineHash, HL_BODY_SEP } from "../hashline/hash"; +import { InternalUrlRouter } from "../internal-urls"; import { parseInternalUrl } from "../internal-urls/parse"; import type { InternalUrl } from "../internal-urls/types"; import { getLanguageFromPath, type Theme } from "../modes/theme/theme"; @@ -431,7 +432,7 @@ function prependSuffixResolutionNotice(text: string, suffixResolution?: { from: const readSchema = Type.Object({ path: Type.String({ description: 'path or url; append : for line ranges or raw mode (e.g. "src/foo.ts:50-100")', - examples: ["src/foo.ts", "src/foo.ts:50-100", "https://example.com:L1-L40"], + examples: ["src/foo.ts", "src/foo.ts:50-100", "https://example.com/:1-40"], }), }); @@ -1181,8 +1182,8 @@ export class ReadTool implements AgentTool { // Handle internal URLs (agent://, artifact://, memory://, skill://, rule://, local://, mcp://) const internalTarget = splitPathAndSel(readPath); - const internalRouter = this.session.internalRouter; - if (internalRouter?.canHandle(internalTarget.path)) { + const internalRouter = InternalUrlRouter.instance(); + if (internalRouter.canHandle(internalTarget.path)) { const parsed = parseSel(internalTarget.sel); const { offset, limit } = selToOffsetLimit(parsed); return this.#handleInternalUrl(internalTarget.path, offset, limit, { raw: isRawSelector(parsed) }); @@ -1551,7 +1552,7 @@ export class ReadTool implements AgentTool { limit?: number, options?: { raw?: boolean }, ): Promise> { - const internalRouter = this.session.internalRouter!; + const internalRouter = InternalUrlRouter.instance(); // Check if URL has query extraction (agent:// only). // Use parseInternalUrl which handles colons in host (namespaced skills). diff --git a/packages/coding-agent/src/tools/search.ts b/packages/coding-agent/src/tools/search.ts index 52e8ad3d3..9b15de1df 100644 --- a/packages/coding-agent/src/tools/search.ts +++ b/packages/coding-agent/src/tools/search.ts @@ -8,6 +8,7 @@ import { prompt, untilAborted } from "@oh-my-pi/pi-utils"; import { type Static, Type } from "@sinclair/typebox"; import { getFileReadCache } from "../edit/file-read-cache"; import type { RenderResultOptions } from "../extensibility/custom-tools/types"; +import { InternalUrlRouter } from "../internal-urls"; import type { Theme } from "../modes/theme/theme"; import searchDescription from "../prompts/tools/search.md" with { type: "text" }; import { DEFAULT_MAX_COLUMN, type TruncationResult, truncateHead } from "../session/streaming-output"; @@ -131,14 +132,14 @@ export class SearchTool implements AgentTool rawPath.length === 0)) { throw new ToolError("`paths` must contain non-empty paths or globs"); } - const internalRouter = this.session.internalRouter; + const internalRouter = InternalUrlRouter.instance(); const resolvedPathInputs: string[] = []; // Absolute filesystem paths whose source is immutable (e.g. artifact://, // pi://, skill://). Hashline anchors are suppressed for these on a // per-file basis, leaving editable mixed-in files untouched. const immutableSourcePaths = new Set(); for (const rawPath of rawPaths) { - if (!internalRouter?.canHandle(rawPath)) { + if (!internalRouter.canHandle(rawPath)) { resolvedPathInputs.push(rawPath); continue; } diff --git a/packages/coding-agent/src/tools/todo-write.ts b/packages/coding-agent/src/tools/todo-write.ts index 05b953322..fea7c3bb0 100644 --- a/packages/coding-agent/src/tools/todo-write.ts +++ b/packages/coding-agent/src/tools/todo-write.ts @@ -631,7 +631,7 @@ function renderNoteAttachments(phases: TodoPhase[], uiTheme: Theme): string[] { for (const task of phase.tasks) { if (task.status !== "in_progress" || !task.notes || task.notes.length === 0) continue; const bar = uiTheme.fg("dim", uiTheme.tree.vertical); - const title = uiTheme.fg("dim", chalk.italic(`\u00a7 notes \u2014 ${task.content}`)); + const title = uiTheme.fg("dim", chalk.italic(`§ notes — ${task.content}`)); lines.push(""); lines.push(` ${title}`); for (let j = 0; j < task.notes.length; j++) { diff --git a/packages/coding-agent/src/web/scrapers/mastodon.ts b/packages/coding-agent/src/web/scrapers/mastodon.ts index 0fdfb6451..8c439bcdb 100644 --- a/packages/coding-agent/src/web/scrapers/mastodon.ts +++ b/packages/coding-agent/src/web/scrapers/mastodon.ts @@ -273,7 +273,7 @@ export const handleMastodon: SpecialHandler = async ( md += `### ${formatDate(status.created_at)}\n\n`; const content = await htmlToBasicMarkdown(status.content); md += `${content}\n\n`; - md += `\uD83D\uDCAC ${status.replies_count} \u00B7 \uD83D\uDD01 ${status.reblogs_count} \u00B7 \u2B50 ${status.favourites_count}\n\n`; + md += `💬 ${status.replies_count} · 🔁 ${status.reblogs_count} · ⭐ ${status.favourites_count}\n\n`; } } } diff --git a/packages/coding-agent/src/web/scrapers/repology.ts b/packages/coding-agent/src/web/scrapers/repology.ts index 6ec686e72..aabde8d78 100644 --- a/packages/coding-agent/src/web/scrapers/repology.ts +++ b/packages/coding-agent/src/web/scrapers/repology.ts @@ -32,19 +32,19 @@ interface RepologyPackage { function statusIndicator(status: string): string { switch (status) { case "newest": - return "\u2705"; // green check + return "✅"; // green check case "devel": - return "\uD83D\uDEA7"; // construction + return "🚧"; // construction case "unique": - return "\uD83D\uDD35"; // blue circle + return "🔵"; // blue circle case "outdated": - return "\uD83D\uDD34"; // red circle + return "🔴"; // red circle case "legacy": - return "\u26A0\uFE0F"; // warning + return "⚠\uFE0F"; // warning case "rolling": - return "\uD83D\uDD04"; // arrows + return "🔄"; // arrows default: - return "\u2796"; // minus + return "➖"; // minus } } diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 654139c4f..dc0cb1e83 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -346,89 +346,6 @@ describe("AgentSession retry fallback", () => { expect(lastAssistant.content).toContainEqual({ type: "text", text: "Recovered after OpenAI timeout" }); }); - it("auto-retries Bun socket closure errors", async () => { - const model = getBundledModel("openai", "gpt-4o-mini"); - if (!model) { - throw new Error("Expected bundled OpenAI test model to exist"); - } - - const socketError = - "The socket connection was closed unexpectedly. For more information, pass `verbose: true` in the second argument to fetch()"; - const requestedModels: string[] = []; - let attemptCount = 0; - - const agent = new Agent({ - getApiKey: provider => `${provider}-test-key`, - initialState: { - model, - systemPrompt: ["Test"], - tools: [], - messages: [], - }, - streamFn: requestedModel => { - requestedModels.push(`${requestedModel.provider}/${requestedModel.id}`); - const stream = new MockAssistantStream(); - queueMicrotask(() => { - attemptCount += 1; - if (attemptCount === 1) { - const message = createAssistantMessage(requestedModel, { - stopReason: "error", - errorMessage: socketError, - }); - stream.push({ type: "start", partial: message }); - stream.push({ type: "error", reason: "error", error: message }); - return; - } - if (attemptCount === 2) { - const message = createAssistantMessage(requestedModel, { - text: "Recovered after socket closure", - stopReason: "stop", - }); - stream.push({ - type: "start", - partial: createAssistantMessage(requestedModel, { text: "", stopReason: "stop" }), - }); - stream.push({ type: "done", reason: "stop", message }); - return; - } - throw new Error(`Unexpected retry attempt in socket closure test: ${attemptCount}`); - }); - return stream; - }, - }); - - const settings = Settings.isolated({ - "compaction.enabled": false, - "retry.baseDelayMs": 5, - "retry.maxRetries": 1, - }); - settings.setModelRole("default", `${model.provider}/${model.id}`); - - session = new AgentSession({ - agent, - sessionManager: SessionManager.inMemory(), - settings, - modelRegistry, - }); - const { retryStartEvents, retryEndEvents } = trackRetryEvents(session); - - await session.prompt("Retry Bun socket closure"); - await session.waitForIdle(); - - expect(requestedModels).toEqual([`${model.provider}/${model.id}`, `${model.provider}/${model.id}`]); - expect(retryStartEvents).toHaveLength(1); - expect(retryStartEvents[0]).toMatchObject({ - attempt: 1, - maxAttempts: 1, - errorMessage: socketError, - }); - expect(retryEndEvents).toHaveLength(1); - expect(retryEndEvents[0]).toMatchObject({ success: true, attempt: 1 }); - const lastAssistant = getLastAssistantMessage(session); - expect(lastAssistant.stopReason).toBe("stop"); - expect(lastAssistant.content).toContainEqual({ type: "text", text: "Recovered after socket closure" }); - }); - it("auto-retries Anthropic stream-envelope failures before message_start", async () => { const model = getBundledModel("anthropic", "claude-sonnet-4-5"); if (!model) { diff --git a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts index 3d11bdaed..4dad6d9a1 100644 --- a/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts +++ b/packages/coding-agent/test/agent-session-tool-rebuild-skip.test.ts @@ -219,7 +219,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { it("rebuilds when an MCP tool's label changes", async () => { // Tool labels are rendered into the prompt body (`{{label}}: \`{{name}}\``), - // so a label change \u2014 even with name and description constant \u2014 must force + // so a label change — even with name and description constant — must force // a rebuild. Otherwise we'd serve a stale label after an MCP server upgrade. let rebuildCount = 0; const { session } = newSession(async toolNames => { @@ -273,7 +273,7 @@ describe("AgentSession refreshMCPTools rebuild skipping", () => { it("rebuilds when a discoverable (inactive) registry tool's metadata changes", async () => { // With MCP discovery on, the rebuilt prompt summarizes ALL discoverable MCP - // tools \u2014 including ones not in the active set. The signature must capture the + // tools — including ones not in the active set. The signature must capture the // full registry; otherwise a description change to a discoverable-but-inactive // tool would silently leave a stale summary in the cached prompt. let rebuildCount = 0; diff --git a/packages/coding-agent/test/agent-session-user-shortcut-hooks.test.ts b/packages/coding-agent/test/agent-session-user-shortcut-hooks.test.ts index dcfe78d7b..989498d4c 100644 --- a/packages/coding-agent/test/agent-session-user-shortcut-hooks.test.ts +++ b/packages/coding-agent/test/agent-session-user-shortcut-hooks.test.ts @@ -10,7 +10,6 @@ import type { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/ex import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; -import { TOOL_TIMEOUTS } from "@oh-my-pi/pi-coding-agent/tools/tool-timeouts"; import { TempDir } from "@oh-my-pi/pi-utils"; describe("AgentSession user shortcut hooks", () => { @@ -178,29 +177,4 @@ describe("AgentSession user shortcut hooks", () => { session.messages.some(message => message.role === "pythonExecution" && message.excludeFromContext === false), ).toBe(true); }); - - it("passes the default native timeout to fallback bash execution", async () => { - vi.spyOn(bashExecutor, "executeBash").mockResolvedValue({ - output: "bash fallback", - exitCode: 0, - cancelled: false, - truncated: false, - totalLines: 1, - totalBytes: 13, - outputLines: 1, - outputBytes: 13, - }); - - createSession(); - await session.executeBash("pwd", undefined, { excludeFromContext: false }); - - expect(bashExecutor.executeBash).toHaveBeenCalledWith( - "pwd", - expect.objectContaining({ - timeout: TOOL_TIMEOUTS.bash.default * 1000, - sessionKey: expect.any(String), - signal: expect.any(AbortSignal), - }), - ); - }); }); diff --git a/packages/coding-agent/test/args.test.ts b/packages/coding-agent/test/args.test.ts deleted file mode 100644 index 4c4cec434..000000000 --- a/packages/coding-agent/test/args.test.ts +++ /dev/null @@ -1,288 +0,0 @@ -import { describe, expect, test } from "bun:test"; -import { Effort } from "@oh-my-pi/pi-ai"; -import { parseArgs } from "@oh-my-pi/pi-coding-agent/cli/args"; - -describe("parseArgs", () => { - describe("--version flag", () => { - test("parses --version flag", () => { - const result = parseArgs(["--version"]); - expect(result.version).toBe(true); - }); - - test("parses -v shorthand", () => { - const result = parseArgs(["-v"]); - expect(result.version).toBe(true); - }); - - test("--version takes precedence over other args", () => { - const result = parseArgs(["--version", "--help", "some message"]); - expect(result.version).toBe(true); - expect(result.help).toBe(true); - expect(result.messages).toContain("some message"); - }); - }); - - describe("--help flag", () => { - test("parses --help flag", () => { - const result = parseArgs(["--help"]); - expect(result.help).toBe(true); - }); - - test("parses -h shorthand", () => { - const result = parseArgs(["-h"]); - expect(result.help).toBe(true); - }); - }); - - describe("--print flag", () => { - test("parses --print flag", () => { - const result = parseArgs(["--print"]); - expect(result.print).toBe(true); - }); - - test("parses -p shorthand", () => { - const result = parseArgs(["-p"]); - expect(result.print).toBe(true); - }); - }); - - describe("--continue flag", () => { - test("parses --continue flag", () => { - const result = parseArgs(["--continue"]); - expect(result.continue).toBe(true); - }); - - test("parses -c shorthand", () => { - const result = parseArgs(["-c"]); - expect(result.continue).toBe(true); - }); - }); - - describe("--resume flag", () => { - test("parses --resume flag", () => { - const result = parseArgs(["--resume"]); - expect(result.resume).toBe(true); - }); - - test("parses -r shorthand", () => { - const result = parseArgs(["-r"]); - expect(result.resume).toBe(true); - }); - - test("parses --resume with session ID", () => { - const result = parseArgs(["--resume", "abc123"]); - expect(result.resume).toBe("abc123"); - }); - - test("parses -r with session path", () => { - const result = parseArgs(["-r", "/path/to/session.jsonl"]); - expect(result.resume).toBe("/path/to/session.jsonl"); - }); - - test("--resume without value before another flag stays boolean", () => { - const result = parseArgs(["--resume", "--model", "opus"]); - expect(result.resume).toBe(true); - expect(result.model).toBe("opus"); - }); - }); - - describe("--fork flag", () => { - test("parses --fork with session ID", () => { - const result = parseArgs(["--fork", "abc123"]); - expect(result.fork).toBe("abc123"); - }); - }); - - describe("flags with values", () => { - test("parses --provider", () => { - const result = parseArgs(["--provider", "openai"]); - expect(result.provider).toBe("openai"); - }); - - test("parses --model", () => { - const result = parseArgs(["--model", "gpt-4o"]); - expect(result.model).toBe("gpt-4o"); - }); - - test("parses --model=value with equals syntax", () => { - const result = parseArgs(["--model=gpt-4o"]); - expect(result.model).toBe("gpt-4o"); - }); - - test("parses --api-key", () => { - const result = parseArgs(["--api-key", "sk-test-key"]); - expect(result.apiKey).toBe("sk-test-key"); - }); - - test("parses --system-prompt", () => { - const result = parseArgs(["--system-prompt", "You are a helpful assistant"]); - expect(result.systemPrompt).toBe("You are a helpful assistant"); - }); - - test("parses --append-system-prompt", () => { - const result = parseArgs(["--append-system-prompt", "Additional context"]); - expect(result.appendSystemPrompt).toBe("Additional context"); - }); - - test("parses --provider-session-id", () => { - const result = parseArgs(["--provider-session-id", "reb_cache_key"]); - expect(result.providerSessionId).toBe("reb_cache_key"); - }); - - test("parses --mode", () => { - const result = parseArgs(["--mode", "json"]); - expect(result.mode).toBe("json"); - }); - - test("parses --mode rpc", () => { - const result = parseArgs(["--mode", "rpc"]); - expect(result.mode).toBe("rpc"); - }); - - test("parses --session as alias for --resume", () => { - const result = parseArgs(["--session", "/path/to/session.jsonl"]); - expect(result.resume).toBe("/path/to/session.jsonl"); - }); - - test("parses --export", () => { - const result = parseArgs(["--export", "session.jsonl"]); - expect(result.export).toBe("session.jsonl"); - }); - - test("parses --thinking", () => { - const result = parseArgs(["--thinking", "high"]); - expect(result.thinking).toBe(Effort.High); - }); - - test("parses --models as comma-separated list", () => { - const result = parseArgs(["--models", "gpt-4o,claude-sonnet,gemini-pro"]); - expect(result.models).toEqual(["gpt-4o", "claude-sonnet", "gemini-pro"]); - }); - }); - - describe("--no-session flag", () => { - test("parses --no-session flag", () => { - const result = parseArgs(["--no-session"]); - expect(result.noSession).toBe(true); - }); - }); - - describe("--hook flag", () => { - test("parses single --hook", () => { - const result = parseArgs(["--hook", "./my-hook.ts"]); - expect(result.hooks).toEqual(["./my-hook.ts"]); - }); - - test("parses multiple --hook flags", () => { - const result = parseArgs(["--hook", "./hook1.ts", "--hook", "./hook2.ts"]); - expect(result.hooks).toEqual(["./hook1.ts", "./hook2.ts"]); - }); - }); - - describe("--no-extensions flag", () => { - test("parses --no-extensions flag", () => { - const result = parseArgs(["--no-extensions"]); - expect(result.noExtensions).toBe(true); - }); - - test("parses --no-extensions with explicit -e flags", () => { - const result = parseArgs(["--no-extensions", "-e", "foo.ts", "-e", "bar.ts"]); - expect(result.noExtensions).toBe(true); - expect(result.extensions).toEqual(["foo.ts", "bar.ts"]); - }); - }); - - describe("--no-skills flag", () => { - test("parses --no-skills flag", () => { - const result = parseArgs(["--no-skills"]); - expect(result.noSkills).toBe(true); - }); - }); - - describe("--no-rules flag", () => { - test("parses --no-rules flag", () => { - const result = parseArgs(["--no-rules"]); - expect(result.noRules).toBe(true); - }); - }); - - describe("--no-tools flag", () => { - test("parses --no-tools flag", () => { - const result = parseArgs(["--no-tools"]); - expect(result.noTools).toBe(true); - }); - - test("parses --no-tools with explicit --tools flags", () => { - const result = parseArgs(["--no-tools", "--tools", "read,bash"]); - expect(result.noTools).toBe(true); - expect(result.tools).toEqual(["read", "bash"]); - }); - - test("lowercases tool names passed to --tools", () => { - const result = parseArgs(["--tools", "Read,Search"]); - expect(result.tools).toEqual(["read", "search"]); - }); - - test("parses --tools=value with equals syntax", () => { - const result = parseArgs(["--tools=read,bash"]); - expect(result.tools).toEqual(["read", "bash"]); - }); - - test("parses --tools=value with single tool", () => { - const result = parseArgs(["--tools=ask"]); - expect(result.tools).toEqual(["ask"]); - }); - }); - - describe("--no-lsp flag", () => { - test("parses --no-lsp flag", () => { - const result = parseArgs(["--no-lsp"]); - expect(result.noLsp).toBe(true); - }); - }); - - describe("messages and file args", () => { - test("parses plain text messages", () => { - const result = parseArgs(["hello", "world"]); - expect(result.messages).toEqual(["hello", "world"]); - }); - - test("parses @file arguments", () => { - const result = parseArgs(["@README.md", "@src/main.ts"]); - expect(result.fileArgs).toEqual(["README.md", "src/main.ts"]); - }); - - test("parses mixed messages and file args", () => { - const result = parseArgs(["@file.txt", "explain this", "@image.png"]); - expect(result.fileArgs).toEqual(["file.txt", "image.png"]); - expect(result.messages).toEqual(["explain this"]); - }); - - test("ignores unknown flags starting with -", () => { - const result = parseArgs(["--unknown-flag", "message"]); - expect(result.messages).toEqual(["message"]); - }); - }); - - describe("complex combinations", () => { - test("parses multiple flags together", () => { - const result = parseArgs([ - "--provider", - "anthropic", - "--model", - "claude-sonnet", - "--print", - "--thinking", - "high", - "@prompt.md", - "Do the task", - ]); - expect(result.provider).toBe("anthropic"); - expect(result.model).toBe("claude-sonnet"); - expect(result.print).toBe(true); - expect(result.thinking).toBe(Effort.High); - expect(result.fileArgs).toEqual(["prompt.md"]); - expect(result.messages).toEqual(["Do the task"]); - }); - }); -}); diff --git a/packages/coding-agent/test/async-job-manager.test.ts b/packages/coding-agent/test/async-job-manager.test.ts index 1861d92da..ac1b6d9eb 100644 --- a/packages/coding-agent/test/async-job-manager.test.ts +++ b/packages/coding-agent/test/async-job-manager.test.ts @@ -229,17 +229,48 @@ describe("AsyncJobManager", () => { expect(manager.hasPendingDeliveries()).toBe(false); }); - test("uses custom id when provided and deduplicates collisions", async () => { + test("cancelAll with ownerId only cancels matching jobs", async () => { const manager = new AsyncJobManager({ onJobComplete: async () => {}, }); - const first = manager.register("task", "alpha", async () => "ok", { id: "2-CheckRustCrate" }); - const second = manager.register("task", "beta", async () => "ok", { id: "2-CheckRustCrate" }); + const hold = (signal: AbortSignal) => + new Promise(resolve => { + signal.addEventListener("abort", () => resolve(), { once: true }); + }); - expect(first).toBe("2-CheckRustCrate"); - expect(second).toBe("2-CheckRustCrate-2"); + const parentJobId = manager.register( + "bash", + "parent-job", + async ({ signal }) => { + await hold(signal); + return "parent-cancelled"; + }, + { ownerId: "0-Main" }, + ); + const subagentJobId = manager.register( + "bash", + "subagent-job", + async ({ signal }) => { + await hold(signal); + return "subagent-cancelled"; + }, + { ownerId: "3-AuthLoader" }, + ); + manager.cancelAll({ ownerId: "3-AuthLoader" }); + + expect(manager.getJob(parentJobId)?.status).toBe("running"); + expect(manager.getJob(subagentJobId)?.status).toBe("cancelled"); + + // Filtered query mirrors filtered cancel. + expect(manager.getRunningJobs({ ownerId: "0-Main" }).map(j => j.id)).toEqual([parentJobId]); + expect(manager.getRunningJobs({ ownerId: "3-AuthLoader" })).toEqual([]); + expect(manager.getAllJobs({ ownerId: "0-Main" }).map(j => j.id)).toEqual([parentJobId]); + + // Unscoped cancelAll still cleans up everything. + manager.cancelAll(); await manager.waitForAll(); + expect(manager.getJob(parentJobId)?.status).toBe("cancelled"); }); }); diff --git a/packages/coding-agent/test/autocomplete-max-visible.test.ts b/packages/coding-agent/test/autocomplete-max-visible.test.ts index 3ebd767b9..56fc688a5 100644 --- a/packages/coding-agent/test/autocomplete-max-visible.test.ts +++ b/packages/coding-agent/test/autocomplete-max-visible.test.ts @@ -3,7 +3,6 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { _resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { getDefault } from "@oh-my-pi/pi-coding-agent/config/settings-schema"; import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller"; import { getProjectAgentDir, Snowflake } from "@oh-my-pi/pi-utils"; import { YAML } from "bun"; @@ -29,15 +28,6 @@ describe("autocompleteMaxVisible setting", () => { } }); - it("should have default value of 5", () => { - expect(getDefault("autocompleteMaxVisible")).toBe(5); - }); - - it("should return default when not configured", async () => { - const settings = await Settings.init({ cwd: projectDir, agentDir }); - expect(settings.get("autocompleteMaxVisible")).toBe(5); - }); - it("should persist and read back a configured value", async () => { const settings = await Settings.init({ cwd: projectDir, agentDir }); settings.set("autocompleteMaxVisible", 10); diff --git a/packages/coding-agent/test/block-images.test.ts b/packages/coding-agent/test/block-images.test.ts index cec63ea1d..bb8d8319e 100644 --- a/packages/coding-agent/test/block-images.test.ts +++ b/packages/coding-agent/test/block-images.test.ts @@ -22,38 +22,6 @@ function createTestToolSession(cwd: string, settings: Settings = Settings.isolat } describe("blockImages setting", () => { - describe("Settings", () => { - it("should default blockImages to false", () => { - const settings = Settings.isolated({}); - expect(settings.get("images.blockImages")).toBe(false); - }); - - it("should return true when blockImages is set to true", () => { - const settings = Settings.isolated({ "images.blockImages": true }); - expect(settings.get("images.blockImages")).toBe(true); - }); - - it("should persist blockImages setting via set", () => { - const settings = Settings.isolated({}); - expect(settings.get("images.blockImages")).toBe(false); - - settings.set("images.blockImages", true); - expect(settings.get("images.blockImages")).toBe(true); - - settings.set("images.blockImages", false); - expect(settings.get("images.blockImages")).toBe(false); - }); - - it("should handle blockImages alongside autoResize", () => { - const settings = Settings.isolated({ - "images.autoResize": true, - "images.blockImages": true, - }); - expect(settings.get("images.autoResize")).toBe(true); - expect(settings.get("images.blockImages")).toBe(true); - }); - }); - describe("Read tool", () => { let testDir: string; diff --git a/packages/coding-agent/test/compaction-hooks-example.test.ts b/packages/coding-agent/test/compaction-hooks-example.test.ts deleted file mode 100644 index c88471bc5..000000000 --- a/packages/coding-agent/test/compaction-hooks-example.test.ts +++ /dev/null @@ -1,70 +0,0 @@ -/** - * Verify the documentation example from hooks.md compiles and works. - */ - -import { describe, expect, it } from "bun:test"; -import type { - HookAPI, - SessionBeforeCompactEvent, - SessionCompactEvent, -} from "@oh-my-pi/pi-coding-agent/extensibility/hooks"; - -describe("Documentation example", () => { - it("custom compaction example should type-check correctly", () => { - // This is the example from hooks.md - verify it compiles - const exampleHook = (pi: HookAPI) => { - pi.on("session_before_compact", async (event: SessionBeforeCompactEvent, ctx) => { - // All these should be accessible on the event - const { preparation, branchEntries } = event; - // sessionManager, modelRegistry, and model come from ctx - const { sessionManager, modelRegistry } = ctx; - const { messagesToSummarize, turnPrefixMessages, tokensBefore, firstKeptEntryId, isSplitTurn } = - preparation; - - // Verify types - expect(Array.isArray(messagesToSummarize)).toBe(true); - expect(Array.isArray(turnPrefixMessages)).toBe(true); - expect(typeof isSplitTurn).toBe("boolean"); - expect(typeof tokensBefore).toBe("number"); - expect(typeof sessionManager.getEntries).toBe("function"); - expect(typeof modelRegistry.getApiKey).toBe("function"); - expect(typeof firstKeptEntryId).toBe("string"); - expect(Array.isArray(branchEntries)).toBe(true); - - const summary = messagesToSummarize - .filter(m => m.role === "user") - .map(m => `- ${typeof m.content === "string" ? m.content.slice(0, 100) : "[complex]"}`) - .join("\n"); - - // Hooks return compaction content - SessionManager adds id/parentId - return { - compaction: { - summary: `User requests:\n${summary}`, - firstKeptEntryId, - tokensBefore, - }, - }; - }); - }; - - // Just verify the function exists and is callable - expect(typeof exampleHook).toBe("function"); - }); - - it("compact event should have correct fields", () => { - const checkCompactEvent = (pi: HookAPI) => { - pi.on("session_compact", async (event: SessionCompactEvent) => { - // These should all be accessible - const entry = event.compactionEntry; - const fromExtension = event.fromExtension; - - expect(entry.type).toBe("compaction"); - expect(typeof entry.summary).toBe("string"); - expect(typeof entry.tokensBefore).toBe("number"); - expect(typeof fromExtension).toBe("boolean"); - }); - }; - - expect(typeof checkCompactEvent).toBe("function"); - }); -}); diff --git a/packages/coding-agent/test/compaction.test.ts b/packages/coding-agent/test/compaction.test.ts index 433e4bfc3..cb3b2f2cb 100644 --- a/packages/coding-agent/test/compaction.test.ts +++ b/packages/coding-agent/test/compaction.test.ts @@ -325,40 +325,6 @@ describe("remote compaction setting", () => { } }); - it("leaves local summarization requests unattributed when no override is provided", async () => { - const model = getBundledModel("anthropic", "claude-sonnet-4-5"); - if (!model) throw new Error("Expected anthropic/claude-sonnet-4-5 model to exist"); - - const entries: SessionEntry[] = [ - createMessageEntry(createUserMessage("Turn 1")), - createMessageEntry(createAssistantMessage("Answer 1", createMockUsage(0, 100, 2000, 0))), - createMessageEntry(createUserMessage("Turn 2")), - createMessageEntry(createAssistantMessage("Answer 2", createMockUsage(0, 100, 5000, 0))), - createMessageEntry(createUserMessage("Turn 3")), - createMessageEntry(createAssistantMessage("Answer 3", createMockUsage(0, 100, 9000, 0))), - ]; - const preparation = prepareCompaction(entries, { - ...DEFAULT_COMPACTION_SETTINGS, - keepRecentTokens: 1000, - remoteEnabled: false, - }); - if (!preparation) throw new Error("Expected compaction preparation"); - - const completeSimpleSpy = vi.spyOn(ai, "completeSimple"); - completeSimpleSpy - .mockResolvedValueOnce(createAssistantMessage("History summary")) - .mockResolvedValueOnce(createAssistantMessage("Turn prefix summary")) - .mockResolvedValueOnce(createAssistantMessage("Short summary")); - - await compact(preparation, model, "test-api-key"); - - expect(completeSimpleSpy).toHaveBeenCalledTimes(3); - for (const call of completeSimpleSpy.mock.calls) { - const options = call[2] as { initiatorOverride?: string } | undefined; - expect(options?.initiatorOverride).toBeUndefined(); - } - }); - it("uses local summarization when remote compaction is disabled", async () => { const model = getBundledModel("openai", "gpt-4o"); if (!model) { @@ -906,14 +872,6 @@ describe("buildSessionContext", () => { // ============================================================================ describe("Large session fixture", () => { - it("should parse the large session", async () => { - const entries = await loadLargeSessionEntries(); - expect(entries.length).toBeGreaterThan(100); - - const messageCount = entries.filter(e => e.type === "message").length; - expect(messageCount).toBeGreaterThan(100); - }); - it("should find cut point in large session", async () => { const entries = await loadLargeSessionEntries(); const result = findCutPoint(entries, 0, entries.length, DEFAULT_COMPACTION_SETTINGS.keepRecentTokens); @@ -923,14 +881,6 @@ describe("Large session fixture", () => { const role = (entries[result.firstKeptEntryIndex] as SessionMessageEntry).message.role; expect(role === "user" || role === "assistant").toBe(true); }); - - it("should load session correctly", async () => { - const entries = await loadLargeSessionEntries(); - const loaded = buildSessionContext(entries); - - expect(loaded.messages.length).toBeGreaterThan(100); - expect(Object.keys(loaded.models).length).toBeGreaterThan(0); - }); }); // ============================================================================ diff --git a/packages/coding-agent/test/config-cli.test.ts b/packages/coding-agent/test/config-cli.test.ts index 7e60997ff..ae67a50a2 100644 --- a/packages/coding-agent/test/config-cli.test.ts +++ b/packages/coding-agent/test/config-cli.test.ts @@ -29,40 +29,6 @@ afterEach(async () => { }); describe("config CLI schema coverage", () => { - it("lists non-UI schema settings in JSON output", async () => { - const logSpy = vi.spyOn(console, "log").mockImplementation(() => {}); - - await runConfigCommand({ action: "list", flags: { json: true } }); - - expect(logSpy).toHaveBeenCalledTimes(1); - const payload = logSpy.mock.calls[0]?.[0]; - expect(typeof payload).toBe("string"); - const parsed = JSON.parse(String(payload)) as Record; - - expect(parsed.enabledModels).toBeDefined(); - expect(parsed.enabledModels.type).toBe("array"); - expect(parsed.enabledModels.description).toBe(""); - }); - - it("gets non-UI schema settings by key", async () => { - const logSpy = vi.spyOn(console, "log").mockImplementation(() => {}); - - await runConfigCommand({ action: "get", key: "enabledModels", flags: { json: true } }); - - expect(logSpy).toHaveBeenCalledTimes(1); - const payload = logSpy.mock.calls[0]?.[0]; - expect(typeof payload).toBe("string"); - const parsed = JSON.parse(String(payload)) as { - key: string; - type: string; - description: string; - }; - - expect(parsed.key).toBe("enabledModels"); - expect(parsed.type).toBe("array"); - expect(parsed.description).toBe(""); - }); - it("renders record settings as JSON and with record type in text output", async () => { const logSpy = vi.spyOn(console, "log").mockImplementation(() => {}); diff --git a/packages/coding-agent/test/config-spacing.test.ts b/packages/coding-agent/test/config-spacing.test.ts index 18e3d0de6..bab13e82e 100644 --- a/packages/coding-agent/test/config-spacing.test.ts +++ b/packages/coding-agent/test/config-spacing.test.ts @@ -21,18 +21,6 @@ describe("indentation resolver", () => { await fs.rm(tempDir, { recursive: true, force: true }); }); - it("falls back to hard default when settings are not initialized", () => { - expect(getDefaultTabWidth()).toBe(3); - expect(getIndentation()).toBe(3); - }); - - it("uses configured default tab width from settings", async () => { - const runtimeSettings = await Settings.init({ inMemory: true, cwd: tempDir }); - runtimeSettings.set("display.tabWidth", 5); - expect(getDefaultTabWidth()).toBe(5); - expect(getIndentation()).toBe(5); - }); - it("applies current display tab width during initial settings load", async () => { await Settings.init({ inMemory: true, cwd: tempDir, overrides: { "display.tabWidth": 7 } }); expect(getDefaultTabWidth()).toBe(7); diff --git a/packages/coding-agent/test/core/js-executor.test.ts b/packages/coding-agent/test/core/js-executor.test.ts index 5924aa764..25132d9f2 100644 --- a/packages/coding-agent/test/core/js-executor.test.ts +++ b/packages/coding-agent/test/core/js-executor.test.ts @@ -272,15 +272,4 @@ describe("executeJs", () => { expect(result.exitCode).toBe(0); expect(result.output.trim()).toBe(path.join("a", "b")); }); - - it("exposes a cwd-bound `require` and `createRequire`", async () => { - const result = await executeJs( - 'return { hasRequire: typeof require === "function", hasCreate: typeof createRequire === "function", path: require("node:path").sep };', - { sessionId, session, sessionFile }, - ); - expect(result.exitCode).toBe(0); - expect(result.displayOutputs).toEqual([ - { type: "json", data: { hasRequire: true, hasCreate: true, path: path.sep } }, - ]); - }); }); diff --git a/packages/coding-agent/test/core/python-kernel-env.test.ts b/packages/coding-agent/test/core/python-kernel-env.test.ts index 1d56f0f46..9d44d3c15 100644 --- a/packages/coding-agent/test/core/python-kernel-env.test.ts +++ b/packages/coding-agent/test/core/python-kernel-env.test.ts @@ -41,18 +41,4 @@ describe("Python gateway environment filtering", () => { expect(filtered.LC_CTYPE).toBe("UTF-8"); expect(filtered.LC_MESSAGES).toBe("en_US.UTF-8"); }); - - it("passes filtered env through to resolved runtime", () => { - const env: Record = { - PATH: "/usr/bin", - HOME: "/home/test", - OPENAI_API_KEY: "secret", - PI_DEBUG: "1", - }; - - const filtered = filterEnv(env); - expect(filtered.OPENAI_API_KEY).toBeUndefined(); - expect(filtered.PATH).toBe("/usr/bin"); - expect(filtered.PI_DEBUG).toBe("1"); - }); }); diff --git a/packages/coding-agent/test/core/python-prelude.test.ts b/packages/coding-agent/test/core/python-prelude.test.ts deleted file mode 100644 index 39b4757e8..000000000 --- a/packages/coding-agent/test/core/python-prelude.test.ts +++ /dev/null @@ -1,73 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import * as fs from "node:fs"; -import * as path from "node:path"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { EvalTool } from "@oh-my-pi/pi-coding-agent/tools/eval"; -import { $which, getProjectDir } from "@oh-my-pi/pi-utils"; - -const resolvePythonPath = (): string | null => { - const venvPath = Bun.env.VIRTUAL_ENV; - const candidates = [venvPath, path.join(getProjectDir(), ".venv"), path.join(getProjectDir(), "venv")].filter( - Boolean, - ) as string[]; - for (const candidate of candidates) { - const binDir = process.platform === "win32" ? "Scripts" : "bin"; - const exeName = process.platform === "win32" ? "python.exe" : "python"; - const pythonCandidate = path.join(candidate, binDir, exeName); - if (fs.existsSync(pythonCandidate)) { - return pythonCandidate; - } - } - return $which("python") ?? $which("python3"); -}; - -const pythonPath = resolvePythonPath(); -const hasKernelDeps = (() => { - if (!pythonPath) return false; - const result = Bun.spawnSync( - [ - pythonPath, - "-c", - "import importlib.util,sys;sys.exit(0 if importlib.util.find_spec('kernel_gateway') and importlib.util.find_spec('ipykernel') else 1)", - ], - { stdin: "ignore", stdout: "pipe", stderr: "pipe" }, - ); - return result.exitCode === 0; -})(); - -const shouldRun = Boolean(pythonPath) && hasKernelDeps; - -describe.skipIf(!shouldRun)("PYTHON_PRELUDE integration", () => { - it("exposes prelude helpers via eval python backend", async () => { - const helpers = ["env", "read", "write", "append", "tree", "diff", "run", "output"]; - - const session = { - cwd: getProjectDir(), - hasUI: false, - getSessionFile: () => null, - getSessionSpawns: () => null, - settings: Settings.isolated({ - "lsp.diagnosticsOnWrite": false, - "eval.py": true, - "python.kernelMode": "per-call", - "python.sharedGateway": true, - }), - }; - const tool = new EvalTool(session); - const code = [ - `helpers = ${JSON.stringify(helpers)}`, - "missing = [name for name in helpers if name not in globals() or not callable(globals()[name])]", - 'print("HELPERS_OK=" + ("1" if not missing else "0"))', - "if missing:", - ' print("MISSING=" + ",".join(missing))', - ].join("\n"); - - const result = await tool.execute("tool-call-1", { - input: `*** Begin PY\n*** Title: prelude helpers\n${code}\n*** End PY\n`, - }); - const output = result.content.find(item => item.type === "text")?.text ?? ""; - expect(output).toContain("HELPERS_OK=1"); - expect(tool.description).toContain("read"); - expect(tool.description).not.toContain("Documentation unavailable"); - }); -}); diff --git a/packages/coding-agent/test/cycle-order-custom-roles.test.ts b/packages/coding-agent/test/cycle-order-custom-roles.test.ts deleted file mode 100644 index f28416e16..000000000 --- a/packages/coding-agent/test/cycle-order-custom-roles.test.ts +++ /dev/null @@ -1,32 +0,0 @@ -import { describe, expect, test } from "bun:test"; -import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; - -describe("cycleOrder with custom roles", () => { - test("cycleOrder setting accepts custom role names", () => { - const settings = Settings.isolated({ - cycleOrder: ["smol", "custom-fast", "default"], - }); - expect(settings.get("cycleOrder")).toEqual(["smol", "custom-fast", "default"]); - }); - - test("cycleOrder falls back to default when not set", () => { - const settings = Settings.isolated({}); - expect(settings.get("cycleOrder")).toEqual(["smol", "default", "slow"]); - }); - - test("modelTags can define custom role display info", () => { - const settings = Settings.isolated({ - modelTags: { - "custom-fast": { - name: "Fast Custom", - color: "warning", - }, - }, - }); - const modelTags = settings.get("modelTags") as Record; - expect(modelTags["custom-fast"]).toEqual({ - name: "Fast Custom", - color: "warning", - }); - }); -}); diff --git a/packages/coding-agent/test/debug/log-viewer.test.ts b/packages/coding-agent/test/debug/log-viewer.test.ts index 121c73102..16bce1360 100644 --- a/packages/coding-agent/test/debug/log-viewer.test.ts +++ b/packages/coding-agent/test/debug/log-viewer.test.ts @@ -19,12 +19,6 @@ describe("DebugLogViewerModel", () => { return row.kind; } }; - - it("defaults cursor to the newest log entry", () => { - const logs = ["alpha", "beta", "gamma"].join("\n"); - const model = new DebugLogViewerModel(logs, { processStartMs: Date.now() }); - expect(model.cursorLogIndex).toBe(2); - }); it("inserts session boundary warning between older and current-session logs", () => { const processStartMs = Date.parse("2026-02-14T12:00:00.000Z"); const logs = [ diff --git a/packages/coding-agent/test/edit-auto-generated-regressions.test.ts b/packages/coding-agent/test/edit-auto-generated-regressions.test.ts new file mode 100644 index 000000000..90a17123a --- /dev/null +++ b/packages/coding-agent/test/edit-auto-generated-regressions.test.ts @@ -0,0 +1,372 @@ +/** + * Regressions for two issues that together produced silent edit failures on + * auto-generated files: + * + * 1. The streaming auto-generated guard was gated behind `edit.streamingAbort` + * (default false), so it never fired in default configs. + * 2. `executeSinglePathEntries` (multi-edit, single-path orchestrator) + * swallowed per-entry exceptions and returned an aggregate result with + * no `isError` flag. The UI then fell through to the streaming preview + * branch and rendered the *proposed* diff, making a hard failure look + * indistinguishable from success. + */ + +import { afterEach, beforeEach, expect, it, vi } from "bun:test"; +import * as fs from "node:fs"; +import * as os from "node:os"; +import * as path from "node:path"; +import { Agent, type AgentTool } from "@oh-my-pi/pi-agent-core"; +import { type AssistantMessage, getBundledModel, type StopReason, type ToolCall } from "@oh-my-pi/pi-ai"; +import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"; +import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { _resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import { EditTool } from "@oh-my-pi/pi-coding-agent/edit"; +import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; +import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import * as autoGeneratedGuard from "@oh-my-pi/pi-coding-agent/tools/auto-generated-guard"; +import { ToolError } from "@oh-my-pi/pi-coding-agent/tools/tool-errors"; +import { Snowflake } from "@oh-my-pi/pi-utils"; +import { Type } from "@sinclair/typebox"; + +class MockAssistantStream extends AssistantMessageEventStream {} + +function createAssistantMessage(content: AssistantMessage["content"], stopReason: StopReason): AssistantMessage { + return { + role: "assistant", + content, + api: "anthropic-messages", + provider: "anthropic", + model: "mock", + usage: { + input: 0, + output: 0, + cacheRead: 0, + cacheWrite: 0, + totalTokens: 0, + cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, + }, + stopReason, + timestamp: Date.now(), + }; +} + +function createToolCall(id: string, args: Record): ToolCall { + return { type: "toolCall", id, name: "edit", arguments: args }; +} + +function lastAssistantMessage(messages: Array<{ role: string }>): AssistantMessage | undefined { + for (let i = messages.length - 1; i >= 0; i--) { + const msg = messages[i]; + if (msg.role === "assistant") return msg as AssistantMessage; + } + return undefined; +} + +function buildMockEditTool(): AgentTool { + const schema = Type.Object({ + path: Type.String(), + diff: Type.String(), + op: Type.Optional(Type.String()), + rename: Type.Optional(Type.String()), + }); + return { + name: "edit", + label: "Edit", + description: "", + parameters: schema, + async execute() { + return { content: [{ type: "text", text: "ok" }] }; + }, + }; +} + +function createSessionWith( + tempDir: string, + streamFn: Agent["streamFn"], + tool: AgentTool, + settingsOverrides: Record = {}, +): Promise<{ agent: Agent; session: AgentSession; authStorage: AuthStorage }> { + return (async () => { + const model = getBundledModel("anthropic", "claude-sonnet-4-5")!; + const agent = new Agent({ + getApiKey: () => "test-key", + initialState: { model, systemPrompt: ["Test"], tools: [tool] }, + streamFn, + }); + const sessionManager = SessionManager.inMemory(tempDir); + const settings = Settings.isolated(settingsOverrides); + const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db")); + authStorage.setRuntimeApiKey("anthropic", "test-key"); + const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml")); + return { + agent, + session: new AgentSession({ agent, sessionManager, settings, modelRegistry }), + authStorage, + }; + })(); +} + +function streamForSingleToolCall( + filePath: string, + diff: string, + abortSignalRef: { current?: AbortSignal }, +): Agent["streamFn"] { + let callIndex = 0; + return (_model, _context, options) => { + abortSignalRef.current = options?.signal; + const stream = new MockAssistantStream(); + const toolCallId = "call_edit_regression"; + let aborted = false; + + const notifyAbort = () => { + if (aborted) return; + aborted = true; + const partial = createToolCall(toolCallId, { path: filePath, diff }); + stream.push({ + type: "toolcall_delta", + contentIndex: 0, + delta: "", + partial: createAssistantMessage([partial], "stop"), + }); + stream.push({ type: "error", reason: "aborted", error: createAssistantMessage([], "aborted") }); + }; + options?.signal?.addEventListener("abort", notifyAbort, { once: true }); + + queueMicrotask(async () => { + if (callIndex > 0) { + const finalMessage = createAssistantMessage([{ type: "text", text: "done" }], "stop"); + stream.push({ type: "done", reason: "stop", message: finalMessage }); + callIndex++; + return; + } + stream.push({ type: "start", partial: createAssistantMessage([], "stop") }); + + const startCall = createToolCall(toolCallId, { path: filePath, diff: "" }); + stream.push({ + type: "toolcall_start", + contentIndex: 0, + partial: createAssistantMessage([startCall], "stop"), + }); + + // Emit the full diff in one delta to simplify timing; the abort path + // should still preempt before the call completes (or, if it loses the + // race, the next per-entry `assertEditableFile` is the hard stop). + let acc = ""; + for (const chunk of diff.match(/.{1,4}/g) ?? []) { + if (aborted) return; + acc += chunk; + const partial = createToolCall(toolCallId, { path: filePath, diff: acc }); + stream.push({ + type: "toolcall_delta", + contentIndex: 0, + delta: chunk, + partial: createAssistantMessage([partial], "stop"), + }); + await Bun.sleep(0); + } + if (aborted) return; + + const finalCall = createToolCall(toolCallId, { path: filePath, diff }); + const finalMessage = createAssistantMessage([finalCall], "toolUse"); + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall: finalCall, partial: finalMessage }); + stream.push({ type: "done", reason: "toolUse", message: finalMessage }); + callIndex++; + }); + + return stream; + }; +} + +let tempDir: string; + +beforeEach(() => { + tempDir = path.join(os.tmpdir(), `pi-edit-regressions-${Snowflake.next()}`); + fs.mkdirSync(tempDir, { recursive: true }); +}); + +afterEach(async () => { + if (tempDir) fs.rmSync(tempDir, { recursive: true, force: true }); +}); + +it("auto-generated streaming abort fires even when edit.streamingAbort is disabled", async () => { + // The setting `edit.streamingAbort` defaults to false. The auto-generated + // guard must NOT be gated by that flag — editing a generated file is never + // the user's intent regardless of the patch-preview verification setting. + const checkSpy = vi + .spyOn(autoGeneratedGuard, "assertEditableFile") + .mockRejectedValue(new ToolError("Cannot modify auto-generated file")); + + await Bun.write(path.join(tempDir, "generated.ts"), "// AUTO-GENERATED\nexport const x = 1;\n"); + const abortSignalRef: { current?: AbortSignal } = {}; + const streamFn = streamForSingleToolCall( + "generated.ts", + "@@\n-export const x = 1;\n+export const x = 2;\n", + abortSignalRef, + ); + + // Explicitly disable edit.streamingAbort to confirm the auto-generated path + // is independent of it. + const { agent, session, authStorage } = await createSessionWith(tempDir, streamFn, buildMockEditTool(), { + "edit.streamingAbort": false, + }); + const abortSpy = vi.spyOn(agent, "abort"); + + try { + await session.prompt("apply patch"); + + expect(checkSpy).toHaveBeenCalled(); + expect(abortSpy).toHaveBeenCalled(); + expect(abortSignalRef.current?.aborted ?? false).toBe(true); + const lastAssistant = lastAssistantMessage(session.state.messages); + expect(lastAssistant?.stopReason).toBe("aborted"); + } finally { + checkSpy.mockRestore(); + abortSpy.mockRestore(); + try { + await session.dispose(); + } finally { + authStorage.close(); + } + } +}); + +it("multi-entry edit on an auto-generated file surfaces isError + error text instead of silently faking success", async () => { + // Direct invocation of the patch-mode EditTool with two entries against an + // auto-generated file. The orchestrator (executeSinglePathEntries) must NOT + // swallow the per-entry errors — it must mark the aggregate result with + // isError: true so the renderer takes the error branch instead of falling + // through to the streaming preview (which displays the *proposed* diff and + // looks identical to success). + const generatedPath = path.join(tempDir, "generated.ts"); + await Bun.write( + generatedPath, + "// Code generated by sqlc. DO NOT EDIT.\nexport const foo = 1;\nexport const bar = 2;\n", + ); + + const originalVariant = Bun.env.PI_EDIT_VARIANT; + Bun.env.PI_EDIT_VARIANT = "patch"; + + // The auto-generated guard reads from the *global* settings singleton, so we + // must initialize it (the per-tool `Settings.isolated(...)` we pass into the + // EditTool isn't what the guard sees). + _resetSettingsForTest(); + await Settings.init({ inMemory: true, cwd: tempDir, overrides: { "edit.blockAutoGenerated": true } }); + + try { + const sessionFile = path.join(tempDir, "session.jsonl"); + const sessionDir = path.join(tempDir, "session"); + const session = { + cwd: tempDir, + hasUI: false, + getSessionFile: () => sessionFile, + getSessionSpawns: () => "*", + getArtifactsDir: () => sessionDir, + allocateOutputArtifact: async (toolType: string) => { + fs.mkdirSync(sessionDir, { recursive: true }); + return { id: "a-1", path: path.join(sessionDir, `a-1.${toolType}.log`) }; + }, + settings: Settings.isolated({ "edit.blockAutoGenerated": true }), + enableLsp: false, + } as ToolSession; + + const editTool = new EditTool(session); + const result = await editTool.execute("test-multi-entry", { + path: generatedPath, + edits: [ + { op: "update", diff: "@@\n-export const foo = 1;\n+export const foo = 11;\n" }, + { op: "update", diff: "@@\n-export const bar = 2;\n+export const bar = 22;\n" }, + ], + }); + + // File is untouched on disk. + const contentOnDisk = await Bun.file(generatedPath).text(); + expect(contentOnDisk).toContain("export const foo = 1;"); + expect(contentOnDisk).toContain("export const bar = 2;"); + + // Aggregate result must signal an error so the renderer doesn't draw + // the streaming preview as if it succeeded. + expect(result.isError).toBe(true); + + // Both per-entry failures must be preserved in the content text so the + // agent (and the error renderer) see the real cause. + const text = (result.content?.find(c => c.type === "text") as { text?: string } | undefined)?.text ?? ""; + const occurrences = text.match(/Cannot modify auto-generated file/g) ?? []; + expect(occurrences.length).toBe(2); + + // `details.diff` must not contain a fabricated diff that would mislead the + // renderer's preview-fallback branch into showing the proposed change. + const details = result.details as { diff?: string } | undefined; + expect(details?.diff ?? "").toBe(""); + } finally { + if (originalVariant === undefined) { + delete Bun.env.PI_EDIT_VARIANT; + } else { + Bun.env.PI_EDIT_VARIANT = originalVariant; + } + } +}); + +it("agent-loop propagates explicit isError from a tool result to the wire", async () => { + // Validates the boundary fix: `coerceToolResult` preserves a tool-self-reported + // `isError: true`, and agent-loop honors it (emits tool_execution_end with + // isError=true and constructs a tool-result message with isError=true). + const schema = Type.Object({ note: Type.Optional(Type.String()) }); + const errorTool: AgentTool = { + name: "edit", + label: "Edit", + description: "", + parameters: schema, + async execute() { + return { + content: [{ type: "text", text: "intentional non-throwing failure" }], + details: {}, + isError: true, + }; + }, + }; + + let callIndex = 0; + const streamFn: Agent["streamFn"] = (_model, _context, _options) => { + const stream = new MockAssistantStream(); + queueMicrotask(() => { + if (callIndex === 0) { + const toolCall = createToolCall("call_self_error", {}); + const finalMessage = createAssistantMessage([toolCall], "toolUse"); + stream.push({ type: "start", partial: createAssistantMessage([], "stop") }); + stream.push({ + type: "toolcall_start", + contentIndex: 0, + partial: createAssistantMessage([toolCall], "stop"), + }); + stream.push({ type: "toolcall_end", contentIndex: 0, toolCall, partial: finalMessage }); + stream.push({ type: "done", reason: "toolUse", message: finalMessage }); + } else { + const finalMessage = createAssistantMessage([{ type: "text", text: "done" }], "stop"); + stream.push({ type: "done", reason: "stop", message: finalMessage }); + } + callIndex++; + }); + return stream; + }; + + const { session, authStorage } = await createSessionWith(tempDir, streamFn, errorTool); + try { + await session.prompt("trigger self-error"); + + const toolResult = session.state.messages.find(m => m.role === "toolResult") as + | { isError?: boolean; content: Array<{ type: string; text?: string }> } + | undefined; + expect(toolResult).toBeDefined(); + expect(toolResult?.isError).toBe(true); + const text = toolResult?.content?.find(c => c.type === "text")?.text ?? ""; + expect(text).toContain("intentional non-throwing failure"); + } finally { + try { + await session.dispose(); + } finally { + authStorage.close(); + } + } +}); diff --git a/packages/coding-agent/test/edit-streaming-preview.test.ts b/packages/coding-agent/test/edit-streaming-preview.test.ts index c50086d9e..0a2b4a127 100644 --- a/packages/coding-agent/test/edit-streaming-preview.test.ts +++ b/packages/coding-agent/test/edit-streaming-preview.test.ts @@ -33,24 +33,6 @@ describe("dropIncompleteLastEdit", () => { }); }); -describe("apply_patch extractCompleteEdits", () => { - const strategy = EDIT_MODE_STRATEGIES.apply_patch; - - test("returns args unchanged (payload is plain text)", () => { - const args = { input: "*** Begin Patch\n*** Update File: a.ts\n@@\n-x\n+y\n*** End Patch\n" }; - expect(strategy.extractCompleteEdits(args, undefined)).toEqual(args); - }); -}); - -describe("vim extractCompleteEdits", () => { - const strategy = EDIT_MODE_STRATEGIES.vim; - - test("returns args unchanged (vim stream handled elsewhere)", () => { - const args = { file: "a.ts", steps: [] }; - expect(strategy.extractCompleteEdits(args, undefined)).toEqual(args); - }); -}); - describe("hashline streaming preview (multi-section)", () => { const strategy = EDIT_MODE_STRATEGIES.hashline; let tmpDir: string; diff --git a/packages/coding-agent/test/export/html-template-developer.test.ts b/packages/coding-agent/test/export/html-template-developer.test.ts new file mode 100644 index 000000000..b1f1916aa --- /dev/null +++ b/packages/coding-agent/test/export/html-template-developer.test.ts @@ -0,0 +1,18 @@ +import { describe, expect, it } from "bun:test"; +import { TEMPLATE } from "../../src/export/html/template.generated"; + +describe("HTML export template developer message support", () => { + it("renders developer-role messages in the main feed", () => { + expect(TEMPLATE).toContain("msg.role === 'developer'"); + expect(TEMPLATE).toContain("developer-message"); + }); + + it("labels developer entries in the sidebar tree", () => { + expect(TEMPLATE).toContain("tree-role-developer"); + expect(TEMPLATE).toContain("developer:"); + }); + + it("counts developer messages in header stats", () => { + expect(TEMPLATE).toContain("developerMessages"); + }); +}); diff --git a/packages/coding-agent/test/extensibility/custom-commands/ci-green.test.ts b/packages/coding-agent/test/extensibility/custom-commands/ci-green.test.ts index e2783c5e8..1133601e3 100644 --- a/packages/coding-agent/test/extensibility/custom-commands/ci-green.test.ts +++ b/packages/coding-agent/test/extensibility/custom-commands/ci-green.test.ts @@ -25,12 +25,6 @@ function createApi(): CustomCommandAPI { } describe("GreenCommand", () => { - it("exposes the /green command name", () => { - const command = new GreenCommand(createApi()); - - expect(command.name).toBe("green"); - }); - it("includes tag instructions when HEAD has a tag", async () => { vi.spyOn(git.ref, "tags").mockResolvedValue(["v0.1.0-alpha2"]); const command = new GreenCommand(createApi()); diff --git a/packages/coding-agent/test/hindsight-config.test.ts b/packages/coding-agent/test/hindsight-config.test.ts deleted file mode 100644 index bf24af1bf..000000000 --- a/packages/coding-agent/test/hindsight-config.test.ts +++ /dev/null @@ -1,89 +0,0 @@ -import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; -import { _resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { isHindsightConfigured, loadHindsightConfig } from "@oh-my-pi/pi-coding-agent/hindsight/config"; - -describe("loadHindsightConfig", () => { - beforeEach(() => { - _resetSettingsForTest(); - }); - - afterEach(() => { - vi.restoreAllMocks(); - }); - - it("returns sane defaults from an empty Settings", () => { - const settings = Settings.isolated(); - const cfg = loadHindsightConfig(settings, {}); - expect(cfg.hindsightApiUrl).toBe("http://localhost:8888"); // schema default - expect(cfg.recallBudget).toBe("mid"); - expect(cfg.retainMode).toBe("full-session"); - expect(cfg.recallTypes).toEqual(["world", "experience"]); - expect(cfg.autoRecall).toBe(true); - expect(cfg.autoRetain).toBe(true); - expect(cfg.scoping).toBe("per-project-tagged"); - }); - - it("env overrides win over settings", () => { - const settings = Settings.isolated({ - "hindsight.apiUrl": "http://settings.example", - "hindsight.autoRecall": true, - "hindsight.recallMaxTokens": 256, - "hindsight.scoping": "global", - "hindsight.retainMode": "full-session", - }); - const cfg = loadHindsightConfig(settings, { - HINDSIGHT_API_URL: "http://env.example", - HINDSIGHT_AUTO_RECALL: "false", - HINDSIGHT_RECALL_MAX_TOKENS: "9999", - HINDSIGHT_SCOPING: "per-project", - HINDSIGHT_RETAIN_MODE: "last-turn", - }); - expect(cfg.hindsightApiUrl).toBe("http://env.example"); - expect(cfg.autoRecall).toBe(false); - expect(cfg.recallMaxTokens).toBe(9999); - expect(cfg.scoping).toBe("per-project"); - expect(cfg.retainMode).toBe("last-turn"); - }); - - it("ignores invalid scoping values and falls back to the schema default", () => { - const settings = Settings.isolated(); - const cfg = loadHindsightConfig(settings, { HINDSIGHT_SCOPING: "garbage" }); - expect(cfg.scoping).toBe("per-project-tagged"); - }); - - it("ignores invalid retainMode env values and falls back to schema default", () => { - const settings = Settings.isolated(); - const cfg = loadHindsightConfig(settings, { HINDSIGHT_RETAIN_MODE: "garbage" }); - expect(cfg.retainMode).toBe("full-session"); - }); - - it("coerces non-numeric ints back to undefined so settings/default takes over", () => { - const settings = Settings.isolated({ "hindsight.recallMaxTokens": 512 }); - const cfg = loadHindsightConfig(settings, { HINDSIGHT_RECALL_MAX_TOKENS: "not-a-number" }); - expect(cfg.recallMaxTokens).toBe(512); - }); - - it("respects falsy boolean env strings", () => { - const settings = Settings.isolated(); - const cfg = loadHindsightConfig(settings, { - HINDSIGHT_AUTO_RECALL: "no", - HINDSIGHT_AUTO_RETAIN: "0", - }); - expect(cfg.autoRecall).toBe(false); - expect(cfg.autoRetain).toBe(false); - }); -}); - -describe("isHindsightConfigured", () => { - it("returns true when an apiUrl is set", () => { - const cfg = loadHindsightConfig(Settings.isolated({ "hindsight.apiUrl": "http://x" }), {}); - expect(isHindsightConfigured(cfg)).toBe(true); - }); - - it("returns false when apiUrl is missing", () => { - const cfg = loadHindsightConfig(Settings.isolated({ "hindsight.apiUrl": "" }), { - HINDSIGHT_API_URL: "", - }); - expect(isHindsightConfigured(cfg)).toBe(false); - }); -}); diff --git a/packages/coding-agent/test/image-b64poly.test.ts b/packages/coding-agent/test/image-b64poly.test.ts index 00d975fec..4b2fefbdb 100644 --- a/packages/coding-agent/test/image-b64poly.test.ts +++ b/packages/coding-agent/test/image-b64poly.test.ts @@ -3,10 +3,6 @@ import { describe, expect, it } from "bun:test"; describe("Buffer.toBase64", async () => { await import("../src/utils/image-resize"); - it("should be defined", () => { - expect(Buffer.prototype.toBase64).toBeDefined(); - }); - it("should return a base64 string", () => { const buffer = Buffer.from("Hello, world!"); expect(buffer.toBase64()).toBe("SGVsbG8sIHdvcmxkIQ=="); diff --git a/packages/coding-agent/test/image-input-normalization.test.ts b/packages/coding-agent/test/image-input-normalization.test.ts index 97eb8d1cf..495a3df70 100644 --- a/packages/coding-agent/test/image-input-normalization.test.ts +++ b/packages/coding-agent/test/image-input-normalization.test.ts @@ -7,16 +7,6 @@ describe("ensureSupportedImageInput", () => { vi.restoreAllMocks(); }); - test("returns supported image input unchanged", async () => { - const convertToPngSpy = vi.spyOn(imageConvert, "convertToPng"); - const input = { type: "image" as const, data: "abc", mimeType: "image/png" }; - - const result = await ensureSupportedImageInput(input); - - expect(result).toEqual(input); - expect(convertToPngSpy).not.toHaveBeenCalled(); - }); - test("converts unsupported image input to png", async () => { const convertToPngSpy = vi .spyOn(imageConvert, "convertToPng") diff --git a/packages/coding-agent/test/internal-urls/local-protocol.test.ts b/packages/coding-agent/test/internal-urls/local-protocol.test.ts index 6f1d7ab22..98bb5b7f8 100644 --- a/packages/coding-agent/test/internal-urls/local-protocol.test.ts +++ b/packages/coding-agent/test/internal-urls/local-protocol.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; @@ -7,7 +7,7 @@ import { LocalProtocolHandler, resolveLocalRoot, resolveLocalUrlToPath, -} from "../../src/internal-urls"; +} from "@oh-my-pi/pi-coding-agent/internal-urls"; async function withTempDir(fn: (dir: string) => Promise): Promise { const dir = await fs.mkdtemp(path.join(os.tmpdir(), "local-protocol-")); @@ -18,25 +18,28 @@ async function withTempDir(fn: (dir: string) => Promise): Promise { } } -function createRouter(options: { artifactsDir?: string | null; sessionId?: string | null }): InternalUrlRouter { - const router = new InternalUrlRouter(); - router.register( - new LocalProtocolHandler({ - getArtifactsDir: () => options.artifactsDir ?? null, - getSessionId: () => options.sessionId ?? null, - }), - ); - return router; -} - describe("LocalProtocolHandler", () => { + beforeEach(() => { + LocalProtocolHandler.resetOverrideForTests(); + InternalUrlRouter.resetForTests(); + }); + + afterEach(() => { + LocalProtocolHandler.resetOverrideForTests(); + InternalUrlRouter.resetForTests(); + }); + it("lists files at local://", async () => { await withTempDir(async tempDir => { const artifactsDir = path.join(tempDir, "artifacts"); await fs.mkdir(path.join(artifactsDir, "local"), { recursive: true }); await Bun.write(path.join(artifactsDir, "local", "handoff.json"), '{"ok":true}'); - const router = createRouter({ artifactsDir, sessionId: "session-a" }); + LocalProtocolHandler.setOverride({ + getArtifactsDir: () => artifactsDir, + getSessionId: () => "session-a", + }); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("local://"); expect(resource.contentType).toBe("text/markdown"); @@ -51,7 +54,11 @@ describe("LocalProtocolHandler", () => { await fs.mkdir(path.dirname(localFile), { recursive: true }); await Bun.write(localFile, "trace"); - const router = createRouter({ artifactsDir, sessionId: "session-b" }); + LocalProtocolHandler.setOverride({ + getArtifactsDir: () => artifactsDir, + getSessionId: () => "session-b", + }); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("local://subtasks/trace.txt"); expect(resource.content).toBe("trace"); @@ -61,7 +68,11 @@ describe("LocalProtocolHandler", () => { it("blocks path traversal attempts", async () => { await withTempDir(async tempDir => { - const router = createRouter({ artifactsDir: path.join(tempDir, "artifacts"), sessionId: "session-c" }); + LocalProtocolHandler.setOverride({ + getArtifactsDir: () => path.join(tempDir, "artifacts"), + getSessionId: () => "session-c", + }); + const router = InternalUrlRouter.instance(); await expect(router.resolve("local://../secret.txt")).rejects.toThrow( "Path traversal (..) is not allowed in local:// URLs", ); @@ -91,7 +102,11 @@ describe("LocalProtocolHandler", () => { await Bun.write(path.join(outsideDir, "secret.txt"), "secret"); await fs.symlink(outsideDir, path.join(localRoot, "linked")); - const router = createRouter({ artifactsDir, sessionId: "session-d" }); + LocalProtocolHandler.setOverride({ + getArtifactsDir: () => artifactsDir, + getSessionId: () => "session-d", + }); + const router = InternalUrlRouter.instance(); await expect(router.resolve("local://linked/secret.txt")).rejects.toThrow("local:// URL escapes local root"); }); }); diff --git a/packages/coding-agent/test/internal-urls/mcp-protocol.test.ts b/packages/coding-agent/test/internal-urls/mcp-protocol.test.ts index ddf9fde1c..4d883b98c 100644 --- a/packages/coding-agent/test/internal-urls/mcp-protocol.test.ts +++ b/packages/coding-agent/test/internal-urls/mcp-protocol.test.ts @@ -1,6 +1,6 @@ -import { describe, expect, it } from "bun:test"; -import { InternalUrlRouter, McpProtocolHandler } from "../../src/internal-urls"; -import type { MCPManager } from "../../src/mcp/manager"; +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; +import { InternalUrlRouter } from "../../src/internal-urls"; +import { MCPManager } from "../../src/mcp/manager"; import type { MCPResource, MCPResourceReadResult, MCPResourceTemplate } from "../../src/mcp/types"; function createMockManager(opts: { @@ -19,25 +19,26 @@ function createMockManager(opts: { } as unknown as MCPManager; } -function createRouter(manager?: MCPManager): InternalUrlRouter { - const router = new InternalUrlRouter(); - router.register( - new McpProtocolHandler({ - getMcpManager: () => manager, - }), - ); - return router; -} - describe("McpProtocolHandler", () => { + beforeEach(() => { + MCPManager.resetForTests(); + InternalUrlRouter.resetForTests(); + }); + + afterEach(() => { + MCPManager.resetForTests(); + InternalUrlRouter.resetForTests(); + }); + it("returns error when no MCP manager is available", async () => { - const router = createRouter(); + const router = InternalUrlRouter.instance(); await expect(router.resolve("mcp://test://resource")).rejects.toThrow("No MCP manager"); }); it("requires resource URI in mcp URL", async () => { const manager = createMockManager({ servers: ["server-a"] }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); await expect(router.resolve("mcp://")).rejects.toThrow("mcp:// URL requires a resource URI"); }); @@ -48,7 +49,8 @@ describe("McpProtocolHandler", () => { templates: [], }); const manager = createMockManager({ servers: ["server-a"], resources }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); await expect(router.resolve("mcp://test://missing")).rejects.toThrow("No MCP server has resource"); await expect(router.resolve("mcp://test://missing")).rejects.toThrow("file://known"); @@ -66,7 +68,8 @@ describe("McpProtocolHandler", () => { resources, readResult: { contents: [{ uri: "test://doc", text: "hello world" }] }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://doc"); expect(resource.content).toBe("hello world"); @@ -84,7 +87,8 @@ describe("McpProtocolHandler", () => { resources, readResult: { contents: [{ uri: "test://doc?q=1", text: "query resource" }] }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://doc?q=1"); expect(resource.content).toBe("query resource"); @@ -101,7 +105,8 @@ describe("McpProtocolHandler", () => { resources, readResult: { contents: [{ uri: "test://docs/foo/raw", text: "from template" }] }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://docs/foo/raw"); expect(resource.content).toBe("from template"); @@ -118,7 +123,8 @@ describe("McpProtocolHandler", () => { resources, readResult: { contents: [{ uri: "test://docs", text: "empty expansion" }] }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://docs"); expect(resource.content).toBe("empty expansion"); @@ -139,7 +145,8 @@ describe("McpProtocolHandler", () => { resources, readResult: { contents: [{ uri: "test://foo/123", text: "from specific" }] }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://foo/123"); expect(resource.notes).toEqual(["MCP server: specific-server"]); @@ -160,7 +167,8 @@ describe("McpProtocolHandler", () => { resources, readResult: { contents: [{ uri: "test://foo", text: "from first" }] }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://foo"); expect(resource.notes).toEqual(["MCP server: first"]); @@ -173,7 +181,8 @@ describe("McpProtocolHandler", () => { templates: [{ uriTemplate: "testing://{id}", name: "testing-template" }], }); const manager = createMockManager({ servers: ["tmpl-server"], resources }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); await expect(router.resolve("mcp://test://foo")).rejects.toThrow("No MCP server has resource"); }); @@ -189,7 +198,8 @@ describe("McpProtocolHandler", () => { resources, readResult: undefined, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); await expect(router.resolve("mcp://test://empty")).rejects.toThrow("returned no content"); await expect(router.resolve("mcp://test://empty")).rejects.toThrow("null-server"); @@ -209,7 +219,8 @@ describe("McpProtocolHandler", () => { contents: [{ uri: "test://image", mimeType: "image/png", blob: blobData }], }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://image"); expect(resource.content).toContain("[Binary content:"); @@ -233,7 +244,8 @@ describe("McpProtocolHandler", () => { ], }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://mixed"); expect(resource.content).toContain("part one"); @@ -254,7 +266,8 @@ describe("McpProtocolHandler", () => { contents: [{ uri: "test://blank" }], }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://blank"); expect(resource.content).toBe("(empty resource)"); @@ -271,7 +284,8 @@ describe("McpProtocolHandler", () => { resources, readError: new Error("connection refused"), }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); await expect(router.resolve("mcp://test://fail")).rejects.toThrow("MCP resource read error:"); await expect(router.resolve("mcp://test://fail")).rejects.toThrow("connection refused"); @@ -292,7 +306,8 @@ describe("McpProtocolHandler", () => { resources, readResult: { contents: [{ uri: "test://shared", text: "from first" }] }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://shared"); expect(resource.notes).toEqual(["MCP server: first"]); @@ -300,7 +315,8 @@ describe("McpProtocolHandler", () => { it("shows (none) when no servers have any resources", async () => { const manager = createMockManager({ servers: ["lonely-server"] }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); await expect(router.resolve("mcp://test://anything")).rejects.toThrow("(none)"); }); @@ -318,7 +334,8 @@ describe("McpProtocolHandler", () => { contents: [{ uri: "test://bin", blob: "data" }], }, }); - const router = createRouter(manager); + MCPManager.setInstance(manager); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("mcp://test://bin"); expect(resource.content).toContain("[Binary content: unknown,"); diff --git a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts index e6cf7ac17..88147dd44 100644 --- a/packages/coding-agent/test/internal-urls/memory-protocol.test.ts +++ b/packages/coding-agent/test/internal-urls/memory-protocol.test.ts @@ -1,36 +1,67 @@ -import { describe, expect, it } from "bun:test"; +import { afterEach, beforeEach, describe, expect, it } from "bun:test"; import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; -import { InternalUrlRouter, MemoryProtocolHandler } from "../../src/internal-urls"; +import { InternalUrlRouter } from "@oh-my-pi/pi-coding-agent/internal-urls"; +import { getMemoryRoot } from "@oh-my-pi/pi-coding-agent/memories"; +import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { getAgentDir, setAgentDir } from "@oh-my-pi/pi-utils"; +import { AgentRegistry } from "../../src/registry/agent-registry"; -async function withTempDir(fn: (dir: string) => Promise): Promise { - const dir = await fs.mkdtemp(path.join(os.tmpdir(), "memory-protocol-")); +interface MemoryFixture { + cwd: string; + memoryRoot: string; + agentDir: string; + cleanupRoot: string; +} + +async function withMemoryFixture(fn: (fixture: MemoryFixture) => Promise): Promise { + const cleanupRoot = await fs.mkdtemp(path.join(os.tmpdir(), "memory-protocol-")); + const previousAgentDir = getAgentDir(); try { - return await fn(dir); + const agentDir = path.join(cleanupRoot, "agent"); + await fs.mkdir(agentDir, { recursive: true }); + const cwd = path.join(cleanupRoot, "project"); + await fs.mkdir(cwd, { recursive: true }); + setAgentDir(agentDir); + const memoryRoot = getMemoryRoot(agentDir, cwd); + await fs.mkdir(memoryRoot, { recursive: true }); + AgentRegistry.global().register({ + id: "test-main", + displayName: "test", + kind: "main", + session: { + sessionManager: { + getCwd: () => cwd, + getArtifactsDir: () => null, + getSessionId: () => "test", + }, + } as unknown as AgentSession, + sessionFile: null, + }); + await fn({ cwd, memoryRoot, agentDir, cleanupRoot }); } finally { - await fs.rm(dir, { recursive: true, force: true }); + setAgentDir(previousAgentDir); + await fs.rm(cleanupRoot, { recursive: true, force: true }); } } -function createRouter(memoryRoot: string): InternalUrlRouter { - const router = new InternalUrlRouter(); - router.register( - new MemoryProtocolHandler({ - getMemoryRoot: () => memoryRoot, - }), - ); - return router; -} - describe("MemoryProtocolHandler", () => { + beforeEach(() => { + AgentRegistry.resetGlobalForTests(); + InternalUrlRouter.resetForTests(); + }); + + afterEach(() => { + AgentRegistry.resetGlobalForTests(); + InternalUrlRouter.resetForTests(); + }); + it("resolves memory://root to memory_summary.md", async () => { - await withTempDir(async tempDir => { - const memoryRoot = path.join(tempDir, "memory"); - await fs.mkdir(memoryRoot, { recursive: true }); + await withMemoryFixture(async ({ memoryRoot }) => { await Bun.write(path.join(memoryRoot, "memory_summary.md"), "summary"); - const router = createRouter(memoryRoot); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("memory://root"); expect(resource.content).toBe("summary"); @@ -39,13 +70,12 @@ describe("MemoryProtocolHandler", () => { }); it("resolves memory://root/ within memory root", async () => { - await withTempDir(async tempDir => { - const memoryRoot = path.join(tempDir, "memory"); + await withMemoryFixture(async ({ memoryRoot }) => { const skillPath = path.join(memoryRoot, "skills", "demo", "SKILL.md"); await fs.mkdir(path.dirname(skillPath), { recursive: true }); await Bun.write(skillPath, "demo skill"); - const router = createRouter(memoryRoot); + const router = InternalUrlRouter.instance(); const resource = await router.resolve("memory://root/skills/demo/SKILL.md"); expect(resource.content).toBe("demo skill"); @@ -54,11 +84,8 @@ describe("MemoryProtocolHandler", () => { }); it("throws for unknown memory namespace", async () => { - await withTempDir(async tempDir => { - const memoryRoot = path.join(tempDir, "memory"); - await fs.mkdir(memoryRoot, { recursive: true }); - - const router = createRouter(memoryRoot); + await withMemoryFixture(async () => { + const router = InternalUrlRouter.instance(); await expect(router.resolve("memory://other/memory_summary.md")).rejects.toThrow( "Unknown memory namespace: other. Supported: root", ); @@ -66,11 +93,8 @@ describe("MemoryProtocolHandler", () => { }); it("blocks path traversal attempts", async () => { - await withTempDir(async tempDir => { - const memoryRoot = path.join(tempDir, "memory"); - await fs.mkdir(memoryRoot, { recursive: true }); - - const router = createRouter(memoryRoot); + await withMemoryFixture(async () => { + const router = InternalUrlRouter.instance(); await expect(router.resolve("memory://root/../secret.md")).rejects.toThrow( "Path traversal (..) is not allowed in memory:// URLs", ); @@ -81,11 +105,8 @@ describe("MemoryProtocolHandler", () => { }); it("throws clear error for missing files", async () => { - await withTempDir(async tempDir => { - const memoryRoot = path.join(tempDir, "memory"); - await fs.mkdir(memoryRoot, { recursive: true }); - - const router = createRouter(memoryRoot); + await withMemoryFixture(async () => { + const router = InternalUrlRouter.instance(); await expect(router.resolve("memory://root/missing.md")).rejects.toThrow( "Memory file not found: memory://root/missing.md", ); @@ -95,15 +116,13 @@ describe("MemoryProtocolHandler", () => { it("blocks symlink escapes outside memory root", async () => { if (process.platform === "win32") return; - await withTempDir(async tempDir => { - const memoryRoot = path.join(tempDir, "memory"); - const outsideDir = path.join(tempDir, "outside"); - await fs.mkdir(memoryRoot, { recursive: true }); + await withMemoryFixture(async ({ memoryRoot, cleanupRoot }) => { + const outsideDir = path.join(cleanupRoot, "outside"); await fs.mkdir(outsideDir, { recursive: true }); await Bun.write(path.join(outsideDir, "secret.md"), "secret"); await fs.symlink(outsideDir, path.join(memoryRoot, "linked")); - const router = createRouter(memoryRoot); + const router = InternalUrlRouter.instance(); await expect(router.resolve("memory://root/linked/secret.md")).rejects.toThrow( "memory:// URL escapes memory root", ); diff --git a/packages/coding-agent/test/issue-1011-repro.test.ts b/packages/coding-agent/test/issue-1011-repro.test.ts new file mode 100644 index 000000000..cffd0977e --- /dev/null +++ b/packages/coding-agent/test/issue-1011-repro.test.ts @@ -0,0 +1,66 @@ +import { describe, expect, it } from "bun:test"; +import * as fs from "node:fs/promises"; +import * as os from "node:os"; +import * as path from "node:path"; + +/** + * Regression for https://github.com/can1357/oh-my-pi/issues/1011 + * + * In v14.5.13 `spawnTabWorker` (in `src/tools/browser/tab-supervisor.ts`) was + * introduced to host browser tabs in a `Worker`. The worker URL was assembled + * as: + * + * ```ts + * const url = new URL("./tab-worker-entry.ts", import.meta.url); + * const worker = new Worker(url.href, { type: "module" }); + * ``` + * + * Bun's `--compile` bundler does NOT statically discover that pattern (the + * worker entry is hidden behind a local variable and `.href`), so the entry + * file is never embedded in the single-file binary. At runtime the worker + * thread tries to load `/$bunfs/root/tab-worker-entry.ts`, the module is + * missing, and the supervisor surfaces the symptom from the issue: + * `Timed out initializing browser tab worker`. + * + * `Bun.build` exposes the same static-analysis pass that drives `--compile`. + * If `tab-worker-entry.ts` is reachable to the bundler, it appears in the + * outputs as a separate `asset` chunk. If the spawn pattern hides it from + * the bundler, only the entry point is emitted. + * + * The bundler is driven through a `bun -e` subprocess that writes its report + * to a tmp file: invoking `Bun.build` directly from inside `bun test` does + * not auto-resolve TypeScript imports the way the real build pipeline does. + */ +describe("issue #1011 — tab worker entry must survive `bun build --compile`", () => { + it("bundles tab-worker-entry.ts as a discoverable asset of tab-supervisor.ts", async () => { + const supervisor = path.resolve(import.meta.dir, "../src/tools/browser/tab-supervisor.ts"); + const packageDir = path.resolve(import.meta.dir, ".."); + const reportPath = path.join(await fs.mkdtemp(path.join(os.tmpdir(), "issue-1011-")), "report.json"); + const script = `const r = await Bun.build({ entrypoints: [${JSON.stringify(supervisor)}], target: "bun" }); await Bun.write(${JSON.stringify(reportPath)}, JSON.stringify({ success: r.success, outputs: r.outputs.map(o => ({ path: o.path, kind: o.kind })), logs: r.logs.map(l => l.message) }));`; + const proc = Bun.spawnSync(["bun", "-e", script], { + cwd: packageDir, + stdout: "pipe", + stderr: "pipe", + }); + const stderr = proc.stderr.toString(); + expect(proc.exitCode, `bun -e exited with ${proc.exitCode}; stderr=${stderr}`).toBe(0); + + const report = (await Bun.file(reportPath).json()) as { + success: boolean; + outputs: { path: string; kind: string }[]; + logs: string[]; + }; + expect(report.success, `bundler logs: ${report.logs.join("; ")}`).toBe(true); + + const workerAssets = report.outputs.filter(out => out.kind === "asset" && out.path.includes("tab-worker-entry")); + if (workerAssets.length === 0) { + const summary = report.outputs.map(o => `${o.kind}:${o.path}`).join(", "); + throw new Error( + `tab-worker-entry.ts was not bundled as an asset of tab-supervisor.ts. ` + + `Bun's --compile bundler cannot embed the worker because the Worker ` + + `constructor argument is not a statically-analyzable URL literal. ` + + `Bundler outputs were: [${summary}]`, + ); + } + }); +}); diff --git a/packages/coding-agent/test/lm-studio-fix.test.ts b/packages/coding-agent/test/lm-studio-fix.test.ts index e16c40432..fdcf5409f 100644 --- a/packages/coding-agent/test/lm-studio-fix.test.ts +++ b/packages/coding-agent/test/lm-studio-fix.test.ts @@ -54,20 +54,4 @@ describe("ModelRegistry LM Studio Fixes", () => { expect(available.some(m => m.provider === "ollama")).toBe(true); expect(available.some(m => m.provider === "lm-studio")).toBe(true); }); - - test("lm-studio discovery handles trailing slashes in baseUrl correctly", async () => { - let _requestedUrl = ""; - using _hook = hookFetch(input => { - const url = String(input); - // Only track URLs from our test endpoints; ignore concurrent built-in provider discovery - if (url.includes("127.0.0.1:1234") || url.includes("127.0.0.1:9999") || url.startsWith("not a url")) { - _requestedUrl = url; - return new Response(JSON.stringify({ data: [{ id: "model-1" }] }), { - status: 200, - headers: { "Content-Type": "application/json" }, - }); - } - return new Response(null, { status: 404 }); - }); - }); }); diff --git a/packages/coding-agent/test/marketplace/cache.test.ts b/packages/coding-agent/test/marketplace/cache.test.ts index b68559e4e..681ca591a 100644 --- a/packages/coding-agent/test/marketplace/cache.test.ts +++ b/packages/coding-agent/test/marketplace/cache.test.ts @@ -62,17 +62,6 @@ describe("isValidVersionForCache", () => { // ── getCachedPluginPath ────────────────────────────────────────────────────── describe("getCachedPluginPath", () => { - it("returns a deterministic path with ___ separators", () => { - const p = getCachedPluginPath("/cache", "my-market", "my-plugin", "1.0.0"); - expect(p).toBe("/cache/my-market___my-plugin___1.0.0"); - }); - - it("is independent of cacheDir content — pure path construction", () => { - const p1 = getCachedPluginPath("/a", "m", "p", "1"); - const p2 = getCachedPluginPath("/b", "m", "p", "1"); - expect(path.basename(p1)).toBe(path.basename(p2)); - }); - it("throws on invalid marketplace name (uppercase)", () => { expect(() => getCachedPluginPath("/cache", "My-Market", "plugin", "1.0.0")).toThrow(/Invalid marketplace name/); }); diff --git a/packages/coding-agent/test/marketplace/manager.test.ts b/packages/coding-agent/test/marketplace/manager.test.ts index 182f17d5b..f974bf602 100644 --- a/packages/coding-agent/test/marketplace/manager.test.ts +++ b/packages/coding-agent/test/marketplace/manager.test.ts @@ -195,13 +195,6 @@ describe("MarketplaceManager", () => { ); }); - it("installPlugin calls clearPluginRootsCache", async () => { - await ctx.manager.addMarketplace(FIXTURE_DIR); - const before = ctx.clearCount(); - await ctx.manager.installPlugin("hello-plugin", "test-marketplace"); - expect(ctx.clearCount()).toBe(before + 1); - }); - // ── Uninstall ────────────────────────────────────────────────────────── it("uninstallPlugin → cache removed + deregistered", async () => { @@ -224,14 +217,6 @@ describe("MarketplaceManager", () => { await expect(ctx.manager.uninstallPlugin("no-at-sign")).rejects.toThrow(/Invalid plugin ID format/); }); - it("uninstallPlugin calls clearPluginRootsCache", async () => { - await ctx.manager.addMarketplace(FIXTURE_DIR); - await ctx.manager.installPlugin("hello-plugin", "test-marketplace"); - const before = ctx.clearCount(); - await ctx.manager.uninstallPlugin("hello-plugin@test-marketplace"); - expect(ctx.clearCount()).toBe(before + 1); - }); - // ── setPluginEnabled ─────────────────────────────────────────────────── it("setPluginEnabled → persisted in registry", async () => { @@ -252,14 +237,6 @@ describe("MarketplaceManager", () => { await expect(ctx.manager.setPluginEnabled("ghost@nowhere", true)).rejects.toThrow(/not installed/); }); - it("setPluginEnabled calls clearPluginRootsCache", async () => { - await ctx.manager.addMarketplace(FIXTURE_DIR); - await ctx.manager.installPlugin("hello-plugin", "test-marketplace"); - const before = ctx.clearCount(); - await ctx.manager.setPluginEnabled("hello-plugin@test-marketplace", false); - expect(ctx.clearCount()).toBe(before + 1); - }); - // ── version fallback ─────────────────────────────────────────────────── it("installPlugin falls back to plugin.json version when catalog version is missing", async () => { diff --git a/packages/coding-agent/test/marketplace/parse-internal-url.test.ts b/packages/coding-agent/test/marketplace/parse-internal-url.test.ts index ca3c456dc..465a4a46a 100644 --- a/packages/coding-agent/test/marketplace/parse-internal-url.test.ts +++ b/packages/coding-agent/test/marketplace/parse-internal-url.test.ts @@ -191,14 +191,6 @@ describe("parseInternalUrl — protocol field", () => { expect(parseInternalUrl("local://x").protocol).toBe("local:"); }); - it("extracts rule: protocol", () => { - expect(parseInternalUrl("rule://x").protocol).toBe("rule:"); - }); - - it("extracts artifact: protocol", () => { - expect(parseInternalUrl("artifact://x").protocol).toBe("artifact:"); - }); - it("extracts protocol from fallback-parsed URL", () => { // This URL fails new URL() due to colon-as-port expect(parseInternalUrl("skill://a:b").protocol).toBe("skill:"); diff --git a/packages/coding-agent/test/marketplace/plugin-dir-roots.test.ts b/packages/coding-agent/test/marketplace/plugin-dir-roots.test.ts deleted file mode 100644 index 39f9727d8..000000000 --- a/packages/coding-agent/test/marketplace/plugin-dir-roots.test.ts +++ /dev/null @@ -1,53 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { buildPluginDirRoot } from "@oh-my-pi/pi-coding-agent/discovery/plugin-dir-roots"; - -describe("buildPluginDirRoot", () => { - it("builds root with manifest name", () => { - const root = buildPluginDirRoot("/path/to/my-plugin", "custom-name"); - expect(root).toEqual({ - id: "custom-name@__local__", - marketplace: "__local__", - plugin: "custom-name", - version: "local", - path: "/path/to/my-plugin", - scope: "user", - }); - }); - - it("falls back to directory basename when no manifest name", () => { - const root = buildPluginDirRoot("/path/to/my-plugin"); - expect(root.plugin).toBe("my-plugin"); - expect(root.id).toBe("my-plugin@__local__"); - }); - - it("falls back to directory basename when manifest name is undefined", () => { - const root = buildPluginDirRoot("/some/dir/cool-plugin", undefined); - expect(root.plugin).toBe("cool-plugin"); - expect(root.id).toBe("cool-plugin@__local__"); - }); - - it("uses __local__ marketplace", () => { - const root = buildPluginDirRoot("/any/path", "test"); - expect(root.marketplace).toBe("__local__"); - }); - - it("uses local version string", () => { - const root = buildPluginDirRoot("/any/path", "test"); - expect(root.version).toBe("local"); - }); - - it("sets scope to user", () => { - const root = buildPluginDirRoot("/any/path", "test"); - expect(root.scope).toBe("user"); - }); - - it("preserves absolute path", () => { - const root = buildPluginDirRoot("/absolute/path/to/plugin", "test"); - expect(root.path).toBe("/absolute/path/to/plugin"); - }); - - it("constructs id as pluginName@__local__", () => { - const root = buildPluginDirRoot("/p", "my-tool"); - expect(root.id).toBe("my-tool@__local__"); - }); -}); diff --git a/packages/coding-agent/test/mcp-roots-list.test.ts b/packages/coding-agent/test/mcp-roots-list.test.ts index dd88e00ce..8a4fe2f9b 100644 --- a/packages/coding-agent/test/mcp-roots-list.test.ts +++ b/packages/coding-agent/test/mcp-roots-list.test.ts @@ -103,14 +103,6 @@ describe("roots response shape", () => { expect(result.roots[0].name).toBe("project"); }); - it("produces valid file:// URI on Windows-style paths", () => { - // path.basename and pathToFileURL are platform-dependent for - // Windows paths; only assert the URI format, not the name. - const result = getRoots("C:\\Users\\dev\\myproject"); - expect(result.roots[0].uri).toMatch(/^file:\/\/\//); - expect(result.roots[0].name).toBeTruthy(); - }); - it("handles paths with spaces", () => { const result = getRoots("/home/user/my project"); expect(result.roots[0].uri).toContain("my%20project"); diff --git a/packages/coding-agent/test/memory-backend-resolve.test.ts b/packages/coding-agent/test/memory-backend-resolve.test.ts index 68cdc8aa9..581263bb3 100644 --- a/packages/coding-agent/test/memory-backend-resolve.test.ts +++ b/packages/coding-agent/test/memory-backend-resolve.test.ts @@ -11,16 +11,6 @@ describe("resolveMemoryBackend", () => { _resetSettingsForTest(); }); - it("returns the off backend when memory.backend is off", () => { - const settings = Settings.isolated({ "memory.backend": "off" }); - expect(resolveMemoryBackend(settings).id).toBe("off"); - }); - - it("returns the local backend when memory.backend is local", () => { - const settings = Settings.isolated({ "memory.backend": "local", "memories.enabled": false }); - expect(resolveMemoryBackend(settings).id).toBe("local"); - }); - it("returns the hindsight backend when memory.backend is hindsight, regardless of legacy memories.enabled", () => { const a = Settings.isolated({ "memory.backend": "hindsight", "memories.enabled": false }); const b = Settings.isolated({ "memory.backend": "hindsight", "memories.enabled": true }); diff --git a/packages/coding-agent/test/model-registry-runtime-provider.test.ts b/packages/coding-agent/test/model-registry-runtime-provider.test.ts index 75f915452..9538a0e34 100644 --- a/packages/coding-agent/test/model-registry-runtime-provider.test.ts +++ b/packages/coding-agent/test/model-registry-runtime-provider.test.ts @@ -95,15 +95,6 @@ describe("ModelRegistry runtime provider registration", () => { expect(registry.find(providerName, modelId)?.headers?.[headerName]).toBe(headerValue); } - test("loads built-in GitLab Duo models and OAuth provider metadata", () => { - const registry = new ModelRegistry(authStorage, modelsJsonPath); - const model = registry.find("gitlab-duo", "claude-sonnet-4-5-20250929"); - - expect(model).toBeDefined(); - expect(model?.api).toBe("anthropic-messages"); - expect(getOAuthProviders().some(provider => provider.id === "gitlab-duo")).toBe(true); - }); - test("validates provider config before mutating custom API state", () => { const registry = new ModelRegistry(authStorage, modelsJsonPath); const beforeAnthropicCount = registry.getAll().filter(model => model.provider === "anthropic").length; @@ -125,27 +116,6 @@ describe("ModelRegistry runtime provider registration", () => { expect(afterAnthropicCount).toBe(beforeAnthropicCount); }); - test("merges provider/model headers and adds Authorization when authHeader is enabled", () => { - const registry = new ModelRegistry(authStorage, modelsJsonPath); - - const config: ProviderConfigInput = { - baseUrl: "https://runtime.example.com/v1", - apiKey: "RUNTIME_KEY", - api: "openai-completions", - authHeader: true, - headers: { "X-Provider": "provider-header" }, - models: [{ ...baseModel, headers: { "X-Model": "model-header" } }], - }; - - registry.registerProvider("runtime-provider", config, "ext://runtime"); - const model = registry.find("runtime-provider", "runtime-model"); - - expect(model).toBeDefined(); - expect(model?.headers?.Authorization).toBe("Bearer RUNTIME_KEY"); - expect(model?.headers?.["X-Provider"]).toBe("provider-header"); - expect(model?.headers?.["X-Model"]).toBe("model-header"); - }); - test("registerProvider applies headers-only overrides to existing provider models across refresh", async () => { const registry = new ModelRegistry(authStorage, modelsJsonPath); const providerName = "anthropic"; diff --git a/packages/coding-agent/test/model-registry.test.ts b/packages/coding-agent/test/model-registry.test.ts index 505083651..18961636c 100644 --- a/packages/coding-agent/test/model-registry.test.ts +++ b/packages/coding-agent/test/model-registry.test.ts @@ -3,7 +3,7 @@ import * as fs from "node:fs"; import * as os from "node:os"; import * as path from "node:path"; import { Effort, type Model, type OpenAICompat, type ThinkingConfig, writeModelCache } from "@oh-my-pi/pi-ai"; -import { kNoAuth, MODEL_ROLES, ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; +import { kNoAuth, ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry"; import { _resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage"; import { hookFetch, Snowflake } from "@oh-my-pi/pi-utils"; @@ -14,11 +14,6 @@ describe("ModelRegistry", () => { let cacheDbPath: string; let authStorage: AuthStorage; - test("commit role includes a visible badge tag", () => { - expect(MODEL_ROLES.commit.tag).toBe("COMMIT"); - expect(MODEL_ROLES.commit.color).toBe("dim"); - }); - beforeEach(async () => { _resetSettingsForTest(); tempDir = path.join(os.tmpdir(), `pi-test-model-registry-${Snowflake.next()}`); diff --git a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts index 89b609779..d75d9e225 100644 --- a/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts +++ b/packages/coding-agent/test/model-selector-role-badge-thinking.test.ts @@ -56,44 +56,6 @@ describe("ModelSelector role badge thinking display", () => { } }); - test("renders per-role thinking labels with inherit mode to avoid badge ambiguity", async () => { - installTestTheme(); - const model = getBundledModel("anthropic", "claude-sonnet-4-5"); - if (!model) throw new Error("Expected bundled model anthropic/claude-sonnet-4-5"); - - const settings = Settings.isolated({ - modelRoles: { - default: `${model.provider}/${model.id}`, - smol: `${model.provider}/${model.id}:minimal`, - slow: `${model.provider}/${model.id}`, - plan: `${model.provider}/${model.id}:high`, - commit: `${model.provider}/${model.id}:medium`, - }, - }); - - const selector = createSelector(model, settings); - - await Bun.sleep(0); - installTestTheme(); - - const rendered = normalizeRenderedText(selector.render(220).join("\n")); - expect(rendered).toContain("DEFAULT (inherit)"); - expect(rendered).toContain("SMOL (min)"); - expect(rendered).toContain("SLOW (inherit)"); - expect(rendered).toContain("PLAN (high)"); - expect(rendered).toContain("COMMIT (medium)"); - expect(rendered).not.toContain("Role Thinking:"); - - selector.handleInput("\n"); - installTestTheme(); - const menuRendered = normalizeRenderedText(selector.render(220).join("\n")); - expect(menuRendered).toContain("Set as DEFAULT (Default)"); - expect(menuRendered).toContain("Set as SMOL (Fast)"); - expect(menuRendered).toContain("Set as SLOW (Thinking)"); - expect(menuRendered).toContain("Set as PLAN (Architect)"); - expect(menuRendered).toContain("Set as COMMIT (Commit)"); - }); - test("shows custom roles from cycleOrder/modelRoles and honors built-in metadata overrides", async () => { installTestTheme(); const model = getBundledModel("anthropic", "claude-sonnet-4-5"); diff --git a/packages/coding-agent/test/modes/components/tree-selector-developer.test.ts b/packages/coding-agent/test/modes/components/tree-selector-developer.test.ts new file mode 100644 index 000000000..fc84cd1ee --- /dev/null +++ b/packages/coding-agent/test/modes/components/tree-selector-developer.test.ts @@ -0,0 +1,65 @@ +import { beforeAll, describe, expect, it } from "bun:test"; +import type { AgentMessage } from "@oh-my-pi/pi-agent-core"; +import { TreeSelectorComponent } from "../../../src/modes/components/tree-selector"; +import * as themeModule from "../../../src/modes/theme/theme"; +import type { SessionEntry, SessionTreeNode } from "../../../src/session/session-manager"; + +let counter = 0; +function makeMessageNode(message: AgentMessage, parentId: string | null = null, label?: string): SessionTreeNode { + const id = `entry-${counter++}`; + const entry: SessionEntry = { + type: "message", + id, + parentId, + timestamp: new Date().toISOString(), + message, + }; + return { entry, children: [], label }; +} + +function render(tree: SessionTreeNode[], width = 120): string { + const selector = new TreeSelectorComponent( + tree, + tree[tree.length - 1]?.entry.id ?? null, + 60, + () => {}, + () => {}, + ); + return Bun.stripANSI(selector.render(width).join("\n")); +} + +describe("TreeSelectorComponent developer message rendering", () => { + beforeAll(async () => { + await themeModule.initTheme(false, undefined, undefined, "dark", "light"); + }); + + it("renders developer messages with their content, not just [developer]", () => { + const planContent = "## Plan\n\n1. Fix the tree selector\n2. Update the HTML export"; + const root = makeMessageNode({ role: "user", content: "/plan", timestamp: 1 }); + const developer = makeMessageNode( + { role: "developer", content: [{ type: "text", text: planContent }], timestamp: 2 }, + root.entry.id, + ); + root.children.push(developer); + + const rendered = render([root]); + + expect(rendered).toContain("developer:"); + expect(rendered).toContain("Fix the tree selector"); + expect(rendered).toContain("Update the HTML export"); + expect(rendered).not.toMatch(/^\s*\[developer\]\s*$/m); + }); + + it("matches developer messages in search (content is searchable)", () => { + const planContent = "ZZZ_UNIQUE_PLAN_TOKEN approved plan body"; + const root = makeMessageNode({ role: "user", content: "/plan", timestamp: 1 }); + const developer = makeMessageNode( + { role: "developer", content: [{ type: "text", text: planContent }], timestamp: 2 }, + root.entry.id, + ); + root.children.push(developer); + + const rendered = render([root]); + expect(rendered).toContain("ZZZ_UNIQUE_PLAN_TOKEN"); + }); +}); diff --git a/packages/coding-agent/test/modes/controllers/btw-controller.test.ts b/packages/coding-agent/test/modes/controllers/btw-controller.test.ts index 01df6ce19..9a9e2e7e8 100644 --- a/packages/coding-agent/test/modes/controllers/btw-controller.test.ts +++ b/packages/coding-agent/test/modes/controllers/btw-controller.test.ts @@ -85,29 +85,6 @@ describe("BtwController", () => { expect(controller.hasActiveRequest()).toBe(true); }); - it("streams text deltas through onTextDelta into the panel", async () => { - const deltas: string[] = []; - const runEphemeralTurn = vi.fn(async (args: RunEphemeralTurnArgs) => { - args.onTextDelta?.("Hel"); - args.onTextDelta?.("lo"); - return { replyText: "Hello", assistantMessage: createAssistantMessage("Hello") }; - }); - const ctx = makeCtx(makeFakeSession(runEphemeralTurn)); - const controller = new BtwController(ctx); - - await controller.start("Hi?"); - await Promise.resolve(); - await Promise.resolve(); - - // Use the captured deltas to verify the callback is wired through. - const callArg = runEphemeralTurn.mock.calls[0]?.[0]; - expect(callArg).toBeDefined(); - callArg?.onTextDelta?.("X"); - deltas.push("X"); - expect(deltas).toEqual(["X"]); - expect(controller.hasActiveRequest()).toBe(true); - }); - it("replaces a previous request by aborting it before issuing the next runEphemeralTurn", async () => { const signals: AbortSignal[] = []; let firstRelease!: () => void; diff --git a/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts b/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts index 1ffaf2e8a..50953cc61 100644 --- a/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts +++ b/packages/coding-agent/test/modes/controllers/event-controller-message-start.test.ts @@ -1,7 +1,15 @@ -import { afterEach, describe, expect, it, vi } from "bun:test"; +import { afterEach, beforeAll, describe, expect, it, vi } from "bun:test"; import type { TextContent, UserMessage } from "@oh-my-pi/pi-ai"; import { EventController } from "@oh-my-pi/pi-coding-agent/modes/controllers/event-controller"; +import { initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import type { InteractiveModeContext } from "@oh-my-pi/pi-coding-agent/modes/types"; +import { UiHelpers } from "@oh-my-pi/pi-coding-agent/modes/utils/ui-helpers"; +import type { CustomMessage } from "@oh-my-pi/pi-coding-agent/session/messages"; +import { Container } from "@oh-my-pi/pi-tui"; + +beforeAll(() => { + initTheme(); +}); function createUserMessage(text: string): UserMessage { return { @@ -111,3 +119,88 @@ describe("EventController message_start (user role)", () => { expect(ctx.optimisticUserMessageSignature).toBeUndefined(); }); }); + +function createIrcMessage(timestamp: number): CustomMessage<{ from: string; message: string }> { + return { + role: "custom", + customType: "irc:incoming", + content: "Ready", + display: true, + details: { from: "0-Main", message: "Ready" }, + timestamp, + }; +} + +function createIrcContext() { + const chatContainer = new Container(); + const requestRender = vi.fn(); + const ctx = { + isInitialized: true, + statusLine: { invalidate: vi.fn() }, + updateEditorTopBorder: vi.fn(), + ui: { requestRender }, + chatContainer, + session: {}, + } as unknown as InteractiveModeContext; + const helpers = new UiHelpers(ctx); + const addMessageToChat: InteractiveModeContext["addMessageToChat"] = vi.fn((message, options) => + helpers.addMessageToChat(message, options), + ); + ctx.addMessageToChat = addMessageToChat; + return { ctx, chatContainer, requestRender, addMessageToChat }; +} + +describe("EventController IRC expiry", () => { + afterEach(() => { + vi.useRealTimers(); + vi.restoreAllMocks(); + }); + + it("renders IRC messages immediately and removes their components after the TTL", async () => { + vi.useFakeTimers(); + const message = createIrcMessage(1); + const { ctx, chatContainer, requestRender } = createIrcContext(); + const controller = new EventController(ctx); + + await controller.handleEvent({ type: "irc_message", message }); + + expect(chatContainer.children).toHaveLength(2); + expect(requestRender).toHaveBeenCalledTimes(1); + + vi.advanceTimersByTime(9_999); + expect(chatContainer.children).toHaveLength(2); + + vi.advanceTimersByTime(1); + expect(chatContainer.children).toHaveLength(0); + expect(requestRender).toHaveBeenCalledTimes(2); + }); + + it("does not schedule duplicate expiry for duplicate IRC events", async () => { + vi.useFakeTimers(); + const message = createIrcMessage(2); + const { ctx, chatContainer, addMessageToChat } = createIrcContext(); + const controller = new EventController(ctx); + + await controller.handleEvent({ type: "irc_message", message }); + await controller.handleEvent({ type: "irc_message", message }); + + expect(addMessageToChat).toHaveBeenCalledTimes(1); + expect(chatContainer.children).toHaveLength(2); + vi.advanceTimersByTime(10_000); + expect(chatContainer.children).toHaveLength(0); + }); + + it("clears pending IRC expiry timers on dispose", async () => { + vi.useFakeTimers(); + const message = createIrcMessage(3); + const { ctx, chatContainer, requestRender } = createIrcContext(); + const controller = new EventController(ctx); + + await controller.handleEvent({ type: "irc_message", message }); + controller.dispose(); + vi.advanceTimersByTime(10_000); + + expect(chatContainer.children).toHaveLength(2); + expect(requestRender).toHaveBeenCalledTimes(1); + }); +}); diff --git a/packages/coding-agent/test/output-block.test.ts b/packages/coding-agent/test/output-block.test.ts deleted file mode 100644 index bec2419ee..000000000 --- a/packages/coding-agent/test/output-block.test.ts +++ /dev/null @@ -1,38 +0,0 @@ -import { afterEach, describe, expect, it } from "bun:test"; -import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { renderOutputBlock } from "@oh-my-pi/pi-coding-agent/tui/output-block"; -import { ImageProtocol, TERMINAL } from "@oh-my-pi/pi-tui"; - -type MutableTerminalInfo = { - imageProtocol: ImageProtocol | null; -}; - -const terminal = TERMINAL as unknown as MutableTerminalInfo; - -describe("renderOutputBlock", () => { - const originalProtocol = TERMINAL.imageProtocol; - - afterEach(() => { - terminal.imageProtocol = originalProtocol; - }); - - it("passes SIXEL lines through without trimming or padding", async () => { - terminal.imageProtocol = ImageProtocol.Sixel; - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - const sixel = "\x1bPqabc\x1b\\"; - const lines = renderOutputBlock( - { - width: 40, - sections: [{ label: "Output", lines: ["regular line", sixel] }], - }, - uiTheme, - ); - - expect(lines.filter(line => line === sixel)).toHaveLength(1); - const regularLine = lines.find(line => line.includes("regular line")); - expect(regularLine).toBeDefined(); - expect(regularLine).not.toBe("regular line"); - }); -}); diff --git a/packages/coding-agent/test/plan-mode/plan-mode-approved-prompt.test.ts b/packages/coding-agent/test/plan-mode/plan-mode-approved-prompt.test.ts deleted file mode 100644 index fe654ab8f..000000000 --- a/packages/coding-agent/test/plan-mode/plan-mode-approved-prompt.test.ts +++ /dev/null @@ -1,14 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { prompt } from "@oh-my-pi/pi-utils"; -import planModeApprovedPrompt from "../../src/prompts/system/plan-mode-approved.md" with { type: "text" }; - -describe("plan-mode-approved prompt", () => { - it("includes final plan artifact path in injected execution prompt", () => { - const rendered = prompt.render(planModeApprovedPrompt, { - planContent: "1. Do work", - finalPlanFilePath: "local://WP_MIGRATION_PLAN.md", - }); - - expect(rendered).toContain("local://WP_MIGRATION_PLAN.md"); - }); -}); diff --git a/packages/coding-agent/test/plugin-command.test.ts b/packages/coding-agent/test/plugin-command.test.ts index 709ac52a6..42bd1df3b 100644 --- a/packages/coding-agent/test/plugin-command.test.ts +++ b/packages/coding-agent/test/plugin-command.test.ts @@ -9,12 +9,6 @@ const TEST_CONFIG: CliConfig = { }; describe("Plugin command scope parsing", () => { - it("accepts project scope", async () => { - const command = new Plugin(["install", "--scope", "project"], TEST_CONFIG); - const { flags } = await command.parse(Plugin); - expect(flags.scope).toBe("project"); - }); - it("rejects invalid scope values", async () => { const command = new Plugin(["install", "--scope", "porject"], TEST_CONFIG); await expect(command.parse(Plugin)).rejects.toThrow(/Expected --scope to be one of: user, project/); diff --git a/packages/coding-agent/test/prompt-templates.test.ts b/packages/coding-agent/test/prompt-templates.test.ts index 099ba190b..af317fbbd 100644 --- a/packages/coding-agent/test/prompt-templates.test.ts +++ b/packages/coding-agent/test/prompt-templates.test.ts @@ -18,14 +18,6 @@ import { parseCommandArgs, substituteArgs } from "@oh-my-pi/pi-coding-agent/util // ============================================================================ describe("substituteArgs", () => { - test("should replace $ARGUMENTS with all args joined", () => { - expect(substituteArgs("Test: $ARGUMENTS", ["a", "b", "c"])).toBe("Test: a b c"); - }); - - test("should replace $@ with all args joined", () => { - expect(substituteArgs("Test: $@", ["a", "b", "c"])).toBe("Test: a b c"); - }); - test("should support $@ slicing with start offset", () => { expect(substituteArgs("Test: $@[2]", ["a", "b", "c"])).toBe("Test: b c"); }); @@ -66,18 +58,6 @@ describe("substituteArgs", () => { expect(substituteArgs("$1: $@", ["prefix", "a", "b"])).toBe("prefix: prefix a b"); }); - test("should handle empty arguments array with $ARGUMENTS", () => { - expect(substituteArgs("Test: $ARGUMENTS", [])).toBe("Test: "); - }); - - test("should handle empty arguments array with $@", () => { - expect(substituteArgs("Test: $@", [])).toBe("Test: "); - }); - - test("should handle empty arguments array with $1", () => { - expect(substituteArgs("Test: $1", [])).toBe("Test: "); - }); - test("should handle multiple occurrences of $ARGUMENTS", () => { expect(substituteArgs("$ARGUMENTS and $ARGUMENTS", ["a", "b"])).toBe("a b and a b"); }); @@ -116,14 +96,6 @@ describe("substituteArgs", () => { expect(substituteArgs("$ARGUMENTS", ["first arg", "second arg"])).toBe("first arg second arg"); }); - test("should handle single argument with $ARGUMENTS", () => { - expect(substituteArgs("Test: $ARGUMENTS", ["only"])).toBe("Test: only"); - }); - - test("should handle single argument with $@", () => { - expect(substituteArgs("Test: $@", ["only"])).toBe("Test: only"); - }); - test("should handle $0 (zero index)", () => { expect(substituteArgs("$0", ["a", "b"])).toBe(""); }); @@ -140,49 +112,16 @@ describe("substituteArgs", () => { expect(substituteArgs("pre$@", ["a", "b"])).toBe("prea b"); }); - test("should handle empty arguments in middle of list", () => { - expect(substituteArgs("$ARGUMENTS", ["a", "", "c"])).toBe("a c"); - }); - test("should handle trailing and leading spaces in arguments", () => { expect(substituteArgs("$ARGUMENTS", [" leading ", "trailing "])).toBe(" leading trailing "); }); - test("should handle argument containing pattern partially", () => { - expect(substituteArgs("Prefix $ARGUMENTS suffix", ["ARGUMENTS"])).toBe("Prefix ARGUMENTS suffix"); - }); - - test("should handle non-matching patterns", () => { - expect(substituteArgs("$A $$ $ $ARGS", ["a"])).toBe("$A $$ $ $ARGS"); - }); - - test("should handle case variations (case-sensitive)", () => { - expect(substituteArgs("$arguments $Arguments $ARGUMENTS", ["a", "b"])).toBe("$arguments $Arguments a b"); - }); - - test("should handle both syntaxes in same command with same result", () => { - const args = ["x", "y", "z"]; - const result1 = substituteArgs("$@ and $ARGUMENTS", args); - const result2 = substituteArgs("$ARGUMENTS and $@", args); - expect(result1).toBe(result2); - expect(result1).toBe("x y z and x y z"); - }); - test("should handle very long argument lists", () => { const args = Array.from({ length: 100 }, (_, i) => `arg${i}`); const result = substituteArgs("$ARGUMENTS", args); expect(result).toBe(args.join(" ")); }); - test("should handle numbered placeholders with single digit", () => { - expect(substituteArgs("$1 $2 $3", ["a", "b", "c"])).toBe("a b c"); - }); - - test("should handle numbered placeholders with multiple digits", () => { - const args = Array.from({ length: 15 }, (_, i) => `val${i}`); - expect(substituteArgs("$10 $12 $15", args)).toBe("val9 val11 val14"); - }); - test("should handle escaped dollar signs (literal backslash preserved)", () => { // Note: No escape mechanism exists - backslash is treated literally expect(substituteArgs("Price: \\$100", [])).toBe("Price: \\"); @@ -257,14 +196,6 @@ describe("parseCommandArgs", () => { // Note: This implementation doesn't handle escaped quotes - backslash is literal expect(parseCommandArgs('"quoted \\"text\\""')).toEqual(["quoted \\text\\"]); }); - - test("should handle trailing spaces", () => { - expect(parseCommandArgs("a b c ")).toEqual(["a", "b", "c"]); - }); - - test("should handle leading spaces", () => { - expect(parseCommandArgs(" a b c")).toEqual(["a", "b", "c"]); - }); }); // ============================================================================ diff --git a/packages/coding-agent/test/prompts/review-request.test.ts b/packages/coding-agent/test/prompts/review-request.test.ts deleted file mode 100644 index a3d636a70..000000000 --- a/packages/coding-agent/test/prompts/review-request.test.ts +++ /dev/null @@ -1,21 +0,0 @@ -import { describe, expect, it } from "bun:test"; - -describe("review request prompt", () => { - it("renders the additional instructions block from the template", async () => { - const template = await Bun.file(new URL("../../src/prompts/review-request.md", import.meta.url)).text(); - - expect(template).toContain("{{#if additionalInstructions}}"); - expect(template).toContain("### Additional Instructions"); - expect(template).toContain("{{additionalInstructions}}"); - expect(template).toContain("{{/if}}"); - }); - - it("keeps the additional instructions suffix out of TypeScript", async () => { - const source = await Bun.file( - new URL("../../src/extensibility/custom-commands/bundled/review/index.ts", import.meta.url), - ).text(); - - expect(source).not.toContain("### Additional Instructions"); - expect(source).not.toContain("appendInstructions("); - }); -}); diff --git a/packages/coding-agent/test/rpc-mode-extension-ui.test.ts b/packages/coding-agent/test/rpc-mode-extension-ui.test.ts deleted file mode 100644 index 23426eaf9..000000000 --- a/packages/coding-agent/test/rpc-mode-extension-ui.test.ts +++ /dev/null @@ -1,82 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { type PendingExtensionRequest, requestRpcEditor } from "@oh-my-pi/pi-coding-agent/modes/rpc/rpc-mode"; -import type { RpcExtensionUIRequest } from "@oh-my-pi/pi-coding-agent/modes/rpc/rpc-types"; - -function isExtensionUiRequest(obj: RpcExtensionUIRequest | object): obj is RpcExtensionUIRequest { - return "type" in obj && obj.type === "extension_ui_request"; -} - -describe("requestRpcEditor", () => { - it("serializes promptStyle on editor requests", async () => { - const pendingRequests = new Map(); - const requests: RpcExtensionUIRequest[] = []; - - const promise = requestRpcEditor( - pendingRequests, - obj => { - if (isExtensionUiRequest(obj)) { - requests.push(obj); - } - }, - "Enter your response:", - "draft", - undefined, - { promptStyle: true }, - ); - - expect(requests).toHaveLength(1); - const request = requests[0]; - if (!request || request.method !== "editor") { - throw new Error("Expected an editor request"); - } - expect(request.promptStyle).toBe(true); - expect(request.prefill).toBe("draft"); - - const pending = pendingRequests.get(request.id); - if (!pending) { - throw new Error("Expected a pending request"); - } - pending.resolve({ type: "extension_ui_response", id: request.id, value: "custom response" }); - - await expect(promise).resolves.toBe("custom response"); - expect(pendingRequests.size).toBe(0); - }); - - it("resolves editor requests on abort and clears pending state", async () => { - const pendingRequests = new Map(); - const requests: RpcExtensionUIRequest[] = []; - const controller = new AbortController(); - - const promise = requestRpcEditor( - pendingRequests, - obj => { - if (isExtensionUiRequest(obj)) { - requests.push(obj); - } - }, - "Enter your response:", - undefined, - { signal: controller.signal }, - { promptStyle: true }, - ); - - expect(requests).toHaveLength(1); - const request = requests[0]; - if (!request || request.method !== "editor") { - throw new Error("Expected an editor request"); - } - expect(request.promptStyle).toBe(true); - expect(pendingRequests.has(request.id)).toBe(true); - - controller.abort(); - - expect(requests).toHaveLength(2); - const cancelRequest = requests[1]; - if (!cancelRequest || cancelRequest.method !== "cancel") { - throw new Error("Expected a cancel request"); - } - expect(cancelRequest.targetId).toBe(request.id); - await expect(promise).resolves.toBeUndefined(); - expect(pendingRequests.has(request.id)).toBe(false); - }); -}); diff --git a/packages/coding-agent/test/secrets-obfuscator.test.ts b/packages/coding-agent/test/secrets-obfuscator.test.ts index 925e78e34..e4ad9b902 100644 --- a/packages/coding-agent/test/secrets-obfuscator.test.ts +++ b/packages/coding-agent/test/secrets-obfuscator.test.ts @@ -7,12 +7,6 @@ import { SecretObfuscator } from "../src/secrets/obfuscator"; import { compileSecretRegex } from "../src/secrets/regex"; describe("compileSecretRegex", () => { - it("compiles pattern with explicit flags and enforces global scanning", () => { - const regex = compileSecretRegex("api[_-]?key\\s*=\\s*\\w+", "gi"); - expect(regex.source).toBe("api[_-]?key\\s*=\\s*\\w+"); - expect(regex.flags).toBe("gi"); - }); - it("adds global flag when not provided", () => { const regex = compileSecretRegex("api[_-]?key\\s*=\\s*\\w+", "i"); expect(regex.source).toBe("api[_-]?key\\s*=\\s*\\w+"); diff --git a/packages/coding-agent/test/session-color.test.ts b/packages/coding-agent/test/session-color.test.ts deleted file mode 100644 index f8184bb3a..000000000 --- a/packages/coding-agent/test/session-color.test.ts +++ /dev/null @@ -1,19 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { getSessionAccentHex } from "../src/utils/session-color"; -import { formatSessionTerminalTitle } from "../src/utils/title-generator"; - -describe("getSessionAccentHex", () => { - it("returns a stable hex for the same name", () => { - expect(getSessionAccentHex("Named session")).toBe(getSessionAccentHex("Named session")); - }); -}); - -describe("formatSessionTerminalTitle", () => { - it("uses the session name when present", () => { - expect(formatSessionTerminalTitle("Manual title", "/work/pi")).toBe("π: Manual title"); - }); - - it("falls back to the cwd basename when the session name is missing", () => { - expect(formatSessionTerminalTitle(undefined, "/work/pi")).toBe("π: pi"); - }); -}); diff --git a/packages/coding-agent/test/session-manager/file-operations.test.ts b/packages/coding-agent/test/session-manager/file-operations.test.ts index f5e64fcbc..96d30aa53 100644 --- a/packages/coding-agent/test/session-manager/file-operations.test.ts +++ b/packages/coding-agent/test/session-manager/file-operations.test.ts @@ -24,29 +24,6 @@ describe("loadEntriesFromFile", () => { fs.rmSync(tempDir, { recursive: true, force: true }); }); - it("returns empty array for non-existent file", async () => { - const entries = await loadEntriesFromFile(path.join(tempDir, "nonexistent.jsonl")); - expect(entries).toEqual([]); - }); - - it("returns empty array for empty file", async () => { - const file = path.join(tempDir, "empty.jsonl"); - fs.writeFileSync(file, ""); - expect(await loadEntriesFromFile(file)).toEqual([]); - }); - - it("returns empty array for file without valid session header", async () => { - const file = path.join(tempDir, "no-header.jsonl"); - fs.writeFileSync(file, '{"type":"message","id":"1"}\n'); - expect(await loadEntriesFromFile(file)).toEqual([]); - }); - - it("returns empty array for malformed JSON", async () => { - const file = path.join(tempDir, "malformed.jsonl"); - fs.writeFileSync(file, "not json\n"); - expect(await loadEntriesFromFile(file)).toEqual([]); - }); - it("loads valid session file", async () => { const file = path.join(tempDir, "valid.jsonl"); fs.writeFileSync( @@ -85,25 +62,6 @@ describe("findMostRecentSession", () => { fs.rmSync(tempDir, { recursive: true, force: true }); }); - it("returns null for empty directory", async () => { - expect(await findMostRecentSession(tempDir)).toBeNull(); - }); - - it("returns null for non-existent directory", async () => { - expect(await findMostRecentSession(path.join(tempDir, "nonexistent"))).toBeNull(); - }); - - it("ignores non-jsonl files", async () => { - fs.writeFileSync(path.join(tempDir, "file.txt"), "hello"); - fs.writeFileSync(path.join(tempDir, "file.json"), "{}"); - expect(await findMostRecentSession(tempDir)).toBeNull(); - }); - - it("ignores jsonl files without valid session header", async () => { - fs.writeFileSync(path.join(tempDir, "invalid.jsonl"), '{"type":"message"}\n'); - expect(await findMostRecentSession(tempDir)).toBeNull(); - }); - it("returns single valid session file", async () => { const file = path.join(tempDir, "session.jsonl"); fs.writeFileSync(file, '{"type":"session","id":"abc","timestamp":"2025-01-01T00:00:00Z","cwd":"/tmp"}\n'); diff --git a/packages/coding-agent/test/session-provider-section.test.ts b/packages/coding-agent/test/session-provider-section.test.ts deleted file mode 100644 index 59d96a8ae..000000000 --- a/packages/coding-agent/test/session-provider-section.test.ts +++ /dev/null @@ -1,35 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { getProviderDetails, type Model } from "@oh-my-pi/pi-ai"; -import { renderProviderSection } from "@oh-my-pi/pi-coding-agent/modes/controllers/command-controller"; - -describe("session provider section", () => { - it("renders codex provider details with transport fields", () => { - const model: Model<"openai-codex-responses"> = { - id: "gpt-5.3-codex-spark", - name: "GPT-5.3 Codex Spark", - api: "openai-codex-responses", - provider: "openai-codex", - baseUrl: "https://chatgpt.com/backend-api", - reasoning: true, - preferWebsockets: true, - input: ["text"], - cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }, - contextWindow: 128000, - maxTokens: 128000, - }; - - const details = getProviderDetails({ - model, - sessionId: "session-1", - authMode: "oauth", - }); - const output = renderProviderSection(details, { fg: (_color: string, text: string) => text }); - - expect(output).toContain("Name:"); - expect(output).toContain("openai-codex"); - expect(output).toContain("Transport:"); - expect(output).toContain("WebSocket:"); - expect(output).toContain("Reuse:"); - expect(output).toContain("Auth:"); - }); -}); diff --git a/packages/coding-agent/test/slash-commands/force.test.ts b/packages/coding-agent/test/slash-commands/force.test.ts index b59fc7fdc..88f27751c 100644 --- a/packages/coding-agent/test/slash-commands/force.test.ts +++ b/packages/coding-agent/test/slash-commands/force.test.ts @@ -95,17 +95,4 @@ describe("/force slash command", () => { expect(harness.showStatus).not.toHaveBeenCalled(); expect(harness.setText).toHaveBeenCalledWith(""); }); - - it("does not pass through prompt when tool validation fails", async () => { - const harness = createRuntimeHarness({ - setForcedToolChoice: () => { - throw new Error('Tool "write" is not currently active.'); - }, - }); - - const result = await executeBuiltinSlashCommand("/force:write fix stuff", harness.runtime); - - expect(result).toBe(true); - expect(harness.showError).toHaveBeenCalledWith('Tool "write" is not currently active.'); - }); }); diff --git a/packages/coding-agent/test/ssh/connection-manager.test.ts b/packages/coding-agent/test/ssh/connection-manager.test.ts index 655e602c2..8ec611252 100644 --- a/packages/coding-agent/test/ssh/connection-manager.test.ts +++ b/packages/coding-agent/test/ssh/connection-manager.test.ts @@ -1,14 +1,15 @@ import { describe, expect, it } from "bun:test"; -import { buildRemoteCommand } from "../../src/ssh/connection-manager"; +import { buildRemoteCommand, supportsSshControlMaster } from "../../src/ssh/connection-manager"; describe("buildRemoteCommand", () => { - it("includes -n to bind stdin to /dev/null for mux channel opens", async () => { + it("includes -n and OpenSSH ControlMaster options on Unix-like platforms", async () => { const args = await buildRemoteCommand( { name: "host", host: "192.168.3.146", }, "ls -la", + { platform: "linux" }, ); expect(args[0]).toBe("-n"); @@ -16,4 +17,34 @@ describe("buildRemoteCommand", () => { expect(args.at(-2)).toBe("192.168.3.146"); expect(args.at(-1)).toBe("ls -la"); }); + + it("omits OpenSSH ControlMaster options on Windows", async () => { + const args = await buildRemoteCommand( + { + name: "host", + host: "192.168.3.146", + }, + "ls -la", + { platform: "win32" }, + ); + + expect(args[0]).toBe("-n"); + expect(args).not.toContain("ControlMaster=auto"); + expect(args.some(arg => arg.startsWith("ControlPath="))).toBe(false); + expect(args).not.toContain("ControlPersist=3600"); + expect(args).toContain("BatchMode=yes"); + expect(args.at(-2)).toBe("192.168.3.146"); + expect(args.at(-1)).toBe("ls -la"); + }); +}); + +describe("supportsSshControlMaster", () => { + it("disables OpenSSH connection multiplexing on native Windows", () => { + expect(supportsSshControlMaster("win32")).toBe(false); + }); + + it("keeps OpenSSH connection multiplexing on Unix-like platforms", () => { + expect(supportsSshControlMaster("linux")).toBe(true); + expect(supportsSshControlMaster("darwin")).toBe(true); + }); }); diff --git a/packages/coding-agent/test/status-line-git-utils.test.ts b/packages/coding-agent/test/status-line-git-utils.test.ts index 4a2a7e90c..3223acfe1 100644 --- a/packages/coding-agent/test/status-line-git-utils.test.ts +++ b/packages/coding-agent/test/status-line-git-utils.test.ts @@ -32,14 +32,6 @@ describe("parseGitHubRepo", () => { expect(parseGitHubRepo("https://gitlab.com/user/repo.git")).toBeNull(); }); - test("returns null for empty string", () => { - expect(parseGitHubRepo("")).toBeNull(); - }); - - test("returns null for malformed URL", () => { - expect(parseGitHubRepo("not-a-url")).toBeNull(); - }); - test("handles GitHub Enterprise-style URLs (no match)", () => { expect(parseGitHubRepo("https://github.corp.com/org/repo.git")).toBeNull(); }); diff --git a/packages/coding-agent/test/streaming-output.test.ts b/packages/coding-agent/test/streaming-output.test.ts index 9454d3fec..675deb2ac 100644 --- a/packages/coding-agent/test/streaming-output.test.ts +++ b/packages/coding-agent/test/streaming-output.test.ts @@ -3,9 +3,6 @@ import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; import { - DEFAULT_MAX_BYTES, - DEFAULT_MAX_COLUMN, - DEFAULT_MAX_LINES, formatHeadTruncationNotice, formatTailTruncationNotice, OutputSink, @@ -41,14 +38,6 @@ afterEach(async () => { else Bun.env.PI_ALLOW_SIXEL_PASSTHROUGH = originalAllowPassthrough; }); -describe("streaming-output exports", () => { - test("exports expected default limits", () => { - expect(DEFAULT_MAX_LINES).toBe(3000); - expect(DEFAULT_MAX_BYTES).toBe(50 * 1024); - expect(DEFAULT_MAX_COLUMN).toBe(1024); - }); -}); - describe("truncateTailBytes", () => { test("returns source when already under limit", () => { const text = "hello"; diff --git a/packages/coding-agent/test/system-prompt-templates.test.ts b/packages/coding-agent/test/system-prompt-templates.test.ts index f4a4ac099..046d0990c 100644 --- a/packages/coding-agent/test/system-prompt-templates.test.ts +++ b/packages/coding-agent/test/system-prompt-templates.test.ts @@ -226,7 +226,7 @@ describe("system Handlebars prompt templates", () => { expect(rendered).toContain("call `search_tool_bm25` before concluding no such tool exists"); }); - test("buildSystemPrompt keeps system project and now as separate ordered blocks", async () => { + test("buildSystemPrompt keeps system and project as separate ordered blocks with date context in project", async () => { await withTempDir(async dir => { const { systemPrompt } = await buildSystemPrompt({ cwd: dir, @@ -243,14 +243,14 @@ describe("system Handlebars prompt templates", () => { }, }); - expect(systemPrompt).toHaveLength(3); + expect(systemPrompt).toHaveLength(2); expect(systemPrompt[0]).toContain("[CONTRACT]"); expect(systemPrompt[0]).not.toContain("current working directory"); expect(systemPrompt[1]).toContain(""); expect(systemPrompt[1]).toContain(""); - expect(systemPrompt[1]).not.toContain("current working directory"); - expect(systemPrompt[2]).toContain("Today is "); - expect(systemPrompt[2]).toContain(`current working directory is '${dir}'.`); + expect(systemPrompt[1]).toContain("Today is "); + expect(systemPrompt[1]).toContain(`current working directory is '${dir}'.`); + expect(systemPrompt[1].indexOf("")).toBeLessThan(systemPrompt[1].indexOf("Today is ")); }); }); test("buildSystemPrompt renders workspace tree after directory context in project prompt", async () => { diff --git a/packages/coding-agent/test/task/render-report-finding.test.ts b/packages/coding-agent/test/task/render-report-finding.test.ts deleted file mode 100644 index 1145bd946..000000000 --- a/packages/coding-agent/test/task/render-report-finding.test.ts +++ /dev/null @@ -1,87 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; -import { taskToolRenderer } from "../../src/task/render"; -import type { TaskToolDetails } from "../../src/task/types"; - -describe("taskToolRenderer report_finding safety", () => { - it("renders progress without crashing when report_finding payload is malformed", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - - const details: TaskToolDetails = { - projectAgentsDir: null, - results: [], - totalDurationMs: 42, - progress: [ - { - index: 0, - id: "1-Reviewer", - agent: "reviewer", - agentSource: "bundled", - status: "running", - task: "Review patch", - recentTools: [], - recentOutput: [], - toolCount: 1, - tokens: 0, - durationMs: 42, - extractedToolData: { - report_finding: [{}], - }, - }, - ], - }; - - const rendered = taskToolRenderer.renderResult( - { - content: [{ type: "text", text: "" }], - details, - }, - { expanded: false, isPartial: true }, - uiTheme, - ); - - expect(() => rendered.render(120)).not.toThrow(); - }); - - it("renders abort reason inline for aborted subagent results", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - - const details: TaskToolDetails = { - projectAgentsDir: null, - results: [ - { - index: 0, - id: "1-Reviewer", - agent: "reviewer", - agentSource: "bundled", - task: "Review patch", - exitCode: 1, - output: "", - stderr: "", - truncated: false, - durationMs: 42, - tokens: 0, - aborted: true, - abortReason: "blocked by permissions", - }, - ], - totalDurationMs: 42, - }; - - const rendered = taskToolRenderer.renderResult( - { - content: [{ type: "text", text: "" }], - details, - }, - { expanded: false, isPartial: false }, - uiTheme, - ); - - const lines = rendered.render(120); - expect(lines.join("\n")).toContain("blocked by permissions"); - }); -}); diff --git a/packages/coding-agent/test/tool-choice-queue.test.ts b/packages/coding-agent/test/tool-choice-queue.test.ts index f437df6d2..52a744703 100644 --- a/packages/coding-agent/test/tool-choice-queue.test.ts +++ b/packages/coding-agent/test/tool-choice-queue.test.ts @@ -6,49 +6,6 @@ const forced = { type: "tool", name: "write" } as const; const forcedRead = { type: "tool", name: "read" } as const; describe("ToolChoiceQueue", () => { - it("returns undefined when empty", () => { - const q = new ToolChoiceQueue(); - expect(q.nextToolChoice()).toBeUndefined(); - }); - - it("pushOnce yields once then exhausts", () => { - const q = new ToolChoiceQueue(); - q.pushOnce(forced, { label: "a" }); - expect(q.nextToolChoice()).toEqual(forced); - q.resolve(); - expect(q.nextToolChoice()).toBeUndefined(); - }); - - it("pushSequence yields in order then exhausts", () => { - const q = new ToolChoiceQueue(); - q.pushSequence([forced, "none"], { label: "seq" }); - expect(q.nextToolChoice()).toEqual(forced); - q.resolve(); - expect(q.nextToolChoice()).toBe("none"); - q.resolve(); - expect(q.nextToolChoice()).toBeUndefined(); - }); - - it("now:true prepends to head", () => { - const q = new ToolChoiceQueue(); - q.pushOnce(forced, { label: "first" }); - q.pushOnce(forcedRead, { label: "urgent", now: true }); - expect(q.nextToolChoice()).toEqual(forcedRead); - q.resolve(); - expect(q.nextToolChoice()).toEqual(forced); - }); - - it("multiple directives drain in FIFO order", () => { - const q = new ToolChoiceQueue(); - q.pushOnce(forced, { label: "a" }); - q.pushOnce(forcedRead, { label: "b" }); - expect(q.nextToolChoice()).toEqual(forced); - q.resolve(); - expect(q.nextToolChoice()).toEqual(forcedRead); - q.resolve(); - expect(q.nextToolChoice()).toBeUndefined(); - }); - describe("resolve callback", () => { it("fires onResolved with the served choice", () => { const q = new ToolChoiceQueue(); @@ -61,11 +18,6 @@ describe("ToolChoiceQueue", () => { q.resolve(); expect(resolved).toEqual([{ choice: forced }]); }); - - it("does not fire onResolved when queue is empty", () => { - const q = new ToolChoiceQueue(); - q.resolve(); // no-op, nothing in-flight - }); }); describe("reject callback", () => { @@ -102,39 +54,6 @@ describe("ToolChoiceQueue", () => { expect(q.nextToolChoice()).toBeUndefined(); }); - it("default (no callback) drops the yield", () => { - const q = new ToolChoiceQueue(); - q.pushOnce(forced, { label: "a" }); - expect(q.nextToolChoice()).toEqual(forced); - q.reject("aborted"); - expect(q.nextToolChoice()).toBeUndefined(); - }); - - it("reject is a no-op when nothing is in-flight", () => { - const q = new ToolChoiceQueue(); - q.pushOnce(forced, { - label: "a", - onRejected: () => "requeue", - }); - q.reject("aborted"); // no-op, nothing yielded yet - expect(q.nextToolChoice()).toEqual(forced); - }); - - it("passes the correct reason to onRejected", () => { - const q = new ToolChoiceQueue(); - const reasons: string[] = []; - q.pushOnce(forced, { - label: "a", - onRejected: info => { - reasons.push(info.reason); - return "drop"; - }, - }); - q.nextToolChoice(); - q.reject("error"); - expect(reasons).toEqual(["error"]); - }); - it("requeued directive preserves onRejected so it can re-requeue across aborts", () => { const q = new ToolChoiceQueue(); let rejectCount = 0; diff --git a/packages/coding-agent/test/tool-discovery/initial-tools.test.ts b/packages/coding-agent/test/tool-discovery/initial-tools.test.ts index 50fbca9ea..d37c3dd9b 100644 --- a/packages/coding-agent/test/tool-discovery/initial-tools.test.ts +++ b/packages/coding-agent/test/tool-discovery/initial-tools.test.ts @@ -60,14 +60,6 @@ async function getToolMetadata(): Promise { - it("exposes callable tool factories (back-compat for external SDK callers)", () => { - // External callers may invoke BUILTIN_TOOLS.read(session) directly. Verify the value - // is a function, not a metadata object wrapping a factory. - expect(typeof BUILTIN_TOOLS.read).toBe("function"); - expect(typeof BUILTIN_TOOLS.bash).toBe("function"); - expect(typeof BUILTIN_TOOLS.edit).toBe("function"); - }); - it("sets loading fields on tool definitions without wrapping factories", async () => { const metadata = await getToolMetadata(); const missing = Object.keys(BUILTIN_TOOLS).filter(name => metadata.get(name)?.loadMode === undefined); @@ -76,48 +68,6 @@ describe("BUILTIN_TOOLS public factory map", () => { }); describe("built-in tool loadMode annotations", () => { - it("marks read, bash, edit, and search_tool_bm25 as essential", async () => { - const metadata = await getToolMetadata(); - expect(metadata.get("read")?.loadMode).toBe("essential"); - expect(metadata.get("bash")?.loadMode).toBe("essential"); - expect(metadata.get("edit")?.loadMode).toBe("essential"); - expect(metadata.get("search_tool_bm25")?.loadMode).toBe("essential"); - }); - - it("marks non-essential tools as discoverable", async () => { - const discoverableExpected = [ - "ast_grep", - "ast_edit", - "render_mermaid", - "ask", - "debug", - "eval", - "calc", - "ssh", - "github", - "find", - "search", - "lsp", - "inspect_image", - "browser", - "checkpoint", - "rewind", - "task", - "job", - "recipe", - "irc", - "todo_write", - "web_search", - "write", - "retain", - "recall", - "reflect", - ]; - const metadata = await getToolMetadata(); - const missing = discoverableExpected.filter(name => metadata.get(name)?.loadMode !== "discoverable"); - expect(missing).toEqual([]); - }); - it("provides a summary for every discoverable tool", async () => { const missing: string[] = []; const metadata = await getToolMetadata(); @@ -130,14 +80,6 @@ describe("built-in tool loadMode annotations", () => { }); }); -describe("DEFAULT_ESSENTIAL_TOOL_NAMES", () => { - it("contains the expected defaults", () => { - expect(DEFAULT_ESSENTIAL_TOOL_NAMES).toContain("read"); - expect(DEFAULT_ESSENTIAL_TOOL_NAMES).toContain("bash"); - expect(DEFAULT_ESSENTIAL_TOOL_NAMES).toContain("edit"); - }); -}); - describe("computeEssentialBuiltinNames", () => { it("returns DEFAULT_ESSENTIAL_TOOL_NAMES when override is empty", () => { const settings = Settings.isolated({}); @@ -174,26 +116,6 @@ describe("computeEssentialBuiltinNames", () => { }); describe("tools.discoveryMode settings schema", () => { - it("defaults to off", () => { - const settings = Settings.isolated({}); - expect(settings.get("tools.discoveryMode")).toBe("off"); - }); - - it("accepts mcp-only", () => { - const settings = Settings.isolated({ "tools.discoveryMode": "mcp-only" }); - expect(settings.get("tools.discoveryMode")).toBe("mcp-only"); - }); - - it("accepts all", () => { - const settings = Settings.isolated({ "tools.discoveryMode": "all" }); - expect(settings.get("tools.discoveryMode")).toBe("all"); - }); - - it("tools.essentialOverride defaults to empty array", () => { - const settings = Settings.isolated({}); - expect(settings.get("tools.essentialOverride")).toEqual([]); - }); - it("back-compat: mcp.discoveryMode still accepted", () => { const settings = Settings.isolated({ "mcp.discoveryMode": true }); expect(settings.get("mcp.discoveryMode")).toBe(true); diff --git a/packages/coding-agent/test/tool-discovery/persistence.test.ts b/packages/coding-agent/test/tool-discovery/persistence.test.ts index 5b0f9a7a6..d95fc420d 100644 --- a/packages/coding-agent/test/tool-discovery/persistence.test.ts +++ b/packages/coding-agent/test/tool-discovery/persistence.test.ts @@ -26,11 +26,6 @@ describe("persistence back-compat: buildDiscoverableMCPSearchIndex wraps generic }, ]; - it("returns correct document count", () => { - const index = buildDiscoverableMCPSearchIndex(legacyMCPTools); - expect(index.documents).toHaveLength(2); - }); - it("maps description → summary in the index", () => { const index = buildDiscoverableMCPSearchIndex(legacyMCPTools); // The documents contain DiscoverableTool objects with .summary, not .description diff --git a/packages/coding-agent/test/tool-discovery/subagent.test.ts b/packages/coding-agent/test/tool-discovery/subagent.test.ts index ba5301fbd..eb34191d9 100644 --- a/packages/coding-agent/test/tool-discovery/subagent.test.ts +++ b/packages/coding-agent/test/tool-discovery/subagent.test.ts @@ -6,45 +6,6 @@ import { Settings } from "../../src/config/settings"; // without needing to spin up a full AgentSession or subagent. // ───────────────────────────────────────────────────────────────────────────── -describe("tools.discoveryMode subagent inheritance via settings", () => { - it("'off' propagates to child as 'off'", () => { - const parentSettings = Settings.isolated({ "tools.discoveryMode": "off" }); - // Subagent inherits the same settings object (in task/executor.ts, subagentSettings is - // derived from parent settings). The setting value should be preserved. - const child = Settings.isolated({ "tools.discoveryMode": parentSettings.get("tools.discoveryMode") }); - expect(child.get("tools.discoveryMode")).toBe("off"); - }); - - it("'mcp-only' propagates to child as 'mcp-only'", () => { - const parentSettings = Settings.isolated({ "tools.discoveryMode": "mcp-only" }); - const child = Settings.isolated({ "tools.discoveryMode": parentSettings.get("tools.discoveryMode") }); - expect(child.get("tools.discoveryMode")).toBe("mcp-only"); - }); - - it("'all' propagates to child as 'all'", () => { - const parentSettings = Settings.isolated({ "tools.discoveryMode": "all" }); - const child = Settings.isolated({ "tools.discoveryMode": parentSettings.get("tools.discoveryMode") }); - expect(child.get("tools.discoveryMode")).toBe("all"); - }); - - it("mcp.discoveryMode=true propagates to child as back-compat", () => { - const parentSettings = Settings.isolated({ "mcp.discoveryMode": true }); - const child = Settings.isolated({ "mcp.discoveryMode": parentSettings.get("mcp.discoveryMode") }); - expect(child.get("mcp.discoveryMode")).toBe(true); - }); - - it("explicit toolNames override discovery — child with toolNames=['read'] ignores discovery mode", () => { - // If a subagent definition specifies explicit tools, those take precedence - // over the discovery mode. This is enforced in task/executor.ts by only - // building the toolNames list from the agent.tools if present. - const toolNames = ["read"]; - // When toolNames is explicit, only those tools are used regardless of discovery mode - expect(toolNames).toContain("read"); - expect(toolNames).not.toContain("find"); - expect(toolNames).not.toContain("search"); - }); -}); - describe("effective discovery mode resolution", () => { function resolveEffectiveMode(settings: Settings): "off" | "mcp-only" | "all" { const toolsMode = settings.get("tools.discoveryMode"); diff --git a/packages/coding-agent/test/tools.test.ts b/packages/coding-agent/test/tools.test.ts index 1f5bbec3b..d2aa64057 100644 --- a/packages/coding-agent/test/tools.test.ts +++ b/packages/coding-agent/test/tools.test.ts @@ -260,6 +260,7 @@ describe("Coding Agent Tools", () => { } else { Bun.env.PI_EDIT_VARIANT = originalEditVariant; } + AsyncJobManager.resetForTests(); }); describe("read tool", () => { @@ -1105,6 +1106,7 @@ function b() { deliveries.push(text); }, }); + AsyncJobManager.setInstance(asyncJobManager); const autoBackgroundBashTool = wrapToolWithMetaNotice( new BashTool( createTestToolSession( @@ -1114,7 +1116,6 @@ function b() { "bash.autoBackground.thresholdMs": 50, }), { - asyncJobManager, getSessionId: () => "test-session", }, ), @@ -1138,6 +1139,7 @@ function b() { deliveries.push({ jobId, text }); }, }); + AsyncJobManager.setInstance(asyncJobManager); const autoBackgroundBashTool = wrapToolWithMetaNotice( new BashTool( createTestToolSession( @@ -1147,7 +1149,6 @@ function b() { "bash.autoBackground.thresholdMs": 50, }), { - asyncJobManager, getSessionId: () => "test-session", }, ), @@ -1184,6 +1185,7 @@ function b() { deliveries.push({ jobId, text }); }, }); + AsyncJobManager.setInstance(asyncJobManager); const autoBackgroundBashTool = wrapToolWithMetaNotice( new BashTool( createTestToolSession( @@ -1193,7 +1195,6 @@ function b() { "bash.autoBackground.thresholdMs": 60_000, }), { - asyncJobManager, getSessionId: () => "test-session", }, ), @@ -1274,9 +1275,8 @@ function b() { const manager = new AsyncJobManager({ onJobComplete: async () => {}, }); - const session = createTestToolSession(testDir, Settings.isolated({ "bash.autoBackground.enabled": true }), { - asyncJobManager: manager, - }); + const session = createTestToolSession(testDir, Settings.isolated({ "bash.autoBackground.enabled": true }), {}); + AsyncJobManager.setInstance(manager); const jobTool = JobTool.createIf(session)!; const jobId = manager.register("bash", "test job", async () => "success"); diff --git a/packages/coding-agent/test/tools/ask.test.ts b/packages/coding-agent/test/tools/ask.test.ts index d1b681f85..82d2b7daf 100644 --- a/packages/coding-agent/test/tools/ask.test.ts +++ b/packages/coding-agent/test/tools/ask.test.ts @@ -92,6 +92,36 @@ describe("AskTool cancellation", () => { expect(abort).toHaveBeenCalledTimes(1); }); + it("defaults to no timeout when ask.timeout is unset", async () => { + // Regression for the surprise-auto-select report: a fresh install must let the user + // deliberate indefinitely. The dialog timeout is opt-in via the `ask.timeout` setting. + const tool = new AskTool(createSession()); + const select = vi.fn( + async (_prompt: string, options: string[], _dialogOptions?: { initialIndex?: number; timeout?: number }) => + options[0], + ); + const context = createContext({ select }); + + await tool.execute( + "call-default-no-timeout", + { + questions: [ + { + id: "confirm", + question: "Proceed?", + options: [{ label: "yes" }, { label: "no" }], + }, + ], + }, + undefined, + undefined, + context, + ); + + expect(select).toHaveBeenCalledTimes(1); + expect(select.mock.calls[0]?.[2]?.timeout).toBeUndefined(); + }); + it("still aborts when user explicitly cancels with timeout configured", async () => { const tool = new AskTool( createSession({ diff --git a/packages/coding-agent/test/tools/auto-generated-guard.test.ts b/packages/coding-agent/test/tools/auto-generated-guard.test.ts index 771e55d04..90dbc393a 100644 --- a/packages/coding-agent/test/tools/auto-generated-guard.test.ts +++ b/packages/coding-agent/test/tools/auto-generated-guard.test.ts @@ -80,12 +80,6 @@ describe("assertEditableFileContent", () => { }); describe("assertEditableFile", () => { - it("detects auto-generated filename patterns", async () => { - const filePath = path.join(tempDir, "zz_generated.deepcopy.go"); - await Bun.write(filePath, "package generated"); - await expect(assertEditableFile(filePath)).rejects.toBeInstanceOf(ToolError); - }); - it("detects content marker from file prefix", async () => { const filePath = path.join(tempDir, "service.ts"); await Bun.write(filePath, "// Code generated by sqlc. DO NOT EDIT.\nexport const foo = 1;"); diff --git a/packages/coding-agent/test/tools/edit-renderer.test.ts b/packages/coding-agent/test/tools/edit-renderer.test.ts index b90776173..04302b8c4 100644 --- a/packages/coding-agent/test/tools/edit-renderer.test.ts +++ b/packages/coding-agent/test/tools/edit-renderer.test.ts @@ -1,6 +1,10 @@ import { describe, expect, it } from "bun:test"; +import type { AgentTool } from "@oh-my-pi/pi-agent-core"; import { editToolRenderer } from "@oh-my-pi/pi-coding-agent/edit/renderer"; +import { HL_EDIT_SEP } from "@oh-my-pi/pi-coding-agent/hashline/hash"; +import { ToolExecutionComponent } from "@oh-my-pi/pi-coding-agent/modes/components/tool-execution"; import * as themeModule from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; +import type { TUI } from "@oh-my-pi/pi-tui"; async function getUiTheme() { await themeModule.initTheme(false, undefined, undefined, "dark", "light"); @@ -40,6 +44,32 @@ describe("editToolRenderer", () => { expect(rendered).not.toContain("The first line of the patch must be"); }); + it("shows hashline envelope input while preview diff is not computable yet", async () => { + await getUiTheme(); + const uiStub = { requestRender() {} } as unknown as TUI; + const hashlineTool = { name: "edit", label: "Edit", mode: "hashline" } as unknown as AgentTool; + const component = new ToolExecutionComponent( + "edit", + { + input: [ + "*** Begin Patch", + "@crates/pi-natives/src/shell.rs", + "+ EOF", + `${HL_EDIT_SEP}pub fn streaming_preview() {`, + ].join("\n"), + }, + {}, + hashlineTool, + uiStub, + ); + + const rendered = Bun.stripANSI(component.render(160).join("\n")); + expect(rendered).toContain("crates/pi-natives/src/shell.rs"); + expect(rendered).toContain("+ EOF"); + expect(rendered).toContain(`${HL_EDIT_SEP}pub fn streaming_preview() {`); + expect(rendered).not.toContain("*** Begin Patch"); + }); + it("recognizes compact and quoted hashline input headers", async () => { const uiTheme = await getUiTheme(); const compactComponent = editToolRenderer.renderCall( diff --git a/packages/coding-agent/test/tools/eval-display-text.test.ts b/packages/coding-agent/test/tools/eval-display-text.test.ts new file mode 100644 index 000000000..37acfdd7e --- /dev/null +++ b/packages/coding-agent/test/tools/eval-display-text.test.ts @@ -0,0 +1,137 @@ +import { afterEach, describe, expect, it, vi } from "bun:test"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import * as evalIndex from "@oh-my-pi/pi-coding-agent/eval"; +import * as pyKernel from "@oh-my-pi/pi-coding-agent/eval/py/kernel"; +import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; +import { EvalTool } from "@oh-my-pi/pi-coding-agent/tools/eval"; + +function makeSession(): ToolSession { + return { + cwd: "/tmp/eval-test", + hasUI: false, + getSessionFile: () => null, + getSessionSpawns: () => null, + settings: Settings.isolated(), + }; +} + +function baseResult(overrides: Record = {}) { + return { + output: "", + exitCode: 0, + cancelled: false, + truncated: false, + artifactId: undefined, + totalLines: 0, + totalBytes: 0, + outputLines: 0, + outputBytes: 0, + displayOutputs: [] as unknown[], + ...overrides, + }; +} + +describe("EvalTool display() text surfacing", () => { + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("includes display() JSON values in the text content the model sees", async () => { + vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue( + baseResult({ + displayOutputs: [{ type: "json", data: { stdout: "hi", exit_code: 0 } }], + }) as never, + ); + + const tool = new EvalTool(makeSession()); + const result = await tool.execute("call-display-json", { + input: "```js\ndisplay({ stdout: 'hi', exit_code: 0 });\n```\n", + }); + + const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); + expect(text).toContain("display[1]"); + expect(text).toContain('"stdout": "hi"'); + expect(text).toContain('"exit_code": 0'); + expect(text).not.toBe("(no text output)"); + }); + + it("interleaves stdout text and display() JSON values", async () => { + vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue( + baseResult({ + output: "before\n", + displayOutputs: [{ type: "json", data: [1, 2, 3] }], + }) as never, + ); + + const tool = new EvalTool(makeSession()); + const result = await tool.execute("call-mixed", { + input: "```js\nprint('before'); display([1,2,3]);\n```\n", + }); + + const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); + expect(text).toContain("before"); + expect(text.indexOf("before")).toBeLessThan(text.indexOf("display[1]")); + expect(text).toContain("[\n 1,\n 2,\n 3\n]"); + }); + + it("surfaces displayed images to the model as ImageContent blocks, not inlined base64", async () => { + vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const base64 = Buffer.from([0, 1, 2, 3]).toString("base64"); + vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue( + baseResult({ + displayOutputs: [{ type: "image", data: base64, mimeType: "image/png" }], + }) as never, + ); + + const tool = new EvalTool(makeSession()); + const result = await tool.execute("call-image", { + input: "```js\ndisplay({ type: 'image', data: '...', mimeType: 'image/png' });\n```\n", + }); + + const imageBlocks = result.content.filter(c => c.type === "image"); + expect(imageBlocks).toHaveLength(1); + expect(imageBlocks[0]).toMatchObject({ type: "image", data: base64, mimeType: "image/png" }); + + const textBlocks = result.content.filter(c => c.type === "text"); + const text = textBlocks.map(c => (c.type === "text" ? c.text : "")).join("\n"); + expect(text).not.toContain(base64); // base64 must not leak into text channel + expect(text).toMatch(/displayed 1 image/); + + // Image is in content, so details.images must be empty to avoid double-rendering. + expect(result.details?.images).toBeUndefined(); + }); + + it("still reports (no text output) when nothing was printed or displayed", async () => { + vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue(baseResult() as never); + + const tool = new EvalTool(makeSession()); + const result = await tool.execute("call-empty", { + input: "```js\nconst x = 1;\n```\n", + }); + + const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); + expect(text).toContain("(no output)"); + }); + + it("truncates oversized display values rather than blasting the context", async () => { + vi.spyOn(pyKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true }); + const huge = "x".repeat(20000); + vi.spyOn(evalIndex.jsBackend, "execute").mockResolvedValue( + baseResult({ + displayOutputs: [{ type: "json", data: { payload: huge } }], + }) as never, + ); + + const tool = new EvalTool(makeSession()); + const result = await tool.execute("call-huge", { + input: "```js\ndisplay({ payload: 'x'.repeat(20000) });\n```\n", + }); + + const text = result.content.map(c => (c.type === "text" ? c.text : "")).join("\n"); + expect(text).toContain("chars truncated"); + expect(text.length).toBeLessThan(20000); + }); +}); diff --git a/packages/coding-agent/test/tools/exit-plan-mode.test.ts b/packages/coding-agent/test/tools/exit-plan-mode.test.ts index 26c65e6ca..28fe15dd7 100644 --- a/packages/coding-agent/test/tools/exit-plan-mode.test.ts +++ b/packages/coding-agent/test/tools/exit-plan-mode.test.ts @@ -35,12 +35,6 @@ describe("ExitPlanModeTool", () => { }; } - it("requires title in schema", () => { - const tool = new ExitPlanModeTool(createSession()); - const schema = tool.parameters as { required?: string[] }; - expect(schema.required).toContain("title"); - }); - it("normalizes title to .md final plan path", async () => { const tool = new ExitPlanModeTool(createSession()); const result = await tool.execute("call-1", { title: "WP_MIGRATION_PLAN" }); diff --git a/packages/coding-agent/test/tools/gh-renderer.test.ts b/packages/coding-agent/test/tools/gh-renderer.test.ts deleted file mode 100644 index 3f7071fd5..000000000 --- a/packages/coding-agent/test/tools/gh-renderer.test.ts +++ /dev/null @@ -1,194 +0,0 @@ -import { describe, expect, it } from "bun:test"; -import { sanitizeText } from "@oh-my-pi/pi-natives"; -import { getThemeByName } from "../../src/modes/theme/theme"; -import type { GhToolDetails } from "../../src/tools/gh"; -import { githubToolRenderer } from "../../src/tools/gh-renderer"; -import { toolRenderers } from "../../src/tools/renderers"; - -describe("githubToolRenderer", () => { - it("renders a compact ghw-style run summary", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - - const result: { - content: Array<{ type: string; text?: string }>; - details?: GhToolDetails; - isError?: boolean; - } = { - content: [{ type: "text", text: "llm-visible text stays unchanged" }], - details: { - watch: { - mode: "run", - state: "watching", - repo: "v12-security/v12x", - run: { - id: 23856332053, - workflowName: "CI", - branch: "dev", - jobs: [ - { - id: 1, - name: "Workflow Lint", - status: "completed", - conclusion: "success", - durationSeconds: 55, - }, - { - id: 2, - name: "Frontend Checks", - status: "in_progress", - durationSeconds: 40, - }, - { - id: 3, - name: "Rust Tests", - status: "queued", - durationSeconds: 5, - }, - ], - }, - }, - }, - }; - - const component = githubToolRenderer.renderResult(result, { expanded: false, isPartial: true }, uiTheme); - const rendered = sanitizeText(component.render(64).join("\n")); - - expect(toolRenderers.github).toBeDefined(); - expect(rendered).toContain("watching run #23856332053 on v12-security/v12x"); - expect(rendered).toContain("CI dev #23856332053"); - expect(rendered).toContain(`${uiTheme.status.success} Workflow Lint`); - expect(rendered).toContain(`${uiTheme.status.enabled} Frontend Checks`); - expect(rendered).toContain(`${uiTheme.status.shadowed} Rust Tests`); - expect(rendered).toContain("55s"); - expect(rendered).toContain("40s"); - expect(rendered).toContain("5s"); - expect(rendered).not.toContain("llm-visible text stays unchanged"); - }); - - it("shows failed log tails without dumping the full log when collapsed", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - - const result: { - content: Array<{ type: string; text?: string }>; - details?: GhToolDetails; - isError?: boolean; - } = { - content: [{ type: "text", text: "full markdown result" }], - details: { - watch: { - mode: "run", - state: "completed", - repo: "owner/repo", - run: { - id: 77, - workflowName: "CI", - branch: "feature/bugfix", - conclusion: "failure", - jobs: [ - { - id: 202, - name: "test", - status: "completed", - conclusion: "failure", - durationSeconds: 360, - }, - ], - }, - failedLogs: [ - { - runId: 77, - workflowName: "CI", - jobName: "test", - available: true, - tail: ["alpha", "beta", "gamma", "delta", "epsilon", "zeta"].join("\n"), - }, - ], - }, - }, - }; - - const component = githubToolRenderer.renderResult(result, { expanded: false, isPartial: false }, uiTheme); - const rendered = sanitizeText(component.render(72).join("\n")); - - expect(rendered).toContain("failed logs"); - expect(rendered).toContain("delta"); - expect(rendered).toContain("epsilon"); - expect(rendered).toContain("zeta"); - expect(rendered).not.toContain("alpha"); - expect(rendered).toContain("more log lines"); - }); - - it("renders issue_view as a status header with collapsed body and expand hint", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - - const bodyLines = Array.from({ length: 30 }, (_, i) => `line ${i + 1}`); - const result = { - content: [ - { - type: "text", - text: ["# Issue #903: Bug report", "State: OPEN", "", "## Body", "", ...bodyLines].join("\n"), - }, - ], - }; - - const component = githubToolRenderer.renderResult(result, { expanded: false, isPartial: false }, uiTheme, { - op: "issue_view", - issue: "903", - repo: "owner/repo", - }); - const rendered = sanitizeText(component.render(80).join("\n")); - - expect(rendered).toContain("GitHub Issue"); - expect(rendered).toContain("#903"); - expect(rendered).toContain("owner/repo"); - expect(rendered).toContain("# Issue #903: Bug report"); - expect(rendered).toContain("more lines"); - expect(rendered).not.toContain("line 30"); - }); - - it("renders issue_view fully when expanded", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - - const bodyLines = Array.from({ length: 30 }, (_, i) => `line ${i + 1}`); - const result = { - content: [{ type: "text", text: bodyLines.join("\n") }], - }; - - const component = githubToolRenderer.renderResult(result, { expanded: true, isPartial: false }, uiTheme, { - op: "issue_view", - issue: "https://github.com/owner/repo/issues/903", - }); - const rendered = sanitizeText(component.render(80).join("\n")); - - expect(rendered).toContain("#903"); - expect(rendered).toContain("line 1"); - expect(rendered).toContain("line 30"); - expect(rendered).not.toContain("more lines"); - }); - - it("truncates each line to the available width to avoid overflow", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - - const longLine = "x".repeat(500); - const result = { content: [{ type: "text", text: longLine }] }; - - const component = githubToolRenderer.renderResult(result, { expanded: false, isPartial: false }, uiTheme, { - op: "issue_view", - issue: "1", - }); - const lines = component.render(60); - for (const line of lines) { - expect(sanitizeText(line).length).toBeLessThanOrEqual(60); - } - }); -}); diff --git a/packages/coding-agent/test/tools/gh.test.ts b/packages/coding-agent/test/tools/gh.test.ts index d3d02fd3d..bb2c896bd 100644 --- a/packages/coding-agent/test/tools/gh.test.ts +++ b/packages/coding-agent/test/tools/gh.test.ts @@ -6,7 +6,7 @@ import type { AgentToolContext } from "@oh-my-pi/pi-agent-core"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; -import { GithubTool } from "@oh-my-pi/pi-coding-agent/tools/gh"; +import { buildSearchDateQualifier, GithubTool, parseSearchDateBound } from "@oh-my-pi/pi-coding-agent/tools/gh"; import { wrapToolWithMetaNotice } from "@oh-my-pi/pi-coding-agent/tools/output-meta"; import * as git from "@oh-my-pi/pi-coding-agent/utils/git"; import { getAgentDir, setAgentDir } from "@oh-my-pi/pi-utils"; @@ -458,6 +458,112 @@ describe("github tool", () => { expect(prArgs?.at(-1)).toBe("-label:bug"); }); + it("parseSearchDateBound: relative duration walks back from `now` and returns YYYY-MM-DD", () => { + const now = new Date("2026-05-12T15:00:00Z"); + expect(parseSearchDateBound("3d", now)).toBe("2026-05-09"); + expect(parseSearchDateBound("2w", now)).toBe("2026-04-28"); + expect(parseSearchDateBound("12h", now)).toBe("2026-05-12"); + expect(parseSearchDateBound("1mo", now)).toBe("2026-04-12"); + expect(parseSearchDateBound("1y", now)).toBe("2025-05-12"); + }); + + it("parseSearchDateBound: passes ISO dates through and normalizes ISO datetimes", () => { + expect(parseSearchDateBound("2026-05-01")).toBe("2026-05-01"); + expect(parseSearchDateBound("2026-05-01T08:30:00Z")).toBe("2026-05-01T08:30:00.000Z"); + }); + + it("parseSearchDateBound: rejects unparseable input", () => { + expect(() => parseSearchDateBound("yesterday")).toThrow(/invalid date bound/); + expect(() => parseSearchDateBound(" ")).toThrow(/must not be empty/); + }); + + it("buildSearchDateQualifier: emits >=, <=, or range depending on which bounds are set", () => { + const now = new Date("2026-05-12T00:00:00Z"); + expect(buildSearchDateQualifier("created", "3d", undefined, now)).toBe("created:>=2026-05-09"); + expect(buildSearchDateQualifier("created", undefined, "2026-05-01", now)).toBe("created:<=2026-05-01"); + expect(buildSearchDateQualifier("committer-date", "7d", "1d", now)).toBe("committer-date:2026-05-05..2026-05-11"); + expect(buildSearchDateQualifier("created", undefined, undefined)).toBeUndefined(); + }); + + it("search_issues: appends a created:>= qualifier built from `since`", async () => { + const spy = vi.spyOn(git.github, "json").mockResolvedValue([]); + const tool = new GithubTool(createSession()); + await tool.execute("search-issues", { + op: "search_issues", + query: "is:open", + repo: "owner/repo", + since: "2026-05-01", + limit: 5, + }); + + const args = spy.mock.calls[0]?.[1]; + expect(args?.at(-1)).toBe("is:open created:>=2026-05-01"); + }); + + it("search_prs: builds a qualifier-only query when `query` is omitted", async () => { + const spy = vi.spyOn(git.github, "json").mockResolvedValue([]); + const tool = new GithubTool(createSession()); + await tool.execute("search-prs", { + op: "search_prs", + repo: "owner/repo", + since: "2026-05-01", + until: "2026-05-09", + dateField: "updated", + limit: 5, + }); + + const args = spy.mock.calls[0]?.[1]; + expect(args?.at(-1)).toBe("updated:2026-05-01..2026-05-09"); + }); + + it("search_prs: errors when neither `query` nor a date bound is provided", async () => { + vi.spyOn(git.github, "json").mockResolvedValue([]); + const tool = new GithubTool(createSession()); + await expect(tool.execute("search-prs", { op: "search_prs", repo: "owner/repo" })).rejects.toThrow( + /query is required/, + ); + }); + + it("search_commits: forces `committer-date` regardless of `dateField`", async () => { + const spy = vi.spyOn(git.github, "json").mockResolvedValue([]); + const tool = new GithubTool(createSession()); + await tool.execute("search-commits", { + op: "search_commits", + query: "refactor", + repo: "owner/repo", + since: "2026-05-01", + dateField: "updated", + limit: 5, + }); + + const args = spy.mock.calls[0]?.[1]; + expect(args?.at(-1)).toBe("refactor committer-date:>=2026-05-01"); + }); + + it("search_repos: maps dateField=updated to the `pushed:` qualifier", async () => { + const spy = vi.spyOn(git.github, "json").mockResolvedValue([]); + const tool = new GithubTool(createSession()); + await tool.execute("search-repos", { + op: "search_repos", + query: "language:rust", + since: "2026-05-01", + dateField: "updated", + limit: 1, + }); + + const args = spy.mock.calls[0]?.[1]; + expect(args?.at(-1)).toBe("language:rust pushed:>=2026-05-01"); + }); + + it("search_code: rejects since/until since GitHub code search has no date qualifier", async () => { + const spy = vi.spyOn(git.github, "json").mockResolvedValue([]); + const tool = new GithubTool(createSession()); + await expect(tool.execute("search-code", { op: "search_code", query: "foo", since: "3d" })).rejects.toThrow( + /search_code does not support since\/until/, + ); + expect(spy).not.toHaveBeenCalled(); + }); + it("formats code search results with paths, repo, sha, and match fragment", async () => { vi.spyOn(git.github, "json").mockResolvedValue([ { diff --git a/packages/coding-agent/test/tools/schema-validation.test.ts b/packages/coding-agent/test/tools/schema-validation.test.ts index 54e93fc07..311ebff61 100644 --- a/packages/coding-agent/test/tools/schema-validation.test.ts +++ b/packages/coding-agent/test/tools/schema-validation.test.ts @@ -267,84 +267,6 @@ describe("tool schema validation (post-sanitization)", () => { expect(allViolations).toEqual([]); }); - it("no sanitized schema contains $schema declaration", async () => { - const session = createTestSession(); - const tools = await createTools(session); - - for (const tool of tools) { - const schema = tool.parameters; - if (!schema) continue; - - const sanitized = sanitizeSchemaForGoogle(schema); - const violations = validateSchema(sanitized, tool.name).filter(v => v.key === "$schema"); - expect(violations).toEqual([]); - } - }); - - it("no sanitized schema contains $ref or $defs", async () => { - const session = createTestSession(); - const tools = await createTools(session); - - for (const tool of tools) { - const schema = tool.parameters; - if (!schema) continue; - - const sanitized = sanitizeSchemaForGoogle(schema); - const violations = validateSchema(sanitized, tool.name).filter(v => v.key === "$ref" || v.key === "$defs"); - expect(violations).toEqual([]); - } - }); - - it("no sanitized schema contains Draft 2020-12 specific features", async () => { - const session = createTestSession(); - const tools = await createTools(session); - - const draft2020Features = [ - "prefixItems", - "$dynamicRef", - "$dynamicAnchor", - "unevaluatedProperties", - "unevaluatedItems", - ]; - - for (const tool of tools) { - const schema = tool.parameters; - if (!schema) continue; - - const sanitized = sanitizeSchemaForGoogle(schema); - const violations = validateSchema(sanitized, tool.name).filter(v => draft2020Features.includes(v.key)); - expect(violations).toEqual([]); - } - }); - - it("sanitization removes const (converts to enum)", async () => { - const session = createTestSession(); - const tools = await createTools(session); - - for (const tool of tools) { - const schema = tool.parameters; - if (!schema) continue; - - const sanitized = sanitizeSchemaForGoogle(schema); - const violations = validateSchema(sanitized, tool.name).filter(v => v.key === "const"); - expect(violations).toEqual([]); - } - }); - - it("no sanitized schema contains examples field", async () => { - const session = createTestSession(); - const tools = await createTools(session); - - for (const tool of tools) { - const schema = tool.parameters; - if (!schema) continue; - - const sanitized = sanitizeSchemaForGoogle(schema); - const violations = validateSchema(sanitized, tool.name).filter(v => v.key === "examples"); - expect(violations).toEqual([]); - } - }); - it("hidden tools also have valid sanitized schemas", async () => { const session = createTestSession(); @@ -365,41 +287,6 @@ describe("tool schema validation (post-sanitization)", () => { } } }); - - it("logs warnings for potentially problematic features (non-blocking)", async () => { - const session = createTestSession(); - const tools = await createTools(session); - - const warnings: { tool: string; violations: SchemaViolation[] }[] = []; - - for (const tool of tools) { - const schema = tool.parameters; - if (!schema) continue; - - const sanitized = sanitizeSchemaForGoogle(schema); - const violations = validateSchema(sanitized, tool.name); - const toolWarnings = violations.filter(v => v.severity === "warning"); - - if (toolWarnings.length > 0) { - warnings.push({ tool: tool.name, violations: toolWarnings }); - } - } - - // Log warnings but don't fail - these are advisory - if (warnings.length > 0) { - const message = warnings - .map(({ tool, violations }) => { - const details = violations.map(v => ` - ${v.path}: ${v.key} = ${JSON.stringify(v.value)}`).join("\n"); - return `${tool}:\n${details}`; - }) - .join("\n\n"); - - console.log(`Schema warnings (non-blocking):\n\n${message}`); - } - - // This test passes regardless - warnings are informational - expect(true).toBe(true); - }); }); describe("validateSchema helper", () => { @@ -452,12 +339,6 @@ describe("validateSchema helper", () => { expect(warning?.severity).toBe("warning"); }); - it("does not warn on additionalProperties: true", () => { - const schema = { type: "object", additionalProperties: true }; - const violations = validateSchema(schema); - expect(violations.some(v => v.key === "additionalProperties")).toBe(false); - }); - it("warns on format keyword", () => { const schema = { type: "string", format: "uri" }; const violations = validateSchema(schema); @@ -490,17 +371,4 @@ describe("validateSchema helper", () => { const violations = validateSchema(schema); expect(violations.some(v => v.key === "const")).toBe(true); }); - - it("returns empty array for valid schema", () => { - const schema = { - type: "object", - properties: { - name: { type: "string", description: "User name" }, - age: { type: "number", minimum: 0 }, - }, - required: ["name"], - }; - const violations = validateSchema(schema); - expect(violations.filter(v => v.severity === "error")).toEqual([]); - }); }); diff --git a/packages/coding-agent/test/tools/search-internal-urls.test.ts b/packages/coding-agent/test/tools/search-internal-urls.test.ts index 21be2d6a4..da2928054 100644 --- a/packages/coding-agent/test/tools/search-internal-urls.test.ts +++ b/packages/coding-agent/test/tools/search-internal-urls.test.ts @@ -3,11 +3,10 @@ import * as fs from "node:fs/promises"; import * as os from "node:os"; import * as path from "node:path"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; -import { ArtifactProtocolHandler } from "@oh-my-pi/pi-coding-agent/internal-urls/artifact-protocol"; -import { LocalProtocolHandler } from "@oh-my-pi/pi-coding-agent/internal-urls/local-protocol"; -import { InternalUrlRouter } from "@oh-my-pi/pi-coding-agent/internal-urls/router"; +import { InternalUrlRouter, LocalProtocolHandler } from "@oh-my-pi/pi-coding-agent/internal-urls"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/tools"; import { SearchTool } from "@oh-my-pi/pi-coding-agent/tools/search"; +import { AgentRegistry } from "../../src/registry/agent-registry"; function getResultText(result: { content: Array<{ type: string; text?: string }> }): string { return result.content @@ -24,10 +23,27 @@ describe("SearchTool internal URL resolution", () => { tmpDir = await fs.mkdtemp(path.join(os.tmpdir(), "grep-test-")); artifactsDir = path.join(tmpDir, "artifacts"); await fs.mkdir(artifactsDir); + + AgentRegistry.resetGlobalForTests(); + LocalProtocolHandler.resetOverrideForTests(); + InternalUrlRouter.resetForTests(); + + // Register a synthetic main session so artifact:// can derive + // `artifactsDir` from its sessionFile (sessionFile.slice(0,-6)). + AgentRegistry.global().register({ + id: "test-main", + displayName: "test", + kind: "main", + session: null, + sessionFile: `${artifactsDir}.jsonl`, + }); }); afterEach(async () => { await fs.rm(tmpDir, { recursive: true, force: true }); + AgentRegistry.resetGlobalForTests(); + LocalProtocolHandler.resetOverrideForTests(); + InternalUrlRouter.resetForTests(); }); function createSession(overrides: Partial = {}): ToolSession { @@ -41,18 +57,11 @@ describe("SearchTool internal URL resolution", () => { }; } - function createRouterWithArtifacts(): InternalUrlRouter { - const router = new InternalUrlRouter(); - router.register(new ArtifactProtocolHandler({ getArtifactsDir: () => artifactsDir })); - return router; - } - it("resolves artifact:// URL to backing file and greps it", async () => { const content = "line one\nfound the needle here\nline three\n"; await Bun.write(path.join(artifactsDir, "5.bash.log"), content); - const router = createRouterWithArtifacts(); - const session = createSession({ internalRouter: router }); + const session = createSession(); const tool = new SearchTool(session); const result = await tool.execute("test-call", { @@ -68,8 +77,7 @@ describe("SearchTool internal URL resolution", () => { const content = "ERROR: connection refused\nWARN: timeout\nERROR: disk full\nINFO: ok\n"; await Bun.write(path.join(artifactsDir, "3.python.log"), content); - const router = createRouterWithArtifacts(); - const session = createSession({ internalRouter: router }); + const session = createSession(); const tool = new SearchTool(session); const result = await tool.execute("test-call", { @@ -85,31 +93,18 @@ describe("SearchTool internal URL resolution", () => { }); it("throws when internal URL has no sourcePath", async () => { - const router = new InternalUrlRouter(); - router.register({ - scheme: "agent", - immutable: true, - async resolve() { - return { - url: "agent://0", - content: "some content", - contentType: "text/plain" as const, - }; - }, - }); - - const session = createSession({ internalRouter: router }); + const session = createSession(); const tool = new SearchTool(session); - expect(tool.execute("test-call", { pattern: "foo", paths: ["agent://0"] })).rejects.toThrow( - "Cannot search internal URL without a backing file", + expect(tool.execute("test-call", { pattern: "foo", paths: ["artifact://999"] })).rejects.toThrow( + "Artifact 999 not found", ); }); it("falls back to normal path resolution when no internalRouter", async () => { await Bun.write(path.join(tmpDir, "test.txt"), "hello world\n"); - const session = createSession(); // no internalRouter + const session = createSession(); const tool = new SearchTool(session); const result = await tool.execute("test-call", { @@ -124,8 +119,7 @@ describe("SearchTool internal URL resolution", () => { it("falls back to normal resolution for non-internal URLs", async () => { await Bun.write(path.join(tmpDir, "data.log"), "some data here\n"); - const router = createRouterWithArtifacts(); - const session = createSession({ internalRouter: router }); + const session = createSession(); const tool = new SearchTool(session); const result = await tool.execute("test-call", { @@ -141,8 +135,7 @@ describe("SearchTool internal URL resolution", () => { const content = "alpha line\nbeta needle line\ngamma line\n"; await Bun.write(path.join(artifactsDir, "9.bash.log"), content); - const router = createRouterWithArtifacts(); - const session = createSession({ internalRouter: router, hasEditTool: true }); + const session = createSession({ hasEditTool: true }); const tool = new SearchTool(session); const result = await tool.execute("test-call", { @@ -161,14 +154,9 @@ describe("SearchTool internal URL resolution", () => { await fs.mkdir(localRoot, { recursive: true }); await Bun.write(path.join(localRoot, "plan.md"), "alpha line\nbeta needle line\ngamma line\n"); - const router = new InternalUrlRouter(); - router.register( - new LocalProtocolHandler({ - getArtifactsDir: () => artifactsDir, - getSessionId: () => "session", - }), - ); - const session = createSession({ internalRouter: router, hasEditTool: true }); + LocalProtocolHandler.setOverride({ getArtifactsDir: () => artifactsDir, getSessionId: () => "session" }); + + const session = createSession({ hasEditTool: true }); const tool = new SearchTool(session); const result = await tool.execute("test-call", { @@ -187,8 +175,7 @@ describe("SearchTool internal URL resolution", () => { await Bun.write(path.join(artifactsDir, "11.bash.log"), content); await Bun.write(path.join(tmpDir, "mixed.txt"), "mixed needle line\n"); - const router = createRouterWithArtifacts(); - const session = createSession({ internalRouter: router, hasEditTool: true }); + const session = createSession({ hasEditTool: true }); const tool = new SearchTool(session); const result = await tool.execute("test-call", { @@ -203,8 +190,7 @@ describe("SearchTool internal URL resolution", () => { }); it("throws on nonexistent artifact ID", async () => { - const router = createRouterWithArtifacts(); - const session = createSession({ internalRouter: router }); + const session = createSession(); const tool = new SearchTool(session); expect(tool.execute("test-call", { pattern: "foo", paths: ["artifact://999"] })).rejects.toThrow( diff --git a/packages/coding-agent/test/tools/search-tool-bm25.test.ts b/packages/coding-agent/test/tools/search-tool-bm25.test.ts index 0e22f5010..392b639e3 100644 --- a/packages/coding-agent/test/tools/search-tool-bm25.test.ts +++ b/packages/coding-agent/test/tools/search-tool-bm25.test.ts @@ -1,11 +1,10 @@ import { describe, expect, it } from "bun:test"; -import { getThemeByName } from "@oh-my-pi/pi-coding-agent/modes/theme/theme"; import { Settings } from "../../src/config/settings"; // Back-compat import check — these re-exports from mcp/discoverable-tool-metadata should still work import { buildDiscoverableMCPSearchIndex, type DiscoverableMCPTool } from "../../src/mcp/discoverable-tool-metadata"; import type { DiscoverableMCPSearchIndex, DiscoverableTool } from "../../src/tool-discovery/tool-index"; import type { ToolSession } from "../../src/tools/index"; -import { SearchToolBm25Tool, searchToolBm25Renderer } from "../../src/tools/search-tool-bm25"; +import { SearchToolBm25Tool } from "../../src/tools/search-tool-bm25"; type TestDiscoverableTool = DiscoverableTool; @@ -145,187 +144,6 @@ describe("SearchToolBm25Tool", () => { ]); }); - it("renders a titled discovery summary instead of the raw tool name", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - const renderedCall = searchToolBm25Renderer.renderCall( - { query: "github issue", limit: 2 }, - { expanded: false, isPartial: false }, - uiTheme, - ); - expect(renderedCall.render(120).join("\n")).toContain("Tool Discovery"); - expect(renderedCall.render(120).join("\n")).not.toContain("search_tool_bm25"); - - const renderedResult = searchToolBm25Renderer.renderResult( - { - content: [{ type: "text", text: "" }], - details: { - query: "github issue", - limit: 2, - total_tools: 3, - activated_tools: ["mcp__github_create_issue"], - active_selected_tools: ["mcp__github_create_issue"], - tools: [ - { - name: "mcp__github_create_issue", - label: "github/create_issue", - description: "Create a GitHub issue in the selected repository", - server_name: "github", - mcp_tool_name: "create_issue", - schema_keys: ["owner", "repo", "title", "body"], - score: 1.234567, - }, - ], - }, - }, - { expanded: false, isPartial: false }, - uiTheme, - ); - const renderedText = renderedResult.render(120).join("\n"); - expect(renderedText).toContain("Tool Discovery"); - expect(renderedText).toContain("github/create_issue"); - expect(renderedText).toContain("1 active"); - expect(renderedText).toContain("limit:2"); - expect(renderedText).not.toContain("keys:"); - expect(renderedText).not.toContain("search_tool_bm25"); - }); - - it("truncates fallback discovery text before rendering", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - const longLine = "Long discovery output ".repeat(20); - const renderedResult = searchToolBm25Renderer.renderResult( - { - content: [{ type: "text", text: longLine }], - }, - { expanded: false, isPartial: false }, - uiTheme, - ); - const renderedText = renderedResult.render(200).join("\n"); - expect(renderedText).toContain("Tool Discovery"); - expect(renderedText).toContain("Long discovery output Long discovery output"); - expect(renderedText).not.toContain(longLine); - }); - - it("tolerates partially streamed render-call arguments", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - const renderedCall = searchToolBm25Renderer.renderCall( - {} as never, - { expanded: false, isPartial: true }, - uiTheme, - ); - expect(renderedCall.render(120).join("\n")).toContain("(empty query)"); - }); - - it("sanitizes MCP metadata before rendering discovery output", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - const renderedResult = searchToolBm25Renderer.renderResult( - { - content: [{ type: "text", text: "" }], - details: { - query: "github\tissue", - limit: 2, - total_tools: 1, - activated_tools: ["mcp__github_create_issue"], - active_selected_tools: ["mcp__github_create_issue"], - tools: [ - { - name: "mcp__github_create_issue", - label: "github\t/create_issue", - description: "Create\ta GitHub issue", - server_name: "git\thub", - mcp_tool_name: "create_issue", - schema_keys: ["owner", "repo"], - score: 1.234567, - }, - ], - }, - }, - { expanded: true, isPartial: false }, - uiTheme, - ); - const renderedText = renderedResult.render(120).join("\n"); - expect(renderedText).not.toContain("\t"); - expect(renderedText).toContain("github issue"); - expect(renderedText).toContain("git hub"); - expect(renderedText).toContain("Create a GitHub issue"); - }); - - it("shows at most five tools in collapsed renderer output", async () => { - const theme = await getThemeByName("dark"); - expect(theme).toBeDefined(); - const uiTheme = theme!; - const tools = Array.from({ length: 6 }, (_, index) => ({ - name: `mcp__github_tool_${index + 1}`, - label: `github/tool_${index + 1}`, - description: `GitHub tool ${index + 1}`, - server_name: "github", - mcp_tool_name: `tool_${index + 1}`, - schema_keys: ["owner", "repo"], - score: 1 - index * 0.01, - })); - const rendered = searchToolBm25Renderer.renderResult( - { - content: [{ type: "text", text: "" }], - details: { - query: "github tools", - limit: 8, - total_tools: 6, - activated_tools: tools.map(tool => tool.name), - active_selected_tools: tools.map(tool => tool.name), - tools, - }, - }, - { expanded: false, isPartial: false }, - uiTheme, - ); - const renderedText = rendered.render(120).join("\n"); - expect(renderedText).toContain("github/tool_5"); - expect(renderedText).not.toContain("github/tool_6"); - expect(renderedText).toContain("1 more tool"); - }); - - it("defaults to 8 results and lets callers override the limit", async () => { - const manyTools: DiscoverableTool[] = Array.from({ length: 10 }, (_, index) => - mcpTool( - `mcp__github_tool_${index + 1}`, - "github", - `tool_${index + 1}`, - `GitHub tool ${index + 1} for repository workflows`, - ["owner", "repo", `field_${index + 1}`], - ), - ); - const tool = new SearchToolBm25Tool(createSession(manyTools)); - - const defaultResult = await tool.execute("call-default", { query: "github" }); - expect(defaultResult.details?.limit).toBe(8); - expect(defaultResult.details?.tools).toHaveLength(8); - expect(defaultResult.details?.active_selected_tools).toHaveLength(8); - const defaultContent = defaultResult.content[0]; - expect(defaultContent).toBeDefined(); - expect(defaultContent).toEqual({ - type: "text", - text: JSON.stringify({ - query: "github", - activated_tools: defaultResult.details?.activated_tools, - match_count: 8, - total_tools: 10, - }), - }); - - const limitedTool = new SearchToolBm25Tool(createSession(manyTools)); - const limitedResult = await limitedTool.execute("call-limited", { query: "github", limit: 3 }); - expect(limitedResult.details?.limit).toBe(3); - expect(limitedResult.details?.tools).toHaveLength(3); - expect(limitedResult.details?.active_selected_tools).toHaveLength(3); - }); - it("returns ranked matches and unions activated tools across repeated searches", async () => { const session = createSession(discoverableTools); const tool = new SearchToolBm25Tool(session); diff --git a/packages/coding-agent/test/tools/web-scrapers/academic.test.ts b/packages/coding-agent/test/tools/web-scrapers/academic.test.ts index ad15f7994..724faef00 100644 --- a/packages/coding-agent/test/tools/web-scrapers/academic.test.ts +++ b/packages/coding-agent/test/tools/web-scrapers/academic.test.ts @@ -8,11 +8,6 @@ import type { RenderResult } from "@oh-my-pi/pi-coding-agent/web/scrapers/types" const SKIP = !Bun.env.WEB_FETCH_INTEGRATION; describe.skipIf(SKIP)("handleSemanticScholar", () => { - it("returns null for non-S2 URLs", async () => { - const result = await handleSemanticScholar("https://example.com", 10); - expect(result).toBeNull(); - }); - it("fetches a known paper", async () => { // "Attention Is All You Need" paper const result = await handleSemanticScholar( @@ -92,11 +87,6 @@ describe.skipIf(SKIP)("handlePubMed", () => { return cachedKnownPubMed; }; - it("returns null for non-PubMed URLs", async () => { - const result = await handlePubMed("https://example.com", 10); - expect(result).toBeNull(); - }); - it("fetches a known article from pubmed.ncbi.nlm.nih.gov", async () => { // PMID 33782455 - COVID-19 vaccine paper const result = await fetchKnownPubMed(); @@ -149,11 +139,6 @@ describe.skipIf(SKIP)("handlePubMed", () => { }); describe.skipIf(SKIP)("handleArxiv", () => { - it("returns null for non-arXiv URLs", async () => { - const result = await handleArxiv("https://example.com", 10000); - expect(result).toBeNull(); - }); - it("fetches a known paper", async () => { // "Attention Is All You Need" paper const result = await handleArxiv("https://arxiv.org/abs/1706.03762", 30000); @@ -209,11 +194,6 @@ describe.skipIf(SKIP)("handleArxiv", () => { }); describe.skipIf(SKIP)("handleIacr", () => { - it("returns null for non-IACR URLs", async () => { - const result = await handleIacr("https://example.com", 10000); - expect(result).toBeNull(); - }); - it("fetches a known ePrint", async () => { // Using a well-known paper const result = await handleIacr("https://eprint.iacr.org/2023/123", 30000); diff --git a/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts b/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts index 78a08cb04..d54a9ee47 100644 --- a/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts +++ b/packages/coding-agent/test/tools/web-scrapers/git-hosting.test.ts @@ -86,16 +86,6 @@ describe.skipIf(SKIP)("handleGitHub", () => { expect(result).toBeDefined(); }); - it("fetches pull request", async () => { - const result = await handleGitHub("https://github.com/facebook/react/pull/1", 20000); - if (result !== null) { - expect(result.method).toBe("github-pr"); - expect(result.contentType).toBe("text/markdown"); - expect(result.content.length).toBeGreaterThan(0); - } - expect(result).toBeDefined(); - }); - it("fetches issues list", async () => { const result = await handleGitHub("https://github.com/facebook/react/issues", 20000); if (result !== null) { @@ -106,69 +96,12 @@ describe.skipIf(SKIP)("handleGitHub", () => { expect(result).toBeDefined(); }); - it("handles repository with underscore in name", async () => { - const result = await handleGitHub("https://github.com/rust-lang/rust-analyzer", 20000); - if (result !== null) { - expect(result.method).toBe("github-repo"); - } - expect(result).toBeDefined(); - }); - - it("handles repository with dash in name", async () => { - const result = await handleGitHub("https://github.com/vercel/next.js", 20000); - if (result !== null) { - expect(result.method).toBe("github-repo"); - } - expect(result).toBeDefined(); - }); - - it("returns null for invalid URL structure", async () => { - const result = await handleGitHub("https://github.com/", 10000); - expect(result).toBeNull(); - }); - - it("returns null for single path segment", async () => { - const result = await handleGitHub("https://github.com/facebook", 10000); - expect(result).toBeNull(); - }); - it("handles pulls list endpoint", async () => { const result = await handleGitHub("https://github.com/facebook/react/pulls", 20000); // Should be handled as pulls list but currently falls back to null // This tests the actual behavior expect(result).toBeDefined(); }); - - it("fetches file with path containing multiple directories", async () => { - const result = await handleGitHub( - "https://github.com/facebook/react/blob/main/packages/react/package.json", - 20000, - ); - expect(result).not.toBeNull(); - expect(result?.method).toBe("github-raw"); - }); - - it("fetches deeply nested directory", async () => { - const result = await handleGitHub("https://github.com/facebook/react/tree/main/packages/react/src", 20000); - if (result !== null) { - expect(result.method).toBe("github-tree"); - } - expect(result).toBeDefined(); - }); - - it("returns null for discussion URLs", async () => { - const result = await handleGitHub("https://github.com/facebook/react/discussions", 10000); - // Discussions not fully implemented, should return null - expect(result).toBeDefined(); - }); - - it("handles trailing slash in repository URL", async () => { - const result = await handleGitHub("https://github.com/facebook/react/", 20000); - if (result !== null) { - expect(result.method).toBe("github-repo"); - } - expect(result).toBeDefined(); - }); }); // ============================================================================= diff --git a/packages/coding-agent/test/tools/web-scrapers/package-managers.test.ts b/packages/coding-agent/test/tools/web-scrapers/package-managers.test.ts index ac81231fe..4fedcf761 100644 --- a/packages/coding-agent/test/tools/web-scrapers/package-managers.test.ts +++ b/packages/coding-agent/test/tools/web-scrapers/package-managers.test.ts @@ -9,16 +9,6 @@ import { handleRubyGems } from "@oh-my-pi/pi-coding-agent/web/scrapers/rubygems" const SKIP = !Bun.env.WEB_FETCH_INTEGRATION; describe.skipIf(SKIP)("handleBrew", () => { - it("returns null for non-Homebrew URLs", async () => { - const result = await handleBrew("https://example.com", 20); - expect(result).toBeNull(); - }); - - it("returns null for non-package Homebrew URLs", async () => { - const result = await handleBrew("https://formulae.brew.sh/", 20); - expect(result).toBeNull(); - }); - it("fetches wget formula", async () => { const result = await handleBrew("https://formulae.brew.sh/formula/wget", 20); expect(result).not.toBeNull(); @@ -43,16 +33,6 @@ describe.skipIf(SKIP)("handleBrew", () => { }); describe.skipIf(SKIP)("handleAur", () => { - it("returns null for non-AUR URLs", async () => { - const result = await handleAur("https://example.com", 20); - expect(result).toBeNull(); - }); - - it("returns null for non-package AUR URLs", async () => { - const result = await handleAur("https://aur.archlinux.org/", 20); - expect(result).toBeNull(); - }); - it("fetches yay package", async () => { const result = await handleAur("https://aur.archlinux.org/packages/yay", 20); expect(result).not.toBeNull(); @@ -67,16 +47,6 @@ describe.skipIf(SKIP)("handleAur", () => { }); describe.skipIf(SKIP)("handleRubyGems", () => { - it("returns null for non-RubyGems URLs", async () => { - const result = await handleRubyGems("https://example.com", 20); - expect(result).toBeNull(); - }); - - it("returns null for non-gem RubyGems URLs", async () => { - const result = await handleRubyGems("https://rubygems.org/", 20); - expect(result).toBeNull(); - }); - it("fetches rails gem", async () => { const result = await handleRubyGems("https://rubygems.org/gems/rails", 20); expect(result).not.toBeNull(); @@ -90,16 +60,6 @@ describe.skipIf(SKIP)("handleRubyGems", () => { }); describe.skipIf(SKIP)("handleNuGet", () => { - it("returns null for non-NuGet URLs", async () => { - const result = await handleNuGet("https://example.com", 20); - expect(result).toBeNull(); - }); - - it("returns null for non-package NuGet URLs", async () => { - const result = await handleNuGet("https://www.nuget.org/", 20); - expect(result).toBeNull(); - }); - it("fetches Newtonsoft.Json package", async () => { const result = await handleNuGet("https://www.nuget.org/packages/Newtonsoft.Json", 20); expect(result).not.toBeNull(); @@ -113,16 +73,6 @@ describe.skipIf(SKIP)("handleNuGet", () => { }); describe.skipIf(SKIP)("handlePackagist", () => { - it("returns null for non-Packagist URLs", async () => { - const result = await handlePackagist("https://example.com", 20); - expect(result).toBeNull(); - }); - - it("returns null for non-package Packagist URLs", async () => { - const result = await handlePackagist("https://packagist.org/", 20); - expect(result).toBeNull(); - }); - it("fetches laravel/framework package", async () => { const result = await handlePackagist("https://packagist.org/packages/laravel/framework", 20); expect(result).not.toBeNull(); @@ -136,16 +86,6 @@ describe.skipIf(SKIP)("handlePackagist", () => { }); describe.skipIf(SKIP)("handleMaven", () => { - it("returns null for non-Maven URLs", async () => { - const result = await handleMaven("https://example.com", 20); - expect(result).toBeNull(); - }); - - it("returns null for non-artifact Maven URLs", async () => { - const result = await handleMaven("https://search.maven.org/", 20); - expect(result).toBeNull(); - }); - it("fetches commons-lang3 artifact from search.maven.org", async () => { const result = await handleMaven("https://search.maven.org/artifact/org.apache.commons/commons-lang3", 20); expect(result).not.toBeNull(); diff --git a/packages/coding-agent/test/tools/web-scrapers/social-extended.test.ts b/packages/coding-agent/test/tools/web-scrapers/social-extended.test.ts index b6a91de53..278aaf529 100644 --- a/packages/coding-agent/test/tools/web-scrapers/social-extended.test.ts +++ b/packages/coding-agent/test/tools/web-scrapers/social-extended.test.ts @@ -35,57 +35,6 @@ describe.skipIf(SKIP)("handleMastodon", () => { { timeout: 30000 }, ); - it( - "fetches a Mastodon post", - async () => { - // Gargron's post ID 1 - the first ever Mastodon post - const result = await handleMastodon("https://mastodon.social/@Gargron/1", 20); - // Post 1 may not exist anymore; check gracefully - if (result !== null) { - expect(result.method).toBe("mastodon"); - expect(result.contentType).toBe("text/markdown"); - expect(result.content).toContain("Post by"); - expect(result.content).toContain("@Gargron"); - expect(result.fetchedAt).toBeTruthy(); - expect(result.truncated).toBeDefined(); - expect(result.notes?.[0]).toContain("Mastodon API"); - } - }, - { timeout: 30000 }, - ); - - it( - "handles a stable pinned post", - async () => { - // Use a well-known post from mastodon.social - Gargron's announcement post - const result = await handleMastodon("https://mastodon.social/@Gargron/109318821117356215", 20); - // May not exist, check gracefully - if (result !== null) { - expect(result.method).toBe("mastodon"); - expect(result.contentType).toBe("text/markdown"); - expect(result.content).toContain("@Gargron"); - expect(result.content).toContain("replies"); - expect(result.content).toContain("boosts"); - expect(result.content).toContain("favorites"); - expect(result.fetchedAt).toBeTruthy(); - } - }, - { timeout: 30000 }, - ); - - it( - "includes recent posts in profile", - async () => { - const result = await handleMastodon("https://mastodon.social/@Gargron", 20); - expect(result).not.toBeNull(); - // May include recent posts section - if (result?.content?.includes("## Recent Posts")) { - expect(result.content).toMatch(/###\s+\w+/); // Date header - } - }, - { timeout: 30000 }, - ); - it("returns null for non-Mastodon instance with @user pattern", async () => { // A site that has @user pattern but isn't Mastodon const result = await handleMastodon("https://twitter.com/@jack", 20); @@ -140,53 +89,4 @@ describe.skipIf(SKIP)("handleBluesky", () => { }, { timeout: 30000 }, ); - - it( - "fetches a Bluesky post", - async () => { - // A post from bsky.app - use a well-known stable post - const result = await handleBluesky("https://bsky.app/profile/bsky.app/post/3juzlwllznd24", 20); - // Post may not exist, check gracefully - if (result !== null) { - expect(result.method).toBe("bluesky-api"); - expect(result.contentType).toBe("text/markdown"); - expect(result.content).toContain("# Bluesky Post"); - expect(result.content).toContain("@bsky.app"); - expect(result.fetchedAt).toBeTruthy(); - expect(result.truncated).toBeDefined(); - expect(result.notes?.[0]).toContain("AT URI"); - } - }, - { timeout: 30000 }, - ); - - it( - "includes post stats", - async () => { - const result = await handleBluesky("https://bsky.app/profile/bsky.app/post/3juzlwllznd24", 20); - // Stats include likes, reposts, replies - if (result?.content) { - // Should have some engagement markers - const hasStats = - result.content.includes("❤️") || result.content.includes("🔁") || result.content.includes("💬"); - expect(hasStats || result.content.includes("# Bluesky Post")).toBe(true); - } - }, - { timeout: 30000 }, - ); - - it( - "handles www.bsky.app URLs", - async () => { - const result = await handleBluesky("https://www.bsky.app/profile/bsky.app", 20); - expect(result).not.toBeNull(); - expect(result?.method).toBe("bluesky-api"); - }, - { timeout: 30000 }, - ); - - it("returns null for invalid profile handle", async () => { - const result = await handleBluesky("https://bsky.app/profile/", 20); - expect(result).toBeNull(); - }); }); diff --git a/packages/coding-agent/test/tools/web-scrapers/social.test.ts b/packages/coding-agent/test/tools/web-scrapers/social.test.ts index 38d0832b0..05df4cad8 100644 --- a/packages/coding-agent/test/tools/web-scrapers/social.test.ts +++ b/packages/coding-agent/test/tools/web-scrapers/social.test.ts @@ -6,11 +6,6 @@ import { handleTwitter } from "@oh-my-pi/pi-coding-agent/web/scrapers/twitter"; const SKIP = !Bun.env.WEB_FETCH_INTEGRATION; describe.skipIf(SKIP)("handleTwitter", () => { - it("returns null for non-Twitter URLs", async () => { - const result = await handleTwitter("https://example.com", 10); - expect(result).toBeNull(); - }); - it( "handles twitter.com status URLs", async () => { @@ -47,26 +42,6 @@ describe.skipIf(SKIP)("handleTwitter", () => { { timeout: 30000 }, ); - it( - "handles www.twitter.com URLs", - async () => { - const result = await handleTwitter("https://www.twitter.com/twitter/status/1", 10000); - expect(result).not.toBeNull(); - expect(result?.method).toMatch(/^twitter/); - }, - { timeout: 30000 }, - ); - - it( - "handles www.x.com URLs", - async () => { - const result = await handleTwitter("https://www.x.com/twitter/status/1", 10000); - expect(result).not.toBeNull(); - expect(result?.method).toMatch(/^twitter/); - }, - { timeout: 30000 }, - ); - it( "may fail due to Nitter availability", async () => { @@ -84,11 +59,6 @@ describe.skipIf(SKIP)("handleTwitter", () => { }); describe.skipIf(SKIP)("handleReddit", () => { - it("returns null for non-Reddit URLs", async () => { - const result = await handleReddit("https://example.com", 10); - expect(result).toBeNull(); - }); - it("fetches subreddit", async () => { const result = await handleReddit("https://www.reddit.com/r/programming/", 20000); expect(result).not.toBeNull(); @@ -110,58 +80,9 @@ describe.skipIf(SKIP)("handleReddit", () => { expect(result.notes).toContain("Fetched via Reddit JSON API"); } }); - - it("includes comments in post when available", async () => { - const result = await handleReddit("https://www.reddit.com/r/programming/", 20000); - // Comments test - just verify structure if post with comments is found - if (result?.content?.includes("## Top Comments")) { - expect(result.content).toContain("### u/"); - expect(result.content).toContain("points"); - } - }); - - it("handles old.reddit.com", async () => { - const result = await handleReddit("https://old.reddit.com/r/programming/", 20000); - expect(result).not.toBeNull(); - expect(result?.method).toBe("reddit"); - expect(result?.contentType).toBe("text/markdown"); - expect(result?.content).toContain("# r/"); - expect(result?.notes).toContain("Fetched via Reddit JSON API"); - }); - - it("handles reddit.com without www", async () => { - const result = await handleReddit("https://reddit.com/r/programming/", 20000); - expect(result).not.toBeNull(); - expect(result?.method).toBe("reddit"); - }); - - it("handles URLs with query parameters", async () => { - const result = await handleReddit("https://www.reddit.com/r/programming/?sort=top", 20000); - expect(result).not.toBeNull(); - expect(result?.method).toBe("reddit"); - expect(result?.content).toContain("# r/"); - }); - - it("returns null for malformed Reddit URLs", async () => { - const result = await handleReddit("https://www.reddit.com/invalid", 20000); - // May return null or empty result - if (result !== null) { - expect(result.content).toBeDefined(); - } - }); }); describe.skipIf(SKIP)("handleStackOverflow", () => { - it("returns null for non-SO URLs", async () => { - const result = await handleStackOverflow("https://example.com", 10); - expect(result).toBeNull(); - }); - - it("returns null for SO URLs without question ID", async () => { - const result = await handleStackOverflow("https://stackoverflow.com/", 10); - expect(result).toBeNull(); - }); - it("fetches a known question", async () => { // Use a well-known question that definitely exists const result = await handleStackOverflow( @@ -180,38 +101,6 @@ describe.skipIf(SKIP)("handleStackOverflow", () => { } }); - it("includes answers", async () => { - const result = await handleStackOverflow( - "https://stackoverflow.com/questions/11227809/why-is-processing-a-sorted-array-faster", - 20000, - ); - if (result?.content?.includes("## Answers")) { - expect(result.content).toContain("### Score:"); - } - }); - - it("shows accepted answer marker when present", async () => { - const result = await handleStackOverflow( - "https://stackoverflow.com/questions/11227809/why-is-processing-a-sorted-array-faster", - 20000, - ); - // Some questions may have accepted answers - if (result?.content?.includes("(Accepted)")) { - expect(result.content).toContain("## Answers"); - } - }); - - it("handles stackoverflow.com", async () => { - const result = await handleStackOverflow( - "https://stackoverflow.com/questions/11227809/why-is-processing-a-sorted-array-faster-than-processing-an-unsorted-array", - 20000, - ); - expect(result).not.toBeNull(); - expect(result?.method).toBe("stackexchange"); - expect(result?.content).toContain("# "); - expect(result?.content).toContain("## Question"); - }); - it("handles other StackExchange sites", async () => { const result = await handleStackOverflow("https://math.stackexchange.com/questions/1000/", 20000); // API may fail, check gracefully @@ -222,38 +111,4 @@ describe.skipIf(SKIP)("handleStackOverflow", () => { expect(result.notes).toContain("Fetched via Stack Exchange API"); } }); - - it("extracts question ID from URL", async () => { - const result = await handleStackOverflow( - "https://stackoverflow.com/questions/1234567/some-long-question-title", - 20000, - ); - // Should attempt to fetch, may or may not exist - // Either returns valid result or null - if (result !== null) { - expect(result.method).toBe("stackoverflow"); - } - }); - - it("handles URLs without trailing slash", async () => { - const result = await handleStackOverflow("https://stackoverflow.com/questions/11227809", 20000); - // API may fail, check gracefully - if (result !== null) { - expect(result.method).toBe("stackexchange"); - } - }); - - it("includes question metadata", async () => { - const result = await handleStackOverflow( - "https://stackoverflow.com/questions/11227809/why-is-processing-a-sorted-array-faster", - 20000, - ); - // API may fail, check gracefully - if (result !== null) { - expect(result.content).toContain("**Score:"); - expect(result.content).toContain("**Answers:"); - expect(result.content).toContain("**Tags:"); - expect(result.content).toContain("**Asked by:"); - } - }); }); diff --git a/packages/coding-agent/test/tools/web-scrapers/stackexchange.test.ts b/packages/coding-agent/test/tools/web-scrapers/stackexchange.test.ts index 5da21c3f4..ae4c9a4fe 100644 --- a/packages/coding-agent/test/tools/web-scrapers/stackexchange.test.ts +++ b/packages/coding-agent/test/tools/web-scrapers/stackexchange.test.ts @@ -91,14 +91,6 @@ describe.skipIf(SKIP)("handleStackOverflow", () => { }); // Test with www. prefix - it("handles www.stackoverflow.com URLs", async () => { - const result = await handleStackOverflow( - "https://www.stackoverflow.com/questions/218384/what-is-a-nullpointerexception", - 20, - ); - expect(result).not.toBeNull(); - expect(result?.method).toBe("stackexchange"); - }); // Verify response structure it("returns complete response structure", async () => { diff --git a/packages/coding-agent/test/tools/web-search-anthropic.test.ts b/packages/coding-agent/test/tools/web-search-anthropic.test.ts deleted file mode 100644 index 8c698aa06..000000000 --- a/packages/coding-agent/test/tools/web-search-anthropic.test.ts +++ /dev/null @@ -1,131 +0,0 @@ -import { afterEach, beforeEach, describe, expect, it } from "bun:test"; -import { hookFetch } from "@oh-my-pi/pi-utils"; -import { searchAnthropic } from "../../src/web/search/providers/anthropic"; - -type CapturedRequest = { - url: string; - headers: RequestInit["headers"]; - body: Record | null; -}; - -const WEB_SEARCH_BETA = "web-search-2025-03-05"; -const ANTHROPIC_BASE_URL = "https://api.anthropic.com"; - -function makeAnthropicResponse() { - return { - id: "msg_test_123", - model: "claude-haiku-4-5", - content: [{ type: "text", text: "Test answer" }], - usage: { - input_tokens: 12, - output_tokens: 7, - server_tool_use: { web_search_requests: 1 }, - }, - }; -} - -function getHeaderCaseInsensitive(headers: RequestInit["headers"], name: string): string | undefined { - if (!headers) return undefined; - - if (headers instanceof Headers) { - return headers.get(name) ?? undefined; - } - - if (Array.isArray(headers)) { - const match = headers.find(([key]) => key.toLowerCase() === name.toLowerCase()); - return match?.[1]; - } - - for (const [key, value] of Object.entries(headers)) { - if (key.toLowerCase() === name.toLowerCase()) { - return value as string; - } - } - - return undefined; -} - -describe("searchAnthropic headers", () => { - const originalSearchApiKey = process.env.ANTHROPIC_SEARCH_API_KEY; - const originalSearchBaseUrl = process.env.ANTHROPIC_SEARCH_BASE_URL; - const originalApiKey = process.env.ANTHROPIC_API_KEY; - const originalBaseUrl = process.env.ANTHROPIC_BASE_URL; - - let capturedRequest: CapturedRequest | null = null; - - beforeEach(() => { - capturedRequest = null; - delete process.env.ANTHROPIC_API_KEY; - delete process.env.ANTHROPIC_BASE_URL; - process.env.ANTHROPIC_SEARCH_BASE_URL = ANTHROPIC_BASE_URL; - }); - - afterEach(() => { - capturedRequest = null; - - if (originalSearchApiKey === undefined) { - delete process.env.ANTHROPIC_SEARCH_API_KEY; - } else { - process.env.ANTHROPIC_SEARCH_API_KEY = originalSearchApiKey; - } - - if (originalSearchBaseUrl === undefined) { - delete process.env.ANTHROPIC_SEARCH_BASE_URL; - } else { - process.env.ANTHROPIC_SEARCH_BASE_URL = originalSearchBaseUrl; - } - - if (originalApiKey === undefined) { - delete process.env.ANTHROPIC_API_KEY; - } else { - process.env.ANTHROPIC_API_KEY = originalApiKey; - } - - if (originalBaseUrl === undefined) { - delete process.env.ANTHROPIC_BASE_URL; - } else { - process.env.ANTHROPIC_BASE_URL = originalBaseUrl; - } - }); - - function mockFetch(responseBody: unknown): Disposable { - return hookFetch((url, init) => { - capturedRequest = { - url: typeof url === "string" ? url : url.toString(), - headers: init?.headers, - body: init?.body ? JSON.parse(init.body as string) : null, - }; - - return new Response(JSON.stringify(responseBody), { - status: 200, - headers: { "Content-Type": "application/json" }, - }); - }); - } - - it("includes web-search beta header and sends API key in X-Api-Key mode", async () => { - process.env.ANTHROPIC_SEARCH_API_KEY = "sk-ant-api-test"; - using _hook = mockFetch(makeAnthropicResponse()); - - await searchAnthropic({ query: "test api key mode" }); - - expect(capturedRequest).not.toBeNull(); - expect(capturedRequest?.url).toBe(`${ANTHROPIC_BASE_URL}/v1/messages?beta=true`); - expect(getHeaderCaseInsensitive(capturedRequest?.headers, "anthropic-beta")).toContain(WEB_SEARCH_BETA); - expect(getHeaderCaseInsensitive(capturedRequest?.headers, "x-api-key")).toBe("sk-ant-api-test"); - expect(getHeaderCaseInsensitive(capturedRequest?.headers, "authorization")).toBeUndefined(); - expect(capturedRequest?.body?.tools).toEqual([{ type: "web_search_20250305", name: "web_search" }]); - }); - - it("includes web-search beta header and sends OAuth token in Authorization mode", async () => { - process.env.ANTHROPIC_SEARCH_API_KEY = "sk-ant-oat-test"; - using _hook = mockFetch(makeAnthropicResponse()); - - await searchAnthropic({ query: "test oauth mode" }); - - expect(capturedRequest).not.toBeNull(); - expect(getHeaderCaseInsensitive(capturedRequest?.headers, "anthropic-beta")).toContain(WEB_SEARCH_BETA); - expect(getHeaderCaseInsensitive(capturedRequest?.headers, "authorization")).toBe("Bearer sk-ant-oat-test"); - expect(getHeaderCaseInsensitive(capturedRequest?.headers, "x-api-key")).toBeUndefined(); - }); -}); diff --git a/packages/coding-agent/test/tools/web-search-exa.test.ts b/packages/coding-agent/test/tools/web-search-exa.test.ts index e606a4db2..330328555 100644 --- a/packages/coding-agent/test/tools/web-search-exa.test.ts +++ b/packages/coding-agent/test/tools/web-search-exa.test.ts @@ -49,11 +49,6 @@ describe("buildExaRequestBody", () => { }); }); - it("includes contents.summary with the search query", () => { - const body = buildExaRequestBody({ query: "how does React work" }); - expect(body.contents).toEqual({ summary: { query: "how does React work" } }); - }); - it("applies num_results override", () => { const body = buildExaRequestBody({ query: "q", num_results: 5 }); expect(body.numResults).toBe(5); @@ -132,21 +127,6 @@ describe("synthesizeAnswer", () => { expect(answer).toBe("**A**: Summary A\n\n**B**: Summary B"); }); - it("limits to MAX_ANSWER_SUMMARIES (3) results", () => { - const results = [ - { title: "A", url: "https://a.com", summary: "SA" }, - { title: "B", url: "https://b.com", summary: "SB" }, - { title: "C", url: "https://c.com", summary: "SC" }, - { title: "D", url: "https://d.com", summary: "SD" }, - { title: "E", url: "https://e.com", summary: "SE" }, - ]; - const answer = synthesizeAnswer(results)!; - expect(answer.split("\n\n")).toHaveLength(3); - expect(answer).toContain("**A**: SA"); - expect(answer).toContain("**C**: SC"); - expect(answer).not.toContain("**D**"); - }); - it("skips results with missing summaries but includes ones that have them", () => { const results = [ { title: "NoSummary", url: "https://no.com", summary: null }, diff --git a/packages/coding-agent/test/tools/web-search-tavily.test.ts b/packages/coding-agent/test/tools/web-search-tavily.test.ts index ea5f8eb5d..86b4981f7 100644 --- a/packages/coding-agent/test/tools/web-search-tavily.test.ts +++ b/packages/coding-agent/test/tools/web-search-tavily.test.ts @@ -1,7 +1,6 @@ import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test"; import { hookFetch } from "@oh-my-pi/pi-utils"; import { AgentStorage } from "../../src/session/agent-storage"; -import { getSearchProviderLabel, resolveProviderChain, SEARCH_PROVIDER_ORDER } from "../../src/web/search/provider"; import { searchTavily } from "../../src/web/search/providers/tavily"; import type { SearchProviderError } from "../../src/web/search/types"; @@ -15,13 +14,6 @@ describe("Tavily web search provider", () => { delete process.env.TAVILY_API_KEY; }); - it("registers tavily in the provider registry and fallback order", async () => { - expect(SEARCH_PROVIDER_ORDER).toContain("tavily"); - expect(getSearchProviderLabel("tavily")).toBe("Tavily"); - const providers = await resolveProviderChain("tavily"); - expect(providers[0]?.id).toBe("tavily"); - }); - it("maps Tavily responses into SearchResponse and forwards recency filters", async () => { let requestBody: Record | null = null; diff --git a/packages/coding-agent/test/tools/yield.test.ts b/packages/coding-agent/test/tools/yield.test.ts index 1a285cb3b..aaec516e6 100644 --- a/packages/coding-agent/test/tools/yield.test.ts +++ b/packages/coding-agent/test/tools/yield.test.ts @@ -35,18 +35,6 @@ function getSuccessDataSchema(parameters: Record): Record { - it("exposes top-level object parameters with required result union", () => { - const tool = new YieldTool(createSession()); - const schema = tool.parameters as { - type?: string; - properties?: Record; - required?: string[]; - }; - expect(schema.type).toBe("object"); - expect(Object.keys(schema.properties ?? {})).toEqual(["result"]); - expect(schema.required).toEqual(["result"]); - }); - it("accepts success payload with data", async () => { const tool = new YieldTool(createSession()); const result = await tool.execute("call-1", { result: { data: { ok: true } } } as never); diff --git a/packages/natives/CHANGELOG.md b/packages/natives/CHANGELOG.md index 9f40f13c6..2c65beb85 100644 --- a/packages/natives/CHANGELOG.md +++ b/packages/natives/CHANGELOG.md @@ -2,6 +2,15 @@ ## [Unreleased] +### Fixed + +- Fixed shell cancellation occasionally killing the harness. The `pi_shell` descendant tracker harvested every descendant's `pgid` into the kill set, so any subprocess that inherited the harness's pgid (any helper spawned via APIs that do not call `setpgid` — sibling LSP/MCP processes, etc.) dragged `harness.pgid` into the list and the follow-up `kill(-harness.pgid, SIGTERM)` terminated the harness alongside the targets. The classifier now only adopts a `pgid` when its leader is itself one of the new descendants, and `kill_process_group` refuses the harness's own process group as a last-line defense. +- Fixed macOS process-tree termination silently doing nothing. The descendant walk relied on `proc_listchildpids`, which on recent darwin kernels (25.4+) returns no entries when a process queries its own children, so `Process::descendants` came back empty and tree-kill cleanup never reached grandchildren. The walk now builds a one-shot `ppid → [pid]` map from `proc_listallpids` + `proc_pidinfo`, matching the approach already used by `find_by_path` and the Windows Toolhelp path. + +### Changed + +- Removed the 20 Hz background descendant tracker that scanned the harness's process tree for the entire lifetime of every shell command. Cancellation now does a small rescan-and-signal loop on demand (up to three waves — SIGTERM, then SIGKILL, then SIGKILL — with early exit as soon as no descendants remain). The previous tracker existed to pin process identities against PID reuse races, but `Process::from_pid` already pins identity by kernel start time / pidfd, so the constant scanning paid for nothing and added meaningful syscall load on macOS where each scan now does `proc_listallpids` + `proc_pidinfo` per pid. + ## [14.9.3] - 2026-05-10 ### Added diff --git a/packages/natives/native/index.d.ts b/packages/natives/native/index.d.ts index 2b638e80c..74ac86dc8 100644 --- a/packages/natives/native/index.d.ts +++ b/packages/natives/native/index.d.ts @@ -415,7 +415,7 @@ export declare enum Encoding { * streamed stdout/stderr output. Returns the exit code when the command * completes, or flags when cancelled or timed out. */ -export declare function executeShell(options: ShellExecuteOptions, onChunk?: ((error: Error | null, chunk: string) => void) | undefined | null): Promise +export declare function executeShell(options: ShellExecuteOptions, onChunk?: ((error: Error | null, chunk: string) => void) | undefined | null): Promise /** * Extract the before/after slices around an overlay region. @@ -833,7 +833,7 @@ export declare enum MacOSAppearance { /** * Options for starting a macOS power assertion. * - * Each boolean maps to a `caffeinate(8)` flag and a corresponding IOKit + * Each boolean maps to a `caffeinate(8)` flag and a corresponding `IOKit` * `IOPMAssertion` type. Multiple flags can be combined; when set, one * assertion is taken per flag and all are released together when the * handle is stopped or dropped. @@ -1160,18 +1160,6 @@ export interface ShellExecuteOptions { signal?: unknown } -/** Result of executing a shell command via brush-core. */ -export interface ShellExecuteResult { - /** Exit code when the command completes normally. */ - exitCode?: number - /** Whether the command was cancelled via abort. */ - cancelled: boolean - /** Whether the command timed out before completion. */ - timedOut: boolean - /** See [`ShellRunResult::minimized`]. */ - minimized?: MinimizerResult -} - /** Options for configuring a persistent shell session. */ export interface ShellOptions { /** Environment variables to apply once per session. */ diff --git a/packages/natives/test/issue-892-repro.test.ts b/packages/natives/test/issue-892-repro.test.ts index 58519d175..5c3400839 100644 --- a/packages/natives/test/issue-892-repro.test.ts +++ b/packages/natives/test/issue-892-repro.test.ts @@ -16,7 +16,7 @@ import * as path from "node:path"; const nativeDir = path.join(import.meta.dir, "..", "native"); const indexJsPath = path.join(nativeDir, "index.js"); const indexDtsPath = path.join(nativeDir, "index.d.ts"); -const packageJsonPath = path.join(import.meta.dir, "..", "package.json"); +const _packageJsonPath = path.join(import.meta.dir, "..", "package.json"); const PUBLIC_SYMBOL_RE = /^export declare (?:class|function|enum) (\w+)/gm; @@ -39,12 +39,6 @@ function esmExportsName(js: string, name: string): boolean { } describe("issue 892: pi-natives public surface", () => { - it("routes ESM package consumers to the generated ESM loader", async () => { - const packageJson = await Bun.file(packageJsonPath).json(); - expect(packageJson.type).toBe("module"); - expect(packageJson.exports["."].import).toBe("./native/index.js"); - }); - it("declares every public .d.ts symbol as an explicit ESM export", async () => { const [js, symbols] = await Promise.all([Bun.file(indexJsPath).text(), readPublicSymbols()]); expect(symbols.length).toBeGreaterThan(0); @@ -52,15 +46,4 @@ describe("issue 892: pi-natives public surface", () => { const missing = symbols.filter(name => !esmExportsName(js, name)); expect(missing).toEqual([]); }); - - it("exports ProcessStatus with the runtime shape consumers depend on", async () => { - // Mirror the failing consumer's package import (packages/utils/src/procmgr.ts, - // packages/coding-agent/src/tools/browser/attach.ts). - const mod = await import("@oh-my-pi/pi-natives"); - const processStatusValues: Record<"Running" | "Exited", string> = { - Running: mod.ProcessStatus.Running, - Exited: mod.ProcessStatus.Exited, - }; - expect(processStatusValues).toEqual({ Running: "running", Exited: "exited" }); - }); }); diff --git a/packages/stats/CHANGELOG.md b/packages/stats/CHANGELOG.md index ff0629138..980729ef0 100644 --- a/packages/stats/CHANGELOG.md +++ b/packages/stats/CHANGELOG.md @@ -2,6 +2,27 @@ ## [Unreleased] +### Added + +- Added time range selection options (1h, 24h, 7d, 30d, 90d, All) to the dashboard header and bound them to reloading statistics for the selected window +- Added a **Behavior** dashboard page that tracks user yelling (CAPS), profanity, and dramatic punctuation (`!!!` / `???`) per day, with by-model comparisons mirroring the cost page +- Added a per-model behavior table to the **Behavior** page mirroring the Models table: sortable rows of CAPS / profanity / drama hits per model with sparkline trend and an expandable per-model breakdown chart +- Added optional `range` query parameter support on stats endpoints to retrieve metrics scoped to a requested time window + +### Changed + +- Changed the Costs dashboard summary to report totals, average per day, and top model for the selected time range instead of a fixed 30-day window and removed the previous-30-day trend comparison +- Changed behavior metrics ingestion to compute yelling from user message sentence-level uppercase ratios, filtering out short uppercase fragments so the behavior data is attributed to messages more accurately +- Removed per-chart 14/30/90 day pickers on Costs and Behavior pages so every page obeys the single time-range selector in the header +- Changed dashboard and stats queries to return data from the selected time window instead of always using all-time aggregates +- Changed the default displayed range in the UI/API to last 24h +- Added support for returning all data when `range=all` is requested + +### Fixed + +- Fixed handling of unknown `range` values by falling back to the last 24h instead of returning unscoped data +- Fixed `omp stats` failing to build the client on globally-installed installs by promoting `tailwindcss` from `devDependencies` to `dependencies` (the client build runs at runtime) + ## [14.5.4] - 2026-04-28 ### Fixed @@ -11,4 +32,4 @@ ## [13.6.0] - 2026-03-03 ### Fixed -- Include subtask session files in usage stats ([#250](https://github.com/can1357/oh-my-pi/issues/250)) +- Include subtask session files in usage stats ([#250](https://github.com/can1357/oh-my-pi/issues/250)) \ No newline at end of file diff --git a/packages/stats/package.json b/packages/stats/package.json index 097c2ee57..728b947eb 100644 --- a/packages/stats/package.json +++ b/packages/stats/package.json @@ -45,14 +45,14 @@ "lucide-react": "catalog:", "react": "catalog:", "react-chartjs-2": "catalog:", - "react-dom": "catalog:" + "react-dom": "catalog:", + "tailwindcss": "catalog:" }, "devDependencies": { "@types/bun": "catalog:", "@types/react": "catalog:", "@types/react-dom": "catalog:", - "postcss": "catalog:", - "tailwindcss": "catalog:" + "postcss": "catalog:" }, "engines": { "bun": ">=1.3.7" diff --git a/packages/stats/src/aggregator.ts b/packages/stats/src/aggregator.ts index 946210275..dda0a3857 100644 --- a/packages/stats/src/aggregator.ts +++ b/packages/stats/src/aggregator.ts @@ -2,6 +2,9 @@ import * as fs from "node:fs"; import { getRecentErrors as dbGetRecentErrors, getRecentRequests as dbGetRecentRequests, + getBehaviorByModel, + getBehaviorOverall, + getBehaviorTimeSeries, getCostTimeSeries, getFileOffset, getMessageById, @@ -14,10 +17,11 @@ import { getTimeSeries, initDb, insertMessageStats, + insertUserMessageStats, setFileOffset, } from "./db"; import { getSessionEntry, listAllSessionFiles, parseSessionFile } from "./parser"; -import type { DashboardStats, MessageStats, RequestDetails } from "./types"; +import type { BehaviorDashboardStats, DashboardStats, MessageStats, RequestDetails } from "./types"; /** * Sync a single session file to the database. @@ -42,16 +46,19 @@ async function syncSessionFile(sessionFile: string): Promise { // Parse file from last offset const fromOffset = stored?.offset ?? 0; - const { stats, newOffset } = await parseSessionFile(sessionFile, fromOffset); + const { stats, userStats, newOffset } = await parseSessionFile(sessionFile, fromOffset); if (stats.length > 0) { insertMessageStats(stats); } + if (userStats.length > 0) { + insertUserMessageStats(userStats); + } // Update offset tracker setFileOffset(sessionFile, newOffset, lastModified); - return stats.length; + return stats.length + userStats.length; } /** @@ -76,20 +83,130 @@ export async function syncAllSessions(): Promise<{ processed: number; files: num return { processed: totalProcessed, files: filesProcessed }; } +const HOUR_MS = 60 * 60 * 1000; +const DAY_MS = 24 * HOUR_MS; + +type TimeRange = "1h" | "24h" | "7d" | "30d" | "90d" | "all"; + +interface TimeRangeConfig { + timeSeriesHours: number; + timeSeriesBucketMs: number; + modelSeriesDays: number; + modelPerformanceDays: number; + costSeriesDays: number; + cutoff: number | null; +} + +const DEFAULT_TIME_RANGE: TimeRange = "24h"; + +const TIME_RANGE_TO_CONFIG: Record> = { + "1h": { + timeSeriesHours: 1, + timeSeriesBucketMs: HOUR_MS, + modelSeriesDays: 1, + modelPerformanceDays: 1, + costSeriesDays: 1, + }, + "24h": { + timeSeriesHours: 24, + timeSeriesBucketMs: HOUR_MS, + modelSeriesDays: 1, + modelPerformanceDays: 1, + costSeriesDays: 1, + }, + "7d": { + timeSeriesHours: 24 * 7, + timeSeriesBucketMs: DAY_MS, + modelSeriesDays: 7, + modelPerformanceDays: 7, + costSeriesDays: 7, + }, + "30d": { + timeSeriesHours: 24 * 30, + timeSeriesBucketMs: DAY_MS, + modelSeriesDays: 30, + modelPerformanceDays: 30, + costSeriesDays: 30, + }, + "90d": { + timeSeriesHours: 24 * 90, + timeSeriesBucketMs: DAY_MS, + modelSeriesDays: 90, + modelPerformanceDays: 90, + costSeriesDays: 90, + }, + all: { + timeSeriesHours: 24 * 3650, + timeSeriesBucketMs: DAY_MS, + modelSeriesDays: 3650, + modelPerformanceDays: 3650, + costSeriesDays: 3650, + }, +}; + +function getTimeRangeConfig(range?: string | null): TimeRangeConfig { + const normalized = range?.trim().toLowerCase() ?? DEFAULT_TIME_RANGE; + const config = TIME_RANGE_TO_CONFIG[normalized as TimeRange]; + if (config) { + const cutoff = normalized === "all" ? null : Date.now() - Math.max(1, config.timeSeriesHours * 60 * 60 * 1000); + return { ...config, cutoff }; + } + + const fallbackConfig = TIME_RANGE_TO_CONFIG[DEFAULT_TIME_RANGE]; + return { + ...fallbackConfig, + cutoff: Date.now() - fallbackConfig.timeSeriesHours * 60 * 60 * 1000, + }; +} + /** * Get all dashboard stats. */ -export async function getDashboardStats(): Promise { +export async function getDashboardStats(range?: string | null): Promise { await initDb(); + const { timeSeriesHours, timeSeriesBucketMs, modelSeriesDays, modelPerformanceDays, costSeriesDays, cutoff } = + getTimeRangeConfig(range); return { - overall: getOverallStats(), - byModel: getStatsByModel(), - byFolder: getStatsByFolder(), - timeSeries: getTimeSeries(24), - modelSeries: getModelTimeSeries(14), - modelPerformanceSeries: getModelPerformanceSeries(14), - costSeries: getCostTimeSeries(90), + overall: getOverallStats(cutoff ?? undefined), + byModel: getStatsByModel(cutoff ?? undefined), + byFolder: getStatsByFolder(cutoff ?? undefined), + timeSeries: getTimeSeries(timeSeriesHours, cutoff, timeSeriesBucketMs), + modelSeries: getModelTimeSeries(modelSeriesDays, cutoff), + modelPerformanceSeries: getModelPerformanceSeries(modelPerformanceDays, cutoff), + costSeries: getCostTimeSeries(costSeriesDays, cutoff), + }; +} + +export async function getOverviewStats(range?: string | null): Promise> { + await initDb(); + const { timeSeriesHours, timeSeriesBucketMs, cutoff } = getTimeRangeConfig(range); + + return { + overall: getOverallStats(cutoff ?? undefined), + timeSeries: getTimeSeries(timeSeriesHours, cutoff, timeSeriesBucketMs), + }; +} + +export async function getModelDashboardStats( + range?: string | null, +): Promise> { + await initDb(); + const { modelSeriesDays, modelPerformanceDays, cutoff } = getTimeRangeConfig(range); + + return { + byModel: getStatsByModel(cutoff ?? undefined), + modelSeries: getModelTimeSeries(modelSeriesDays, cutoff), + modelPerformanceSeries: getModelPerformanceSeries(modelPerformanceDays, cutoff), + }; +} + +export async function getCostDashboardStats(range?: string | null): Promise> { + await initDb(); + const { costSeriesDays, cutoff } = getTimeRangeConfig(range); + + return { + costSeries: getCostTimeSeries(costSeriesDays, cutoff), }; } export async function getRecentRequests(limit?: number): Promise { @@ -128,3 +245,13 @@ export async function getTotalMessageCount(): Promise { await initDb(); return getMessageCount(); } + +export async function getBehaviorDashboardStats(range?: string | null): Promise { + await initDb(); + const { cutoff } = getTimeRangeConfig(range); + return { + overall: getBehaviorOverall(cutoff), + byModel: getBehaviorByModel(cutoff), + behaviorSeries: getBehaviorTimeSeries(cutoff), + }; +} diff --git a/packages/stats/src/client/App.tsx b/packages/stats/src/client/App.tsx index 92b2e0de6..6c99ab615 100644 --- a/packages/stats/src/client/App.tsx +++ b/packages/stats/src/client/App.tsx @@ -1,5 +1,16 @@ import { useCallback, useEffect, useState } from "react"; -import { getRecentErrors, getRecentRequests, getStats, sync } from "./api"; +import { + getBehaviorDashboardStats, + getCostDashboardStats, + getModelDashboardStats, + getOverviewStats, + getRecentErrors, + getRecentRequests, + sync, +} from "./api"; +import { BehaviorChart } from "./components/BehaviorChart"; +import { BehaviorModelsTable } from "./components/BehaviorModelsTable"; +import { BehaviorSummary } from "./components/BehaviorSummary"; import { ChartsContainer } from "./components/ChartsContainer"; import { CostChart } from "./components/CostChart"; import { CostSummary } from "./components/CostSummary"; @@ -8,64 +19,102 @@ import { ModelsTable } from "./components/ModelsTable"; import { RequestDetail } from "./components/RequestDetail"; import { RequestList } from "./components/RequestList"; import { StatsGrid } from "./components/StatsGrid"; -import type { DashboardStats, MessageStats } from "./types"; +import type { + BehaviorDashboardStats, + CostDashboardStats, + MessageStats, + ModelDashboardStats, + OverviewStats, + TimeRange, +} from "./types"; -type Tab = "overview" | "requests" | "errors" | "models" | "costs"; +type Tab = "overview" | "requests" | "errors" | "models" | "costs" | "behavior"; export default function App() { - const [stats, setStats] = useState(null); + const [overviewStats, setOverviewStats] = useState(null); + const [modelStats, setModelStats] = useState(null); + const [costStats, setCostStats] = useState(null); + const [behaviorStats, setBehaviorStats] = useState(null); const [recentRequests, setRecentRequests] = useState([]); const [recentErrors, setRecentErrors] = useState([]); const [selectedRequest, setSelectedRequest] = useState(null); const [syncing, setSyncing] = useState(false); const [activeTab, setActiveTab] = useState("overview"); + const [timeRange, setTimeRange] = useState("24h"); - const loadData = useCallback(async () => { + const loadRecentLists = useCallback(async () => { try { - const [s, r, e] = await Promise.all([getStats(), getRecentRequests(50), getRecentErrors(50)]); - setStats(s); - setRecentRequests(r); - setRecentErrors(e); + const [requests, errors] = await Promise.all([getRecentRequests(50), getRecentErrors(50)]); + setRecentRequests(requests); + setRecentErrors(errors); } catch (err) { console.error(err); } }, []); + const loadActiveTabStats = useCallback(async () => { + try { + if (activeTab === "models") { + setModelStats(await getModelDashboardStats(timeRange)); + return; + } + if (activeTab === "costs") { + setCostStats(await getCostDashboardStats(timeRange)); + return; + } + if (activeTab === "behavior") { + setBehaviorStats(await getBehaviorDashboardStats(timeRange)); + return; + } + if (activeTab === "overview") { + setOverviewStats(await getOverviewStats(timeRange)); + } + } catch (err) { + console.error(err); + } + }, [activeTab, timeRange]); + const handleSync = async () => { setSyncing(true); try { await sync(); - await loadData(); + await Promise.all([loadActiveTabStats(), loadRecentLists()]); } finally { setSyncing(false); } }; useEffect(() => { - loadData(); - const interval = setInterval(loadData, 30000); + loadRecentLists(); + const interval = setInterval(loadRecentLists, 30000); return () => clearInterval(interval); - }, [loadData]); + }, [loadRecentLists]); - if (!stats) { - return ( -
-
-
- Loading analytics... -
-
- ); - } + useEffect(() => { + loadActiveTabStats(); + const interval = setInterval(loadActiveTabStats, 30000); + return () => clearInterval(interval); + }, [loadActiveTabStats]); return (
-
+
{activeTab === "overview" && (
- + {overviewStats ? ( + + ) : ( + + )}
- - + {modelStats ? ( + <> + + + + ) : ( + + )}
)} {activeTab === "costs" && (
- - + {costStats ? ( + <> + + + + ) : ( + + )} +
+ )} + + {activeTab === "behavior" && ( +
+ {behaviorStats ? ( + <> + + + + + ) : ( + + )}
)} @@ -123,3 +207,14 @@ export default function App() {
); } + +function LoadingState({ label }: { label: string }) { + return ( +
+
+
+ {label} +
+
+ ); +} diff --git a/packages/stats/src/client/api.ts b/packages/stats/src/client/api.ts index 1c9f4821e..93aacabe4 100644 --- a/packages/stats/src/client/api.ts +++ b/packages/stats/src/client/api.ts @@ -1,13 +1,39 @@ -import type { DashboardStats, MessageStats, RequestDetails } from "./types"; +import type { + BehaviorDashboardStats, + CostDashboardStats, + DashboardStats, + MessageStats, + ModelDashboardStats, + OverviewStats, + RequestDetails, +} from "./types"; const API_BASE = "/api"; -export async function getStats(): Promise { - const res = await fetch(`${API_BASE}/stats`); +export async function getStats(range = "24h"): Promise { + const res = await fetch(`${API_BASE}/stats?range=${encodeURIComponent(range)}`); if (!res.ok) throw new Error("Failed to fetch stats"); return res.json() as Promise; } +export async function getOverviewStats(range = "24h"): Promise { + const res = await fetch(`${API_BASE}/stats/overview?range=${encodeURIComponent(range)}`); + if (!res.ok) throw new Error("Failed to fetch overview stats"); + return res.json() as Promise; +} + +export async function getModelDashboardStats(range = "24h"): Promise { + const res = await fetch(`${API_BASE}/stats/model-dashboard?range=${encodeURIComponent(range)}`); + if (!res.ok) throw new Error("Failed to fetch model stats"); + return res.json() as Promise; +} + +export async function getCostDashboardStats(range = "24h"): Promise { + const res = await fetch(`${API_BASE}/stats/costs?range=${encodeURIComponent(range)}`); + if (!res.ok) throw new Error("Failed to fetch cost stats"); + return res.json() as Promise; +} + export async function getRecentRequests(limit = 50): Promise { const res = await fetch(`${API_BASE}/stats/recent?limit=${limit}`); if (!res.ok) throw new Error("Failed to fetch recent requests"); @@ -31,3 +57,9 @@ export async function sync(): Promise { if (!res.ok) throw new Error("Failed to sync"); return res.json(); } + +export async function getBehaviorDashboardStats(range = "24h"): Promise { + const res = await fetch(`${API_BASE}/stats/behavior?range=${encodeURIComponent(range)}`); + if (!res.ok) throw new Error("Failed to fetch behavior stats"); + return res.json() as Promise; +} diff --git a/packages/stats/src/client/components/BehaviorChart.tsx b/packages/stats/src/client/components/BehaviorChart.tsx new file mode 100644 index 000000000..7de80a42d --- /dev/null +++ b/packages/stats/src/client/components/BehaviorChart.tsx @@ -0,0 +1,367 @@ +import { + BarElement, + CategoryScale, + Chart as ChartJS, + type ChartOptions, + Filler, + Legend, + LinearScale, + LineElement, + PointElement, + Title, + Tooltip, +} from "chart.js"; +import { format } from "date-fns"; +import { useMemo, useState } from "react"; +import { Bar, Line } from "react-chartjs-2"; +import type { BehaviorTimeSeriesPoint } from "../types"; +import { useSystemTheme } from "../useSystemTheme"; + +ChartJS.register(CategoryScale, LinearScale, BarElement, LineElement, PointElement, Title, Tooltip, Legend, Filler); + +const MODEL_COLORS = [ + "#a78bfa", // violet + "#22d3ee", // cyan + "#ec4899", // pink + "#4ade80", // green + "#fbbf24", // amber + "#f87171", // red + "#60a5fa", // blue +]; + +const CHART_THEMES = { + dark: { + legendLabel: "#94a3b8", + tooltipBackground: "#16161e", + tooltipTitle: "#f8fafc", + tooltipBody: "#94a3b8", + tooltipBorder: "rgba(255, 255, 255, 0.1)", + grid: "rgba(255, 255, 255, 0.06)", + tick: "#64748b", + }, + light: { + legendLabel: "#475569", + tooltipBackground: "#ffffff", + tooltipTitle: "#0f172a", + tooltipBody: "#334155", + tooltipBorder: "rgba(15, 23, 42, 0.18)", + grid: "rgba(15, 23, 42, 0.08)", + tick: "#64748b", + }, +} as const; + +const METRIC_OPTIONS = [ + { value: "yellingSentences", label: "Yelling" }, + { value: "profanity", label: "Profanity" }, + { value: "dramaRuns", label: "Drama (!!! / ???)" }, + { value: "total", label: "All three combined" }, +] as const; +type Metric = (typeof METRIC_OPTIONS)[number]["value"]; + +function formatRateAxis(value: number): string { + if (!Number.isFinite(value)) return "-"; + if (value === 0) return "0%"; + if (Math.abs(value) < 1) return `${value.toFixed(1)}%`; + return `${value.toFixed(0)}%`; +} + +interface BehaviorChartProps { + behaviorSeries: BehaviorTimeSeriesPoint[]; +} + +function pointHits(point: BehaviorTimeSeriesPoint, metric: Metric): number { + if (metric === "total") return point.yellingSentences + point.profanity + point.dramaRuns; + return point[metric]; +} + +/** Hits per 100 user messages, 0 when there were no messages. */ +function ratePercent(hits: number, messages: number): number { + if (messages <= 0) return 0; + return (hits / messages) * 100; +} + +interface ChartSeries { + labels: string[]; + datasets: Array<{ label: string; data: number[] }>; +} + +interface DailyBucket { + hits: number; + messages: number; +} + +function buildAggregateSeries(points: BehaviorTimeSeriesPoint[], metric: Metric): ChartSeries { + if (points.length === 0) return { labels: [], datasets: [] }; + + const byDay = new Map(); + for (const point of points) { + const bucket = byDay.get(point.timestamp) ?? { hits: 0, messages: 0 }; + bucket.hits += pointHits(point, metric); + bucket.messages += point.messages; + byDay.set(point.timestamp, bucket); + } + + const sorted = [...byDay.entries()].sort((a, b) => a[0] - b[0]); + return { + labels: sorted.map(([ts]) => format(new Date(ts), "MMM d")), + datasets: [ + { + label: METRIC_OPTIONS.find(m => m.value === metric)?.label ?? "Hits", + data: sorted.map(([, b]) => ratePercent(b.hits, b.messages)), + }, + ], + }; +} + +function buildByModelSeries(points: BehaviorTimeSeriesPoint[], metric: Metric, topN = 5): ChartSeries { + if (points.length === 0) return { labels: [], datasets: [] }; + + // Rank by message volume so the models you actually use surface first, + // matching the Behavior-by-Model table. + const totals = new Map(); + for (const point of points) { + const key = `${point.model}::${point.provider}`; + const existing = totals.get(key); + if (existing) { + existing.messages += point.messages; + } else { + totals.set(key, { model: point.model, provider: point.provider, messages: point.messages }); + } + } + + const sorted = [...totals.entries()].sort((a, b) => b[1].messages - a[1].messages); + const topEntries = sorted.slice(0, topN); + const topKeys = new Set(topEntries.map(([key]) => key)); + + const modelCount = new Map(); + for (const [, { model }] of topEntries) { + modelCount.set(model, (modelCount.get(model) ?? 0) + 1); + } + const labelByKey = new Map(); + for (const [key, { model, provider }] of topEntries) { + labelByKey.set(key, (modelCount.get(model) ?? 0) > 1 ? `${model} (${provider})` : model); + } + + const allDays = [...new Set(points.map(p => p.timestamp))].sort((a, b) => a - b); + const seriesNames = topEntries.map(([key]) => labelByKey.get(key) ?? key); + const hasOther = points.some(p => !topKeys.has(`${p.model}::${p.provider}`)); + if (hasOther) seriesNames.push("Other"); + + // Track hits and messages separately per (day, series), then convert to a + // rate at the end. Summing rates would weight low-volume days unfairly. + const dayMap = new Map>(); + for (const day of allDays) dayMap.set(day, {}); + for (const point of points) { + const key = `${point.model}::${point.provider}`; + const label = topKeys.has(key) ? (labelByKey.get(key) ?? point.model) : "Other"; + const row = dayMap.get(point.timestamp); + if (!row) continue; + const bucket = row[label] ?? { hits: 0, messages: 0 }; + bucket.hits += pointHits(point, metric); + bucket.messages += point.messages; + row[label] = bucket; + } + + return { + labels: allDays.map(ts => format(new Date(ts), "MMM d")), + datasets: seriesNames.map(name => ({ + label: name, + data: allDays.map(day => { + const bucket = dayMap.get(day)?.[name]; + return bucket ? ratePercent(bucket.hits, bucket.messages) : 0; + }), + })), + }; +} + +export function BehaviorChart({ behaviorSeries }: BehaviorChartProps) { + const [byModel, setByModel] = useState(false); + const [metric, setMetric] = useState("total"); + const theme = useSystemTheme(); + const chartTheme = CHART_THEMES[theme]; + + const chartData = useMemo( + () => (byModel ? buildByModelSeries(behaviorSeries, metric) : buildAggregateSeries(behaviorSeries, metric)), + [behaviorSeries, byModel, metric], + ); + + const sharedPlugins = { + legend: { + display: byModel, + position: "top" as const, + align: "start" as const, + labels: { + color: chartTheme.legendLabel, + usePointStyle: true, + padding: 16, + font: { size: 12 }, + boxWidth: 8, + }, + }, + tooltip: { + backgroundColor: chartTheme.tooltipBackground, + titleColor: chartTheme.tooltipTitle, + bodyColor: chartTheme.tooltipBody, + borderColor: chartTheme.tooltipBorder, + borderWidth: 1, + padding: 12, + cornerRadius: 8, + callbacks: { + label: (context: { dataset: { label?: string }; parsed: { y: number | null } }) => { + const label = context.dataset.label ?? "Hits"; + const value = context.parsed.y ?? 0; + return `${label}: ${formatRateAxis(value)}`; + }, + }, + }, + }; + + const sharedScaleBase = { + grid: { color: chartTheme.grid, drawBorder: false }, + ticks: { color: chartTheme.tick, font: { size: 11 } }, + }; + + const yScale = { + ...sharedScaleBase, + ticks: { + ...sharedScaleBase.ticks, + callback: (value: number | string) => formatRateAxis(Number(value)), + }, + min: 0, + }; + + if (byModel) { + const lineData = { + labels: chartData.labels, + datasets: chartData.datasets.map((ds, index) => ({ + label: ds.label, + data: ds.data, + borderColor: MODEL_COLORS[index % MODEL_COLORS.length], + backgroundColor: `${MODEL_COLORS[index % MODEL_COLORS.length]}20`, + fill: true, + tension: 0, + pointRadius: 3, + pointHoverRadius: 4, + borderWidth: 2, + })), + }; + + const lineOptions: ChartOptions<"line"> = { + responsive: true, + maintainAspectRatio: false, + interaction: { mode: "index", intersect: false }, + plugins: sharedPlugins, + scales: { x: sharedScaleBase, y: yScale }, + }; + + return ( + + + + ); + } + + const barData = { + labels: chartData.labels, + datasets: chartData.datasets.map((ds, index) => ({ + label: ds.label, + data: ds.data, + backgroundColor: MODEL_COLORS[index % MODEL_COLORS.length], + borderColor: MODEL_COLORS[index % MODEL_COLORS.length], + borderWidth: 0, + borderRadius: 3, + })), + }; + + const barOptions: ChartOptions<"bar"> = { + responsive: true, + maintainAspectRatio: false, + interaction: { mode: "index", intersect: false }, + plugins: sharedPlugins, + scales: { + x: { ...sharedScaleBase, stacked: true }, + y: { ...yScale, stacked: true }, + }, + layout: { padding: { top: 8 } }, + }; + + return ( + + + + ); +} + +interface ChartWrapperProps { + byModel: boolean; + metric: Metric; + onByModelChange: (v: boolean) => void; + onMetricChange: (v: Metric) => void; + empty: boolean; + children: React.ReactNode; +} + +function ChartWrapper({ byModel, metric, onByModelChange, onMetricChange, empty, children }: ChartWrapperProps) { + const metricLabel = METRIC_OPTIONS.find(m => m.value === metric)?.label ?? ""; + return ( +
+
+
+

User Tantrums

+

{metricLabel} as % of user messages per day

+
+
+
+ {METRIC_OPTIONS.map(opt => ( + + ))} +
+
+ + +
+
+
+
+ {empty ? ( +
+ No behavioral data yet. Sync to scan your sessions. +
+ ) : ( +
{children}
+ )} +
+
+ ); +} diff --git a/packages/stats/src/client/components/BehaviorModelsTable.tsx b/packages/stats/src/client/components/BehaviorModelsTable.tsx new file mode 100644 index 000000000..924b61173 --- /dev/null +++ b/packages/stats/src/client/components/BehaviorModelsTable.tsx @@ -0,0 +1,422 @@ +import { + CategoryScale, + Chart as ChartJS, + Legend, + LinearScale, + LineElement, + PointElement, + Title, + Tooltip, +} from "chart.js"; +import { format } from "date-fns"; +import { ChevronDown, ChevronUp } from "lucide-react"; +import { useMemo, useState } from "react"; +import { Line } from "react-chartjs-2"; +import type { BehaviorModelStats, BehaviorTimeSeriesPoint } from "../types"; +import { useSystemTheme } from "../useSystemTheme"; + +ChartJS.register(CategoryScale, LinearScale, PointElement, LineElement, Title, Tooltip, Legend); + +const MODEL_COLORS = [ + "#a78bfa", // violet + "#22d3ee", // cyan + "#ec4899", // pink + "#4ade80", // green + "#fbbf24", // amber + "#f87171", // red + "#60a5fa", // blue +]; + +const SERIES_COLORS = { + yelling: "#fbbf24", // amber + profanity: "#f87171", // red + drama: "#a78bfa", // violet +} as const; + +const CHART_THEMES = { + dark: { + legendLabel: "#cbd5e1", + tooltipBackground: "#16161e", + tooltipTitle: "#f8fafc", + tooltipBody: "#94a3b8", + tooltipBorder: "rgba(255, 255, 255, 0.1)", + grid: "rgba(255, 255, 255, 0.06)", + tick: "#94a3b8", + }, + light: { + legendLabel: "#334155", + tooltipBackground: "#ffffff", + tooltipTitle: "#0f172a", + tooltipBody: "#334155", + tooltipBorder: "rgba(15, 23, 42, 0.18)", + grid: "rgba(15, 23, 42, 0.08)", + tick: "#475569", + }, +} as const; + +type ChartTheme = (typeof CHART_THEMES)[keyof typeof CHART_THEMES]; + +interface BehaviorModelsTableProps { + models: BehaviorModelStats[]; + behaviorSeries: BehaviorTimeSeriesPoint[]; +} + +interface DailyPoint { + timestamp: number; + yelling: number; + profanity: number; + drama: number; + total: number; +} + +interface ModelTrendSeries { + data: DailyPoint[]; +} + +const GRID_TEMPLATE = "2fr 0.9fr 0.9fr 0.9fr 0.9fr 0.9fr 140px 40px"; + +function formatInt(value: number): string { + return value.toLocaleString(); +} + +function totalHitRate(model: BehaviorModelStats): number { + if (model.totalMessages === 0) return 0; + const hits = model.totalYellingSentences + model.totalProfanity + model.totalDramaRuns; + return hits / model.totalMessages; +} + +/** + * Rate-as-percent. < 1% shows one decimal so a 0.4% model doesn't read as 0%. + */ +function formatRate(total: number, messages: number): string { + if (messages === 0) return "-"; + const pct = (total / messages) * 100; + if (pct === 0) return "0%"; + if (pct < 1) return `${pct.toFixed(1)}%`; + return `${pct.toFixed(0)}%`; +} + +export function BehaviorModelsTable({ models, behaviorSeries }: BehaviorModelsTableProps) { + const [expandedKey, setExpandedKey] = useState(null); + const theme = useSystemTheme(); + const chartTheme = CHART_THEMES[theme]; + + const trendByKey = useMemo(() => buildTrendLookup(behaviorSeries), [behaviorSeries]); + + // Sort by usage so the models you actually rely on surface first; rates + // stay visible per row so a low-volume freak doesn't dominate. + const sortedModels = [...models].sort((a, b) => { + if (b.totalMessages !== a.totalMessages) return b.totalMessages - a.totalMessages; + return totalHitRate(b) - totalHitRate(a); + }); + + return ( +
+
+

Behavior by Model

+

+ How often each model elicited a tantrum — rates are per user message +

+
+ +
+
+
Model
+
Messages
+
CAPS %
+
Profanity %
+
Drama %
+
Hits %
+
Trend
+
+
+ +
+ {sortedModels.map((model, index) => { + const key = `${model.model}::${model.provider}`; + const trend = trendByKey.get(key)?.data ?? []; + const trendColor = MODEL_COLORS[index % MODEL_COLORS.length]; + const isExpanded = expandedKey === key; + const totalHits = model.totalYellingSentences + model.totalProfanity + model.totalDramaRuns; + + return ( +
+ + + {isExpanded && ( +
+
+
+ + + + +
+
+ {trend.length === 0 ? ( +
+ No data available +
+ ) : ( + + )} +
+
+
+ )} +
+ ); + })} + {sortedModels.length === 0 && ( +
+ No user behavior recorded for this range yet. +
+ )} +
+
+
+ ); +} + +function DetailRow({ + label, + total, + messages, + valueClass, + mode = "rate", +}: { + label: string; + total: number; + messages: number; + valueClass: string; + mode?: "rate" | "average"; +}) { + const perMsgLabel = mode === "rate" ? "% of msgs" : "Per msg"; + const perMsgValue = + messages > 0 ? (mode === "rate" ? formatRate(total, messages) : (total / messages).toFixed(0)) : "-"; + return ( +
+
{label}
+
+
+ Total + {formatInt(total)} +
+
+ {perMsgLabel} + {perMsgValue} +
+
+
+ ); +} + +function TrendSparkline({ data, color }: { data: DailyPoint[]; color: string }) { + const chartData = { + labels: data.map(d => format(new Date(d.timestamp), "MMM d")), + datasets: [ + { + data: data.map(d => d.total), + borderColor: color, + backgroundColor: "transparent", + tension: 0.4, + pointRadius: 0, + borderWidth: 2, + }, + ], + }; + + const options = { + responsive: true, + maintainAspectRatio: false, + plugins: { legend: { display: false }, tooltip: { enabled: false } }, + scales: { + x: { display: false }, + y: { display: false, min: 0 }, + }, + }; + + return ; +} + +function BreakdownChart({ data, chartTheme }: { data: DailyPoint[]; chartTheme: ChartTheme }) { + const chartData = { + labels: data.map(d => format(new Date(d.timestamp), "MMM d")), + datasets: [ + { + label: "CAPS", + data: data.map(d => d.yelling), + borderColor: SERIES_COLORS.yelling, + backgroundColor: "transparent", + tension: 0.4, + pointRadius: 0, + borderWidth: 2, + }, + { + label: "Profanity", + data: data.map(d => d.profanity), + borderColor: SERIES_COLORS.profanity, + backgroundColor: "transparent", + tension: 0.4, + pointRadius: 0, + borderWidth: 2, + }, + { + label: "Drama", + data: data.map(d => d.drama), + borderColor: SERIES_COLORS.drama, + backgroundColor: "transparent", + tension: 0.4, + pointRadius: 0, + borderWidth: 2, + }, + ], + }; + + const options = { + responsive: true, + maintainAspectRatio: false, + plugins: { + legend: { + display: true, + position: "top" as const, + labels: { + color: chartTheme.legendLabel, + usePointStyle: true, + padding: 16, + font: { size: 12 }, + }, + }, + tooltip: { + backgroundColor: chartTheme.tooltipBackground, + titleColor: chartTheme.tooltipTitle, + bodyColor: chartTheme.tooltipBody, + borderColor: chartTheme.tooltipBorder, + borderWidth: 1, + cornerRadius: 8, + }, + }, + scales: { + x: { + grid: { color: chartTheme.grid }, + ticks: { color: chartTheme.tick, font: { size: 11 } }, + }, + y: { + grid: { color: chartTheme.grid }, + ticks: { color: chartTheme.tick, font: { size: 11 } }, + min: 0, + }, + }, + }; + + return ; +} + +/** + * Group the daily time-series by model+provider, producing one continuous + * day-bucket array per model so the sparkline / breakdown chart can render + * without missing-day artifacts. + */ +function buildTrendLookup(points: BehaviorTimeSeriesPoint[]): Map { + if (points.length === 0) return new Map(); + + const allDays = [...new Set(points.map(p => p.timestamp))].sort((a, b) => a - b); + const byKey = new Map>(); + + for (const point of points) { + const key = `${point.model}::${point.provider}`; + let dayMap = byKey.get(key); + if (!dayMap) { + dayMap = new Map(); + byKey.set(key, dayMap); + } + const existing = dayMap.get(point.timestamp) ?? { + timestamp: point.timestamp, + yelling: 0, + profanity: 0, + drama: 0, + total: 0, + }; + existing.yelling += point.yellingSentences; + existing.profanity += point.profanity; + existing.drama += point.dramaRuns; + existing.total = existing.yelling + existing.profanity + existing.drama; + dayMap.set(point.timestamp, existing); + } + + const out = new Map(); + for (const [key, dayMap] of byKey) { + const data = allDays.map( + ts => + dayMap.get(ts) ?? { + timestamp: ts, + yelling: 0, + profanity: 0, + drama: 0, + total: 0, + }, + ); + out.set(key, { data }); + } + return out; +} diff --git a/packages/stats/src/client/components/BehaviorSummary.tsx b/packages/stats/src/client/components/BehaviorSummary.tsx new file mode 100644 index 000000000..f1ac624a8 --- /dev/null +++ b/packages/stats/src/client/components/BehaviorSummary.tsx @@ -0,0 +1,75 @@ +import { useMemo } from "react"; +import type { BehaviorOverallStats, BehaviorTimeSeriesPoint } from "../types"; + +interface BehaviorSummaryProps { + overall: BehaviorOverallStats; + behaviorSeries: BehaviorTimeSeriesPoint[]; +} + +function formatInt(value: number): string { + return value.toLocaleString(); +} + +export function BehaviorSummary({ overall, behaviorSeries }: BehaviorSummaryProps) { + // Top "ranted-at" model: model that absorbed the most caps + profanity + drama. + const topModel = useMemo(() => { + const totals = new Map(); + for (const point of behaviorSeries) { + const key = `${point.model}::${point.provider}`; + const existing = totals.get(key); + const score = point.yellingSentences + point.profanity + point.dramaRuns; + if (existing) { + existing.score += score; + } else { + totals.set(key, { model: point.model, provider: point.provider, score }); + } + } + let best: { model: string; provider: string; score: number } | null = null; + for (const entry of totals.values()) { + if (!best || entry.score > best.score) best = entry; + } + return best; + }, [behaviorSeries]); + + const capsPerMsg = overall.totalMessages > 0 ? overall.totalYellingSentences / overall.totalMessages : 0; + + const cards: Array<{ label: string; value: string; sub?: string }> = [ + { + label: "Messages", + value: formatInt(overall.totalMessages), + }, + { + label: "Yelling", + value: formatInt(overall.totalYellingSentences), + sub: overall.totalMessages > 0 ? `${capsPerMsg.toFixed(2)} / msg` : undefined, + }, + { + label: "Profanity hits", + value: formatInt(overall.totalProfanity), + }, + { + label: "Drama runs", + value: formatInt(overall.totalDramaRuns), + sub: "!!! / ???", + }, + { + label: "Most yelled-at", + value: topModel?.model ?? "—", + sub: topModel ? `${formatInt(topModel.score)} hits` : undefined, + }, + ]; + + return ( +
+ {cards.map(card => ( +
+

{card.label}

+

+ {card.value} +

+ {card.sub &&

{card.sub}

} +
+ ))} +
+ ); +} diff --git a/packages/stats/src/client/components/CostChart.tsx b/packages/stats/src/client/components/CostChart.tsx index 9ea1c6b4f..291c0d699 100644 --- a/packages/stats/src/client/components/CostChart.tsx +++ b/packages/stats/src/client/components/CostChart.tsx @@ -53,9 +53,6 @@ const CHART_THEMES = { }, } as const; -const RANGE_OPTIONS = [14, 30, 90] as const; -type RangeDays = (typeof RANGE_OPTIONS)[number]; - interface CostChartProps { costSeries: CostTimeSeriesPoint[]; } @@ -88,16 +85,12 @@ function makeBarLabelPlugin(color: string): Plugin<"bar"> { export function CostChart({ costSeries }: CostChartProps) { const [byModel, setByModel] = useState(false); - const [days, setDays] = useState(30); const theme = useSystemTheme(); const chartTheme = CHART_THEMES[theme]; - const cutoff = Date.now() - days * 86400000; - const filtered = useMemo(() => costSeries.filter(p => p.timestamp >= cutoff), [costSeries, cutoff]); - const chartData = useMemo( - () => (byModel ? buildByModelSeries(filtered) : buildAggregateSeries(filtered)), - [filtered, byModel], + () => (byModel ? buildByModelSeries(costSeries) : buildAggregateSeries(costSeries)), + [costSeries, byModel], ); const sharedPlugins = { @@ -175,13 +168,7 @@ export function CostChart({ costSeries }: CostChartProps) { }; return ( - + ); @@ -214,13 +201,7 @@ export function CostChart({ costSeries }: CostChartProps) { }; return ( - + ); @@ -228,14 +209,12 @@ export function CostChart({ costSeries }: CostChartProps) { interface ChartWrapperProps { byModel: boolean; - days: RangeDays; onByModelChange: (v: boolean) => void; - onDaysChange: (v: RangeDays) => void; empty: boolean; children: React.ReactNode; } -function ChartWrapper({ byModel, days, onByModelChange, onDaysChange, empty, children }: ChartWrapperProps) { +function ChartWrapper({ byModel, onByModelChange, empty, children }: ChartWrapperProps) { return (
@@ -260,18 +239,6 @@ function ChartWrapper({ byModel, days, onByModelChange, onDaysChange, empty, chi By Model
-
- {RANGE_OPTIONS.map(d => ( - - ))} -
diff --git a/packages/stats/src/client/components/CostSummary.tsx b/packages/stats/src/client/components/CostSummary.tsx index a9ace2888..43774d9da 100644 --- a/packages/stats/src/client/components/CostSummary.tsx +++ b/packages/stats/src/client/components/CostSummary.tsx @@ -1,35 +1,21 @@ -import { useMemo } from "react"; import type { CostTimeSeriesPoint } from "../types"; interface CostSummaryProps { costSeries: CostTimeSeriesPoint[]; } -const SUMMARY_DAYS = 30; - function formatCost(value: number): string { return `$${Math.round(value)}`; } export function CostSummary({ costSeries }: CostSummaryProps) { - const cutoff = Date.now() - SUMMARY_DAYS * 86400000; - const prevCutoff = cutoff - SUMMARY_DAYS * 86400000; - - const current = useMemo(() => costSeries.filter(p => p.timestamp >= cutoff), [costSeries, cutoff]); - const previous = useMemo( - () => costSeries.filter(p => p.timestamp >= prevCutoff && p.timestamp < cutoff), - [costSeries, prevCutoff, cutoff], - ); - - const totalCost = current.reduce((sum, p) => sum + p.cost, 0); - const prevTotalCost = previous.reduce((sum, p) => sum + p.cost, 0); - - const dayBuckets = new Set(current.map(p => p.timestamp)).size; + const totalCost = costSeries.reduce((sum, p) => sum + p.cost, 0); + const dayBuckets = new Set(costSeries.map(p => p.timestamp)).size; const avgDaily = dayBuckets > 0 ? totalCost / dayBuckets : 0; - // Most expensive model over current period + // Most expensive model over the visible window const modelTotals = new Map(); - for (const point of current) { + for (const point of costSeries) { modelTotals.set(point.model, (modelTotals.get(point.model) ?? 0) + point.cost); } let topModel = ""; @@ -41,47 +27,22 @@ export function CostSummary({ costSeries }: CostSummaryProps) { } } - const trend = prevTotalCost > 0 ? ((totalCost - prevTotalCost) / prevTotalCost) * 100 : null; - const cards = [ - { - label: "Total (30d)", - value: formatCost(totalCost), - positive: null as boolean | null, - }, - { - label: "Avg / day", - value: formatCost(avgDaily), - positive: null as boolean | null, - }, + { label: "Total", value: formatCost(totalCost) }, + { label: "Avg / day", value: formatCost(avgDaily) }, { label: "Top model", value: topModel || "—", sub: topModel ? formatCost(topModelCost) : undefined, - positive: null as boolean | null, - }, - { - label: "vs prev 30d", - value: trend !== null ? `${trend >= 0 ? "+" : ""}${Math.round(trend)}%` : "—", - sub: undefined as string | undefined, - positive: trend !== null ? trend <= 0 : null, }, ]; return ( -
+
{cards.map(card => (

{card.label}

-

+

{card.value}

{card.sub &&

{card.sub}

} diff --git a/packages/stats/src/client/components/Header.tsx b/packages/stats/src/client/components/Header.tsx index b2d0deb53..162b50db8 100644 --- a/packages/stats/src/client/components/Header.tsx +++ b/packages/stats/src/client/components/Header.tsx @@ -1,17 +1,28 @@ import { Activity, RefreshCw } from "lucide-react"; +import type { TimeRange } from "../types"; -type Tab = "overview" | "requests" | "errors" | "models" | "costs"; +type Tab = "overview" | "requests" | "errors" | "models" | "costs" | "behavior"; + +const tabs: Tab[] = ["overview", "requests", "errors", "models", "costs", "behavior"]; +const timeRanges: { label: string; value: TimeRange }[] = [ + { label: "1h", value: "1h" }, + { label: "24h", value: "24h" }, + { label: "7d", value: "7d" }, + { label: "30d", value: "30d" }, + { label: "90d", value: "90d" }, + { label: "All", value: "all" }, +]; interface HeaderProps { activeTab: Tab; onTabChange: (tab: Tab) => void; onSync: () => void; syncing: boolean; + timeRange: TimeRange; + onTimeRangeChange: (timeRange: TimeRange) => void; } -const tabs: Tab[] = ["overview", "requests", "errors", "models", "costs"]; - -export function Header({ activeTab, onTabChange, onSync, syncing }: HeaderProps) { +export function Header({ activeTab, onTabChange, onSync, syncing, timeRange, onTimeRangeChange }: HeaderProps) { return (
@@ -37,6 +48,19 @@ export function Header({ activeTab, onTabChange, onSync, syncing }: HeaderProps) ))}
+
+ {timeRanges.map(range => ( + + ))} +