Merge upstream/main into feat/profiles-and-alias
This commit is contained in:
Generated
+8
-8
@@ -2330,7 +2330,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-ast"
|
||||
version = "15.10.9"
|
||||
version = "15.10.10"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"ast-grep-core",
|
||||
@@ -2398,7 +2398,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-iso"
|
||||
version = "15.10.9"
|
||||
version = "15.10.10"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"libc",
|
||||
@@ -2410,7 +2410,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-natives"
|
||||
version = "15.10.9"
|
||||
version = "15.10.10"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"arboard",
|
||||
@@ -2456,7 +2456,7 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "pi-shell"
|
||||
version = "15.10.9"
|
||||
version = "15.10.10"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"brush-builtins",
|
||||
@@ -4946,18 +4946,18 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy"
|
||||
version = "0.8.50"
|
||||
version = "0.8.52"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3b065d4f0e55f82fae73202e189638116a87c55ab6b8e6c2721e13dd9d854ad1"
|
||||
checksum = "ce1022995ff5ff5d841ad7d994facc23098cd40152f2c1d11cd607c6f530653f"
|
||||
dependencies = [
|
||||
"zerocopy-derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "zerocopy-derive"
|
||||
version = "0.8.50"
|
||||
version = "0.8.52"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0b631b19d36a892ab55420c92dbc83ccd79274f25be714855d3074aa71cab639"
|
||||
checksum = "1ae7f38b72ec2a254e2b87ef277cf2cd4fb97cbebf944faa6f33354da0867930"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
|
||||
+1
-1
@@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"]
|
||||
resolver = "3"
|
||||
|
||||
[workspace.package]
|
||||
version = "15.10.9"
|
||||
version = "15.10.10"
|
||||
edition = "2024"
|
||||
license = "MIT"
|
||||
authors = ["Can Boluk"]
|
||||
|
||||
@@ -15,7 +15,7 @@
|
||||
},
|
||||
"packages/agent": {
|
||||
"name": "@oh-my-pi/pi-agent-core",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-ai": "catalog:",
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
@@ -30,7 +30,7 @@
|
||||
},
|
||||
"packages/ai": {
|
||||
"name": "@oh-my-pi/pi-ai",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"dependencies": {
|
||||
"@bufbuild/protobuf": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
@@ -44,7 +44,7 @@
|
||||
},
|
||||
"packages/coding-agent": {
|
||||
"name": "@oh-my-pi/pi-coding-agent",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"bin": {
|
||||
"omp": "src/cli.ts",
|
||||
},
|
||||
@@ -90,7 +90,7 @@
|
||||
},
|
||||
"packages/hashline": {
|
||||
"name": "@oh-my-pi/hashline",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"dependencies": {
|
||||
"diff": "catalog:",
|
||||
"lru-cache": "catalog:",
|
||||
@@ -101,7 +101,7 @@
|
||||
},
|
||||
"packages/mnemopi": {
|
||||
"name": "@oh-my-pi/pi-mnemopi",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"bin": {
|
||||
"mnemopi": "src/cli.ts",
|
||||
},
|
||||
@@ -118,7 +118,7 @@
|
||||
},
|
||||
"packages/natives": {
|
||||
"name": "@oh-my-pi/pi-natives",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"devDependencies": {
|
||||
"@napi-rs/cli": "catalog:",
|
||||
"@types/bun": "catalog:",
|
||||
@@ -126,7 +126,7 @@
|
||||
},
|
||||
"packages/stats": {
|
||||
"name": "@oh-my-pi/omp-stats",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"bin": {
|
||||
"omp-stats": "./src/index.ts",
|
||||
},
|
||||
@@ -151,7 +151,7 @@
|
||||
},
|
||||
"packages/swarm-extension": {
|
||||
"name": "@oh-my-pi/swarm-extension",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"bin": {
|
||||
"omp-swarm": "src/cli.ts",
|
||||
},
|
||||
@@ -167,7 +167,7 @@
|
||||
},
|
||||
"packages/tui": {
|
||||
"name": "@oh-my-pi/pi-tui",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
"@oh-my-pi/pi-utils": "catalog:",
|
||||
@@ -208,7 +208,7 @@
|
||||
},
|
||||
"packages/utils": {
|
||||
"name": "@oh-my-pi/pi-utils",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"dependencies": {
|
||||
"@oh-my-pi/pi-natives": "catalog:",
|
||||
"beautiful-mermaid": "catalog:",
|
||||
@@ -248,15 +248,15 @@
|
||||
"@huggingface/transformers": "^4.2.0",
|
||||
"@mozilla/readability": "^0.6.0",
|
||||
"@napi-rs/cli": "3.7.0",
|
||||
"@oh-my-pi/hashline": "15.10.9",
|
||||
"@oh-my-pi/omp-stats": "15.10.9",
|
||||
"@oh-my-pi/pi-agent-core": "15.10.9",
|
||||
"@oh-my-pi/pi-ai": "15.10.9",
|
||||
"@oh-my-pi/pi-coding-agent": "15.10.9",
|
||||
"@oh-my-pi/pi-mnemopi": "15.10.9",
|
||||
"@oh-my-pi/pi-natives": "15.10.9",
|
||||
"@oh-my-pi/pi-tui": "15.10.9",
|
||||
"@oh-my-pi/pi-utils": "15.10.9",
|
||||
"@oh-my-pi/hashline": "15.10.10",
|
||||
"@oh-my-pi/omp-stats": "15.10.10",
|
||||
"@oh-my-pi/pi-agent-core": "15.10.10",
|
||||
"@oh-my-pi/pi-ai": "15.10.10",
|
||||
"@oh-my-pi/pi-coding-agent": "15.10.10",
|
||||
"@oh-my-pi/pi-mnemopi": "15.10.10",
|
||||
"@oh-my-pi/pi-natives": "15.10.10",
|
||||
"@oh-my-pi/pi-tui": "15.10.10",
|
||||
"@oh-my-pi/pi-utils": "15.10.10",
|
||||
"@opentelemetry/api": "^1.9.1",
|
||||
"@opentelemetry/context-async-hooks": "^2.7.1",
|
||||
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
|
||||
|
||||
+127
-36
@@ -14,9 +14,15 @@
|
||||
//! // JS: await native.glob({ pattern: "*.rs", path: "." })
|
||||
//! ```
|
||||
|
||||
use std::{cmp::Ordering, collections::BinaryHeap, path::Path};
|
||||
use std::{
|
||||
cmp::Ordering,
|
||||
collections::BinaryHeap,
|
||||
path::Path,
|
||||
sync::{Arc, Mutex},
|
||||
};
|
||||
|
||||
use globset::GlobSet;
|
||||
use ignore::{ParallelVisitor, ParallelVisitorBuilder, WalkState};
|
||||
use napi::{
|
||||
bindgen_prelude::*,
|
||||
threadsafe_function::{ThreadsafeFunction, ThreadsafeFunctionCallMode},
|
||||
@@ -226,57 +232,142 @@ fn filter_entries(
|
||||
Ok(matches)
|
||||
}
|
||||
|
||||
struct SortedMatchVisitor<'a> {
|
||||
glob_set: &'a GlobSet,
|
||||
config: &'a GlobConfig,
|
||||
on_match: Option<&'a ThreadsafeFunction<GlobMatch>>,
|
||||
top_matches: BinaryHeap<RankedGlobMatch>,
|
||||
shared: Arc<Mutex<Vec<GlobMatch>>>,
|
||||
error: Arc<Mutex<Option<String>>>,
|
||||
ct: &'a task::CancelToken,
|
||||
visited: usize,
|
||||
}
|
||||
|
||||
impl Drop for SortedMatchVisitor<'_> {
|
||||
fn drop(&mut self) {
|
||||
if self.top_matches.is_empty() {
|
||||
return;
|
||||
}
|
||||
let drained = std::mem::take(&mut self.top_matches);
|
||||
self
|
||||
.shared
|
||||
.lock()
|
||||
.expect("glob match collection lock poisoned")
|
||||
.extend(drained.into_iter().map(|ranked| ranked.entry));
|
||||
}
|
||||
}
|
||||
|
||||
impl ParallelVisitor for SortedMatchVisitor<'_> {
|
||||
fn visit(&mut self, entry: std::result::Result<ignore::DirEntry, ignore::Error>) -> WalkState {
|
||||
if self.visited == 0 || self.visited >= 128 {
|
||||
self.visited = 0;
|
||||
if let Err(err) = self.ct.heartbeat() {
|
||||
*self.error.lock().expect("error lock poisoned") = Some(err.to_string());
|
||||
return WalkState::Quit;
|
||||
}
|
||||
}
|
||||
self.visited += 1;
|
||||
|
||||
let Ok(entry) = entry else {
|
||||
return WalkState::Continue;
|
||||
};
|
||||
let Some(mut matched_entry) =
|
||||
fs_cache::collect_entry(&self.config.root, &entry, fs_cache::ScanDetail::Full)
|
||||
else {
|
||||
return WalkState::Continue;
|
||||
};
|
||||
if fs_cache::should_skip_path(
|
||||
Path::new(&matched_entry.path),
|
||||
self.config.mentions_node_modules,
|
||||
) {
|
||||
return WalkState::Continue;
|
||||
}
|
||||
if !self.glob_set.is_match(&matched_entry.path) {
|
||||
return WalkState::Continue;
|
||||
}
|
||||
let Some(effective_file_type) = apply_file_type_filter(&matched_entry, self.config) else {
|
||||
return WalkState::Continue;
|
||||
};
|
||||
matched_entry.file_type = effective_file_type;
|
||||
let streamable = self.on_match.map(|cb| (cb, matched_entry.clone()));
|
||||
// Admission into the per-thread heap over-approximates the global top-N,
|
||||
// so streamed partials are a superset; callers dedup and re-rank.
|
||||
if push_bounded_match(&mut self.top_matches, matched_entry, self.config.max_results)
|
||||
&& let Some((callback, payload)) = streamable
|
||||
{
|
||||
callback.call(Ok(payload), ThreadsafeFunctionCallMode::NonBlocking);
|
||||
}
|
||||
WalkState::Continue
|
||||
}
|
||||
}
|
||||
|
||||
struct SortedMatchVisitorBuilder<'a> {
|
||||
glob_set: &'a GlobSet,
|
||||
config: &'a GlobConfig,
|
||||
on_match: Option<&'a ThreadsafeFunction<GlobMatch>>,
|
||||
shared: Arc<Mutex<Vec<GlobMatch>>>,
|
||||
error: Arc<Mutex<Option<String>>>,
|
||||
ct: &'a task::CancelToken,
|
||||
}
|
||||
|
||||
impl<'a> ParallelVisitorBuilder<'a> for SortedMatchVisitorBuilder<'a> {
|
||||
fn build(&mut self) -> Box<dyn ParallelVisitor + 'a> {
|
||||
Box::new(SortedMatchVisitor {
|
||||
glob_set: self.glob_set,
|
||||
config: self.config,
|
||||
on_match: self.on_match,
|
||||
top_matches: BinaryHeap::with_capacity(self.config.max_results.min(1024)),
|
||||
shared: Arc::clone(&self.shared),
|
||||
error: Arc::clone(&self.error),
|
||||
ct: self.ct,
|
||||
visited: 0,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// Walk the tree in parallel, keeping a bounded top-`max_results` heap per
|
||||
/// worker. The union of per-thread heaps always contains the global top-N;
|
||||
/// `run_glob` re-sorts and truncates afterwards, so the final ranking is
|
||||
/// deterministic (mtime desc, path tiebreak) regardless of walk order.
|
||||
fn collect_sorted_matches_uncached(
|
||||
glob_set: &GlobSet,
|
||||
config: &GlobConfig,
|
||||
on_match: Option<&ThreadsafeFunction<GlobMatch>>,
|
||||
ct: &task::CancelToken,
|
||||
) -> Result<Vec<GlobMatch>> {
|
||||
let builder = fs_cache::build_walker(
|
||||
let mut builder = fs_cache::build_walker(
|
||||
&config.root,
|
||||
config.include_hidden,
|
||||
config.use_gitignore,
|
||||
!config.mentions_node_modules,
|
||||
false,
|
||||
);
|
||||
let mut top_matches = BinaryHeap::with_capacity(config.max_results.min(1024));
|
||||
let mut visited = 0usize;
|
||||
let workers = fs_cache::grep_workers();
|
||||
if workers > 0 {
|
||||
builder.threads(workers);
|
||||
}
|
||||
let shared = Arc::new(Mutex::new(Vec::new()));
|
||||
let error = Arc::new(Mutex::new(None));
|
||||
let mut visitor_builder = SortedMatchVisitorBuilder {
|
||||
glob_set,
|
||||
config,
|
||||
on_match,
|
||||
shared: Arc::clone(&shared),
|
||||
error: Arc::clone(&error),
|
||||
ct,
|
||||
};
|
||||
ct.heartbeat()?;
|
||||
builder.build_parallel().visit(&mut visitor_builder);
|
||||
|
||||
for entry in builder.build() {
|
||||
if visited == 0 || visited >= 128 {
|
||||
visited = 0;
|
||||
ct.heartbeat()?;
|
||||
}
|
||||
visited += 1;
|
||||
|
||||
let Ok(entry) = entry else {
|
||||
continue;
|
||||
};
|
||||
let Some(mut matched_entry) =
|
||||
fs_cache::collect_entry(&config.root, &entry, fs_cache::ScanDetail::Full)
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
if fs_cache::should_skip_path(Path::new(&matched_entry.path), config.mentions_node_modules) {
|
||||
continue;
|
||||
}
|
||||
if !glob_set.is_match(&matched_entry.path) {
|
||||
continue;
|
||||
}
|
||||
let Some(effective_file_type) = apply_file_type_filter(&matched_entry, config) else {
|
||||
continue;
|
||||
};
|
||||
matched_entry.file_type = effective_file_type;
|
||||
let streamable = on_match.map(|cb| (cb, matched_entry.clone()));
|
||||
if push_bounded_match(&mut top_matches, matched_entry, config.max_results)
|
||||
&& let Some((callback, payload)) = streamable
|
||||
{
|
||||
callback.call(Ok(payload), ThreadsafeFunctionCallMode::NonBlocking);
|
||||
}
|
||||
let walk_error = error.lock().expect("error lock poisoned").take();
|
||||
if let Some(error) = walk_error {
|
||||
return Err(Error::from_reason(error));
|
||||
}
|
||||
|
||||
let mut matches: Vec<GlobMatch> = top_matches.into_iter().map(|ranked| ranked.entry).collect();
|
||||
let mut matches =
|
||||
std::mem::take(&mut *shared.lock().expect("glob match collection lock poisoned"));
|
||||
matches.sort_by(compare_matches_by_rank);
|
||||
matches.truncate(config.max_results);
|
||||
Ok(matches)
|
||||
}
|
||||
|
||||
|
||||
+248
-115
@@ -12,7 +12,10 @@ use std::{
|
||||
fs::File,
|
||||
io::{self, Read},
|
||||
path::{Path, PathBuf},
|
||||
sync::{Arc, Mutex},
|
||||
sync::{
|
||||
Arc, Mutex,
|
||||
atomic::{AtomicU64, Ordering},
|
||||
},
|
||||
};
|
||||
|
||||
use globset::GlobSet;
|
||||
@@ -87,41 +90,45 @@ pub struct SearchOptions {
|
||||
#[napi(object)]
|
||||
pub struct GrepOptions<'env> {
|
||||
/// Regex pattern to search for.
|
||||
pub pattern: String,
|
||||
pub pattern: String,
|
||||
/// Directory or file to search.
|
||||
pub path: String,
|
||||
pub path: String,
|
||||
/// Glob filter for filenames (e.g., "*.ts").
|
||||
pub glob: Option<String>,
|
||||
pub glob: Option<String>,
|
||||
/// Filter by file type (e.g., "js", "py", "rust").
|
||||
pub r#type: Option<String>,
|
||||
pub r#type: Option<String>,
|
||||
/// Case-insensitive search.
|
||||
pub ignore_case: Option<bool>,
|
||||
pub ignore_case: Option<bool>,
|
||||
/// Enable multiline matching.
|
||||
pub multiline: Option<bool>,
|
||||
pub multiline: Option<bool>,
|
||||
/// Include hidden files (default: true).
|
||||
pub hidden: Option<bool>,
|
||||
pub hidden: Option<bool>,
|
||||
/// Respect .gitignore files (default: true).
|
||||
pub gitignore: Option<bool>,
|
||||
pub gitignore: Option<bool>,
|
||||
/// Enable shared filesystem scan cache (default: false).
|
||||
pub cache: Option<bool>,
|
||||
pub cache: Option<bool>,
|
||||
/// Maximum number of matches to return.
|
||||
pub max_count: Option<u32>,
|
||||
pub max_count: Option<u32>,
|
||||
/// Skip first N matches.
|
||||
pub offset: Option<u32>,
|
||||
pub offset: Option<u32>,
|
||||
/// Lines of context before matches.
|
||||
pub context_before: Option<u32>,
|
||||
pub context_before: Option<u32>,
|
||||
/// Lines of context after matches.
|
||||
pub context_after: Option<u32>,
|
||||
pub context_after: Option<u32>,
|
||||
/// Lines of context before/after matches (legacy).
|
||||
pub context: Option<u32>,
|
||||
pub context: Option<u32>,
|
||||
/// Truncate lines longer than this (characters).
|
||||
pub max_columns: Option<u32>,
|
||||
pub max_columns: Option<u32>,
|
||||
/// Output mode (content, filesWithMatches, or count).
|
||||
pub mode: Option<GrepOutputMode>,
|
||||
pub mode: Option<GrepOutputMode>,
|
||||
/// Maximum matches collected per file (content mode). Keeps one hot file
|
||||
/// from exhausting the global `max_count` budget before other files are
|
||||
/// reached.
|
||||
pub max_count_per_file: Option<u32>,
|
||||
/// Abort signal for cancelling the operation.
|
||||
pub signal: Option<Unknown<'env>>,
|
||||
pub signal: Option<Unknown<'env>>,
|
||||
/// Timeout in milliseconds for the operation.
|
||||
pub timeout_ms: Option<u32>,
|
||||
pub timeout_ms: Option<u32>,
|
||||
}
|
||||
|
||||
/// A context line (before or after a match).
|
||||
@@ -196,6 +203,8 @@ pub struct GrepResult {
|
||||
pub files_searched: u32,
|
||||
/// Whether the limit/offset stopped the search early.
|
||||
pub limit_reached: Option<bool>,
|
||||
/// Number of files skipped because they exceed the size limit.
|
||||
pub skipped_oversized: Option<u32>,
|
||||
}
|
||||
|
||||
enum TypeFilter {
|
||||
@@ -268,6 +277,16 @@ enum FileBytes {
|
||||
Owned(Vec<u8>),
|
||||
}
|
||||
|
||||
/// Outcome of attempting to read a file for searching.
|
||||
enum ReadFile {
|
||||
Bytes(FileBytes),
|
||||
/// File exceeds [`MAX_FILE_BYTES`]; callers count these so the skip can be
|
||||
/// surfaced instead of silently returning no matches.
|
||||
Oversized,
|
||||
/// Unreadable or not a regular file; silently skipped.
|
||||
Skipped,
|
||||
}
|
||||
|
||||
impl FileBytes {
|
||||
fn as_slice(&self) -> &[u8] {
|
||||
match self {
|
||||
@@ -503,12 +522,14 @@ fn resolve_context(
|
||||
|
||||
#[derive(Clone, Copy)]
|
||||
struct SearchParams {
|
||||
context_before: u32,
|
||||
context_after: u32,
|
||||
max_columns: Option<u32>,
|
||||
mode: OutputMode,
|
||||
max_count: Option<u64>,
|
||||
offset: u64,
|
||||
context_before: u32,
|
||||
context_after: u32,
|
||||
max_columns: Option<u32>,
|
||||
mode: OutputMode,
|
||||
max_count: Option<u64>,
|
||||
max_count_per_file: Option<u64>,
|
||||
offset: u64,
|
||||
multiline: bool,
|
||||
}
|
||||
|
||||
fn run_search(
|
||||
@@ -552,44 +573,46 @@ fn build_searcher_for_params(params: SearchParams) -> Searcher {
|
||||
} else {
|
||||
0
|
||||
},
|
||||
params.multiline,
|
||||
)
|
||||
}
|
||||
|
||||
fn build_searcher(context_before: u32, context_after: u32) -> Searcher {
|
||||
fn build_searcher(context_before: u32, context_after: u32, multiline: bool) -> Searcher {
|
||||
SearcherBuilder::new()
|
||||
.binary_detection(BinaryDetection::quit(b'\x00'))
|
||||
.line_number(true)
|
||||
.multi_line(multiline)
|
||||
.before_context(context_before as usize)
|
||||
.after_context(context_after as usize)
|
||||
.build()
|
||||
}
|
||||
|
||||
/// Read file bytes, returning `None` for oversized or non-file paths.
|
||||
fn read_file_bytes(path: &Path) -> io::Result<Option<FileBytes>> {
|
||||
/// Read file bytes, distinguishing oversized files from other skips.
|
||||
fn read_file_bytes(path: &Path) -> io::Result<ReadFile> {
|
||||
let file = match File::open(path) {
|
||||
Ok(file) => file,
|
||||
Err(err)
|
||||
if matches!(err.kind(), io::ErrorKind::NotFound | io::ErrorKind::PermissionDenied) =>
|
||||
{
|
||||
return Ok(None);
|
||||
return Ok(ReadFile::Skipped);
|
||||
},
|
||||
Err(err) => return Err(err),
|
||||
};
|
||||
let metadata = file.metadata()?;
|
||||
if !metadata.is_file() {
|
||||
return Ok(None);
|
||||
return Ok(ReadFile::Skipped);
|
||||
}
|
||||
let size = metadata.len();
|
||||
if size > MAX_FILE_BYTES {
|
||||
return Ok(None);
|
||||
return Ok(ReadFile::Oversized);
|
||||
} else if size == 0 {
|
||||
return Ok(Some(FileBytes::Owned(Vec::new())));
|
||||
return Ok(ReadFile::Bytes(FileBytes::Owned(Vec::new())));
|
||||
}
|
||||
if size <= SMALL_FILE_READ_BYTES {
|
||||
let mut buffer = Vec::with_capacity(size as usize);
|
||||
let mut handle = file;
|
||||
handle.read_to_end(&mut buffer)?;
|
||||
return Ok(Some(FileBytes::Owned(buffer)));
|
||||
return Ok(ReadFile::Bytes(FileBytes::Owned(buffer)));
|
||||
}
|
||||
|
||||
let mapping = unsafe {
|
||||
@@ -608,7 +631,7 @@ fn read_file_bytes(path: &Path) -> io::Result<Option<FileBytes>> {
|
||||
FileBytes::Owned(buffer)
|
||||
};
|
||||
|
||||
Ok(Some(bytes))
|
||||
Ok(ReadFile::Bytes(bytes))
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -683,22 +706,23 @@ const fn empty_search_result(error: Option<String>) -> SearchResult {
|
||||
|
||||
/// Internal configuration for grep, extracted from options.
|
||||
struct GrepConfig {
|
||||
pattern: String,
|
||||
path: String,
|
||||
glob: Option<String>,
|
||||
type_filter: Option<String>,
|
||||
ignore_case: Option<bool>,
|
||||
multiline: Option<bool>,
|
||||
hidden: Option<bool>,
|
||||
gitignore: Option<bool>,
|
||||
cache: Option<bool>,
|
||||
max_count: Option<u32>,
|
||||
offset: Option<u32>,
|
||||
context_before: Option<u32>,
|
||||
context_after: Option<u32>,
|
||||
context: Option<u32>,
|
||||
max_columns: Option<u32>,
|
||||
mode: Option<GrepOutputMode>,
|
||||
pattern: String,
|
||||
path: String,
|
||||
glob: Option<String>,
|
||||
type_filter: Option<String>,
|
||||
ignore_case: Option<bool>,
|
||||
multiline: Option<bool>,
|
||||
hidden: Option<bool>,
|
||||
gitignore: Option<bool>,
|
||||
cache: Option<bool>,
|
||||
max_count: Option<u32>,
|
||||
offset: Option<u32>,
|
||||
context_before: Option<u32>,
|
||||
context_after: Option<u32>,
|
||||
context: Option<u32>,
|
||||
max_columns: Option<u32>,
|
||||
mode: Option<GrepOutputMode>,
|
||||
max_count_per_file: Option<u32>,
|
||||
}
|
||||
|
||||
fn collect_files(
|
||||
@@ -979,22 +1003,23 @@ mod tests {
|
||||
#[cfg(unix)]
|
||||
fn base_grep_config(path: &Path) -> GrepConfig {
|
||||
GrepConfig {
|
||||
pattern: "needle".to_string(),
|
||||
path: path.to_string_lossy().into_owned(),
|
||||
glob: None,
|
||||
type_filter: None,
|
||||
ignore_case: None,
|
||||
multiline: None,
|
||||
hidden: None,
|
||||
gitignore: Some(false),
|
||||
cache: Some(false),
|
||||
max_count: None,
|
||||
offset: None,
|
||||
context_before: None,
|
||||
context_after: None,
|
||||
context: None,
|
||||
max_columns: None,
|
||||
mode: None,
|
||||
pattern: "needle".to_string(),
|
||||
path: path.to_string_lossy().into_owned(),
|
||||
glob: None,
|
||||
type_filter: None,
|
||||
ignore_case: None,
|
||||
multiline: None,
|
||||
hidden: None,
|
||||
gitignore: Some(false),
|
||||
cache: Some(false),
|
||||
max_count: None,
|
||||
offset: None,
|
||||
context_before: None,
|
||||
context_after: None,
|
||||
context: None,
|
||||
max_columns: None,
|
||||
mode: None,
|
||||
max_count_per_file: None,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1139,6 +1164,49 @@ mod tests {
|
||||
assert_eq!(result.files_searched, 0);
|
||||
assert_eq!(result.limit_reached, None);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn grep_multiline_matches_cross_line_patterns() {
|
||||
let root = TempDirGuard::new();
|
||||
write_file(&root.path().join("code.txt"), "fn foo() {\n return 1;\n}\n");
|
||||
|
||||
let mut config = base_grep_config(root.path());
|
||||
config.pattern = r"foo\(\) \{\n return".to_string();
|
||||
config.multiline = Some(true);
|
||||
|
||||
let result = grep_sync(config, None, task::CancelToken::default())
|
||||
.expect("multiline grep should succeed");
|
||||
|
||||
assert_eq!(result.total_matches, 1, "cross-line pattern should match across lines");
|
||||
assert_eq!(result.matches.len(), 1);
|
||||
assert_eq!(result.matches[0].path, "code.txt");
|
||||
assert_eq!(result.matches[0].line_number, 1);
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn grep_per_file_max_count_preserves_file_diversity() {
|
||||
let root = TempDirGuard::new();
|
||||
write_file(&root.path().join("a.txt"), "needle 1\nneedle 2\nneedle 3\nneedle 4\nneedle 5\n");
|
||||
write_file(&root.path().join("z.txt"), "needle z\n");
|
||||
|
||||
let mut config = base_grep_config(root.path());
|
||||
config.max_count = Some(4);
|
||||
config.max_count_per_file = Some(2);
|
||||
|
||||
let result = grep_sync(config, None, task::CancelToken::default())
|
||||
.expect("directory grep should succeed");
|
||||
|
||||
let paths: Vec<&str> = result
|
||||
.matches
|
||||
.iter()
|
||||
.map(|matched| matched.path.as_str())
|
||||
.collect();
|
||||
assert_eq!(paths, ["a.txt", "a.txt", "z.txt"], "hot file must not starve later files");
|
||||
assert_eq!(result.files_with_matches, 2);
|
||||
assert_eq!(result.limit_reached, Some(true));
|
||||
}
|
||||
}
|
||||
|
||||
fn build_matcher(
|
||||
@@ -1169,9 +1237,15 @@ fn build_matcher(
|
||||
|
||||
fn per_file_params(params: SearchParams) -> SearchParams {
|
||||
let file_limit = match params.mode {
|
||||
OutputMode::Content => params
|
||||
.max_count
|
||||
.map(|max| max.saturating_add(params.offset)),
|
||||
OutputMode::Content => {
|
||||
let global = params
|
||||
.max_count
|
||||
.map(|max| max.saturating_add(params.offset));
|
||||
match (global, params.max_count_per_file) {
|
||||
(Some(global), Some(per_file)) => Some(global.min(per_file)),
|
||||
(global, per_file) => global.or(per_file),
|
||||
}
|
||||
},
|
||||
OutputMode::Count => None,
|
||||
OutputMode::FilesWithMatches => Some(1),
|
||||
};
|
||||
@@ -1182,6 +1256,7 @@ fn run_parallel_search(
|
||||
entries: &[FileEntry],
|
||||
matcher: &grep_regex::RegexMatcher,
|
||||
params: SearchParams,
|
||||
skipped_oversized: &AtomicU64,
|
||||
) -> Vec<FileSearchResult> {
|
||||
let file_params = per_file_params(params);
|
||||
let raw: Vec<Option<FileSearchResult>> = entries
|
||||
@@ -1189,7 +1264,14 @@ fn run_parallel_search(
|
||||
.map_init(
|
||||
|| build_searcher_for_params(file_params),
|
||||
|searcher, entry| {
|
||||
let bytes = read_file_bytes(&entry.path).ok()??;
|
||||
let bytes = match read_file_bytes(&entry.path).ok()? {
|
||||
ReadFile::Bytes(bytes) => bytes,
|
||||
ReadFile::Oversized => {
|
||||
skipped_oversized.fetch_add(1, Ordering::Relaxed);
|
||||
return None;
|
||||
},
|
||||
ReadFile::Skipped => return None,
|
||||
};
|
||||
let search = if file_params.mode == OutputMode::FilesWithMatches {
|
||||
let matched = matcher.is_match(bytes.as_slice()).ok()?;
|
||||
SearchResultInternal {
|
||||
@@ -1215,17 +1297,18 @@ fn run_parallel_search(
|
||||
}
|
||||
|
||||
struct StreamingGrepVisitor<'a> {
|
||||
root: &'a Path,
|
||||
matcher: &'a grep_regex::RegexMatcher,
|
||||
glob_set: Option<&'a GlobSet>,
|
||||
type_filter: Option<&'a TypeFilter>,
|
||||
params: SearchParams,
|
||||
searcher: Searcher,
|
||||
results: Vec<FileSearchResult>,
|
||||
shared_results: Arc<Mutex<Vec<Vec<FileSearchResult>>>>,
|
||||
error: Arc<Mutex<Option<String>>>,
|
||||
ct: &'a task::CancelToken,
|
||||
visited: usize,
|
||||
root: &'a Path,
|
||||
matcher: &'a grep_regex::RegexMatcher,
|
||||
glob_set: Option<&'a GlobSet>,
|
||||
type_filter: Option<&'a TypeFilter>,
|
||||
params: SearchParams,
|
||||
searcher: Searcher,
|
||||
results: Vec<FileSearchResult>,
|
||||
shared_results: Arc<Mutex<Vec<Vec<FileSearchResult>>>>,
|
||||
error: Arc<Mutex<Option<String>>>,
|
||||
skipped_oversized: Arc<AtomicU64>,
|
||||
ct: &'a task::CancelToken,
|
||||
visited: usize,
|
||||
}
|
||||
|
||||
impl Drop for StreamingGrepVisitor<'_> {
|
||||
@@ -1278,8 +1361,13 @@ impl ParallelVisitor for StreamingGrepVisitor<'_> {
|
||||
return WalkState::Continue;
|
||||
}
|
||||
|
||||
let Ok(Some(bytes)) = read_file_bytes(entry.path()) else {
|
||||
return WalkState::Continue;
|
||||
let bytes = match read_file_bytes(entry.path()) {
|
||||
Ok(ReadFile::Bytes(bytes)) => bytes,
|
||||
Ok(ReadFile::Oversized) => {
|
||||
self.skipped_oversized.fetch_add(1, Ordering::Relaxed);
|
||||
return WalkState::Continue;
|
||||
},
|
||||
Ok(ReadFile::Skipped) | Err(_) => return WalkState::Continue,
|
||||
};
|
||||
let search = if self.params.mode == OutputMode::FilesWithMatches {
|
||||
let Ok(matched) = self.matcher.is_match(bytes.as_slice()) else {
|
||||
@@ -1311,30 +1399,32 @@ impl ParallelVisitor for StreamingGrepVisitor<'_> {
|
||||
}
|
||||
|
||||
struct StreamingGrepVisitorBuilder<'a> {
|
||||
root: &'a Path,
|
||||
matcher: &'a grep_regex::RegexMatcher,
|
||||
glob_set: Option<&'a GlobSet>,
|
||||
type_filter: Option<&'a TypeFilter>,
|
||||
params: SearchParams,
|
||||
shared_results: Arc<Mutex<Vec<Vec<FileSearchResult>>>>,
|
||||
error: Arc<Mutex<Option<String>>>,
|
||||
ct: &'a task::CancelToken,
|
||||
root: &'a Path,
|
||||
matcher: &'a grep_regex::RegexMatcher,
|
||||
glob_set: Option<&'a GlobSet>,
|
||||
type_filter: Option<&'a TypeFilter>,
|
||||
params: SearchParams,
|
||||
shared_results: Arc<Mutex<Vec<Vec<FileSearchResult>>>>,
|
||||
error: Arc<Mutex<Option<String>>>,
|
||||
skipped_oversized: Arc<AtomicU64>,
|
||||
ct: &'a task::CancelToken,
|
||||
}
|
||||
|
||||
impl<'a> ParallelVisitorBuilder<'a> for StreamingGrepVisitorBuilder<'a> {
|
||||
fn build(&mut self) -> Box<dyn ParallelVisitor + 'a> {
|
||||
Box::new(StreamingGrepVisitor {
|
||||
root: self.root,
|
||||
matcher: self.matcher,
|
||||
glob_set: self.glob_set,
|
||||
type_filter: self.type_filter,
|
||||
params: self.params,
|
||||
searcher: build_searcher_for_params(self.params),
|
||||
results: Vec::new(),
|
||||
shared_results: Arc::clone(&self.shared_results),
|
||||
error: Arc::clone(&self.error),
|
||||
ct: self.ct,
|
||||
visited: 0,
|
||||
root: self.root,
|
||||
matcher: self.matcher,
|
||||
glob_set: self.glob_set,
|
||||
type_filter: self.type_filter,
|
||||
params: self.params,
|
||||
searcher: build_searcher_for_params(self.params),
|
||||
results: Vec::new(),
|
||||
shared_results: Arc::clone(&self.shared_results),
|
||||
error: Arc::clone(&self.error),
|
||||
skipped_oversized: Arc::clone(&self.skipped_oversized),
|
||||
ct: self.ct,
|
||||
visited: 0,
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -1349,7 +1439,7 @@ fn run_streaming_grep(
|
||||
use_gitignore: bool,
|
||||
skip_node_modules: bool,
|
||||
ct: &task::CancelToken,
|
||||
) -> Result<Vec<FileSearchResult>> {
|
||||
) -> Result<(Vec<FileSearchResult>, u64)> {
|
||||
let mut builder =
|
||||
fs_cache::build_walker(search_path, include_hidden, use_gitignore, skip_node_modules, false);
|
||||
let workers = fs_cache::grep_workers();
|
||||
@@ -1359,6 +1449,7 @@ fn run_streaming_grep(
|
||||
let file_params = per_file_params(params);
|
||||
let shared_results = Arc::new(Mutex::new(Vec::new()));
|
||||
let error = Arc::new(Mutex::new(None));
|
||||
let skipped_oversized = Arc::new(AtomicU64::new(0));
|
||||
let mut visitor_builder = StreamingGrepVisitorBuilder {
|
||||
root: search_path,
|
||||
matcher,
|
||||
@@ -1367,6 +1458,7 @@ fn run_streaming_grep(
|
||||
params: file_params,
|
||||
shared_results: Arc::clone(&shared_results),
|
||||
error: Arc::clone(&error),
|
||||
skipped_oversized: Arc::clone(&skipped_oversized),
|
||||
ct,
|
||||
};
|
||||
ct.heartbeat()?;
|
||||
@@ -1384,7 +1476,7 @@ fn run_streaming_grep(
|
||||
.flatten()
|
||||
.collect();
|
||||
results.sort_unstable_by(|a, b| a.relative_path.cmp(&b.relative_path));
|
||||
Ok(results)
|
||||
Ok((results, skipped_oversized.load(Ordering::Relaxed)))
|
||||
}
|
||||
|
||||
fn push_count_match(matches: &mut Vec<GrepMatch>, path: String, match_count: u64) {
|
||||
@@ -1532,8 +1624,16 @@ fn search_sync(content: &[u8], options: SearchOptions) -> SearchResult {
|
||||
let max_columns = options.max_columns;
|
||||
let max_count = options.max_count.map(u64::from);
|
||||
let offset = options.offset.unwrap_or(0) as u64;
|
||||
let params =
|
||||
SearchParams { context_before, context_after, max_columns, mode, max_count, offset };
|
||||
let params = SearchParams {
|
||||
context_before,
|
||||
context_after,
|
||||
max_columns,
|
||||
mode,
|
||||
max_count,
|
||||
max_count_per_file: None,
|
||||
offset,
|
||||
multiline,
|
||||
};
|
||||
let result = match run_search(&matcher, content, params) {
|
||||
Ok(result) => result,
|
||||
Err(err) => return empty_search_result(Some(err.to_string())),
|
||||
@@ -1582,7 +1682,9 @@ fn grep_sync(
|
||||
max_columns,
|
||||
mode: output_mode,
|
||||
max_count,
|
||||
max_count_per_file: options.max_count_per_file.map(u64::from),
|
||||
offset,
|
||||
multiline,
|
||||
};
|
||||
|
||||
if !metadata.is_file() && !metadata.is_dir() {
|
||||
@@ -1592,6 +1694,7 @@ fn grep_sync(
|
||||
files_with_matches: 0,
|
||||
files_searched: 0,
|
||||
limit_reached: None,
|
||||
skipped_oversized: None,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1605,17 +1708,32 @@ fn grep_sync(
|
||||
files_with_matches: 0,
|
||||
files_searched: 0,
|
||||
limit_reached: None,
|
||||
skipped_oversized: None,
|
||||
});
|
||||
}
|
||||
|
||||
let Ok(Some(bytes)) = read_file_bytes(&search_path) else {
|
||||
return Ok(GrepResult {
|
||||
matches: Vec::new(),
|
||||
total_matches: 0,
|
||||
files_with_matches: 0,
|
||||
files_searched: 0,
|
||||
limit_reached: None,
|
||||
});
|
||||
let bytes = match read_file_bytes(&search_path) {
|
||||
Ok(ReadFile::Bytes(bytes)) => bytes,
|
||||
Ok(ReadFile::Oversized) => {
|
||||
return Ok(GrepResult {
|
||||
matches: Vec::new(),
|
||||
total_matches: 0,
|
||||
files_with_matches: 0,
|
||||
files_searched: 0,
|
||||
limit_reached: None,
|
||||
skipped_oversized: Some(1),
|
||||
});
|
||||
},
|
||||
Ok(ReadFile::Skipped) | Err(_) => {
|
||||
return Ok(GrepResult {
|
||||
matches: Vec::new(),
|
||||
total_matches: 0,
|
||||
files_with_matches: 0,
|
||||
files_searched: 0,
|
||||
limit_reached: None,
|
||||
skipped_oversized: None,
|
||||
});
|
||||
},
|
||||
};
|
||||
|
||||
if output_mode == OutputMode::FilesWithMatches && max_count.is_none() && offset == 0 {
|
||||
@@ -1629,6 +1747,7 @@ fn grep_sync(
|
||||
files_with_matches: 0,
|
||||
files_searched: 1,
|
||||
limit_reached: None,
|
||||
skipped_oversized: None,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1647,6 +1766,7 @@ fn grep_sync(
|
||||
files_with_matches: 1,
|
||||
files_searched: 1,
|
||||
limit_reached: None,
|
||||
skipped_oversized: None,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1660,6 +1780,7 @@ fn grep_sync(
|
||||
files_with_matches: 0,
|
||||
files_searched: 1,
|
||||
limit_reached: None,
|
||||
skipped_oversized: None,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1702,6 +1823,7 @@ fn grep_sync(
|
||||
files_with_matches: 1,
|
||||
files_searched: 1,
|
||||
limit_reached: if limit_reached { Some(true) } else { None },
|
||||
skipped_oversized: None,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1739,9 +1861,12 @@ fn grep_sync(
|
||||
files_with_matches: 0,
|
||||
files_searched: 0,
|
||||
limit_reached: None,
|
||||
skipped_oversized: None,
|
||||
});
|
||||
}
|
||||
run_parallel_search(&entries, &matcher, params)
|
||||
let skipped = AtomicU64::new(0);
|
||||
let results = run_parallel_search(&entries, &matcher, params, &skipped);
|
||||
(results, skipped.load(Ordering::Relaxed))
|
||||
} else {
|
||||
run_streaming_grep(
|
||||
&search_path,
|
||||
@@ -1755,6 +1880,7 @@ fn grep_sync(
|
||||
&ct,
|
||||
)?
|
||||
};
|
||||
let (results, skipped_oversized) = results;
|
||||
let (matches, total_matches, files_with_matches, files_searched, limit_reached) =
|
||||
aggregate_parallel_results(results, params);
|
||||
|
||||
@@ -1772,6 +1898,11 @@ fn grep_sync(
|
||||
files_with_matches,
|
||||
files_searched,
|
||||
limit_reached: if limit_reached { Some(true) } else { None },
|
||||
skipped_oversized: if skipped_oversized > 0 {
|
||||
Some(crate::utils::clamp_u32(skipped_oversized))
|
||||
} else {
|
||||
None
|
||||
},
|
||||
})
|
||||
}
|
||||
|
||||
@@ -1880,6 +2011,7 @@ pub fn grep(
|
||||
context,
|
||||
max_columns,
|
||||
mode,
|
||||
max_count_per_file,
|
||||
timeout_ms,
|
||||
signal,
|
||||
} = options;
|
||||
@@ -1895,6 +2027,7 @@ pub fn grep(
|
||||
gitignore,
|
||||
cache,
|
||||
max_count,
|
||||
max_count_per_file,
|
||||
offset,
|
||||
context_before,
|
||||
context_after,
|
||||
|
||||
@@ -68,5 +68,5 @@ use napi_derive::napi;
|
||||
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
|
||||
/// `packages/natives/native/index.js` (which derives the name from
|
||||
/// `package.json#version`).
|
||||
#[napi(js_name = "__piNativesV15_10_9")]
|
||||
#[napi(js_name = "__piNativesV15_10_10")]
|
||||
pub const fn pi_natives_version_sentinel() {}
|
||||
|
||||
@@ -314,6 +314,7 @@ Extra conditional behavior:
|
||||
| `PI_TASK_MAX_OUTPUT_BYTES` | Max captured output bytes per subagent (default `500000`) |
|
||||
| `PI_TASK_MAX_OUTPUT_LINES` | Max captured output lines per subagent (default `5000`) |
|
||||
| `PI_TIMING` | If set (any non-empty value), prints a hierarchical timing-span tree to **stderr** via `logger.printTimings()`. In interactive mode the tree prints once the agent is ready (before the TUI starts); in print mode it prints after the whole prompt batch completes. Print-mode prompts are wrapped in `print:prompt:initial` / `print:prompt:next` spans so each user message shows up as its own row. `PI_TIMING=x` exits the process with code 0 right after printing in interactive mode (use to measure cold startup only). `PI_TIMING=full` lists every module-load entry instead of just the top N. |
|
||||
| `PI_DEBUG_STARTUP` | If set (any non-empty value), streams one synchronous `[startup] <phase>:start` / `:done` marker line to **stderr** as each startup phase begins/ends — including command-module imports (`cli:load:<name>`) and the native addon extraction/`dlopen` (`native:*`). Unlike `PI_TIMING` (which prints only once startup completes), the markers survive a hard hang: the last line on stderr names the phase the process is stuck in. Combine with `PI_TIMING` freely; markers and the span tree share the same phase names. |
|
||||
| `PI_PACKAGE_DIR` | Overrides package asset base dir resolution (`docs/`, `examples/`, `CHANGELOG.md`) |
|
||||
| `PI_DISABLE_LSPMUX` | If `1`, disables lspmux detection/integration and forces direct LSP server spawning |
|
||||
| `PI_RPC_EMIT_TITLE` | Boolean-like flag enabling title events in RPC mode |
|
||||
@@ -391,11 +392,9 @@ These are read as runtime signals; they are usually set by the terminal/OS rathe
|
||||
| `PI_NOTIFICATIONS` | `off` / `0` / `false` suppress desktop notifications |
|
||||
| `PI_TUI_WRITE_LOG` | If set, logs TUI writes to file |
|
||||
| `PI_HARDWARE_CURSOR` | If `1`, enables hardware cursor mode |
|
||||
| `PI_CLEAR_ON_SHRINK` | If `1`, clears empty rows when content shrinks |
|
||||
| `PI_NO_SYNC_OUTPUT` | If `1`, disables DEC 2026 synchronized-output wrappers while keeping TUI autowrap guards |
|
||||
| `PI_NO_DECCARA` | If set (truthy), disables Kitty DECCARA rectangular-SGR background fills (forces padded-string rendering) |
|
||||
| `PI_DEBUG_REDRAW` | If `1`, enables redraw debug logging |
|
||||
| `PI_TUI_DEBUG` | If `1`, enables deep TUI debug dump path |
|
||||
| `PI_FORCE_IMAGE_PROTOCOL` | Forces terminal image protocol detection (`kitty`, `iterm2`/`iterm`, `sixel`, `none`) |
|
||||
|
||||
---
|
||||
|
||||
+1
-1
@@ -125,7 +125,7 @@ Implemented in `packages/coding-agent/src/eval/js/worker-core.ts`, `packages/cod
|
||||
- Persistent worker-backed VM sessions keyed by `js:${sessionId}`
|
||||
- `reset: true` calls `resetVmContext(sessionKey)` before the cell executes; reset is destructive for all live runs on that JS session
|
||||
- Top-level `await` and bare `return` are supported by wrapping code in an async IIFE when `wrapCode()` sees `await` or `return`
|
||||
- Top-level static `import ... from ...` and dynamic `import(...)` calls are routed through `rewriteImports()`, which sends them via `__omp_import__` so the specifier resolves against the session cwd
|
||||
- Top-level static `import ... from ...` and dynamic `import(...)` calls are routed through `rewriteImports()`, which sends them via `__omp_import__` so the specifier resolves against the session cwd. Dynamic-import call sites are swapped for a guarded shim (`typeof __omp_import__ === "function" ? __omp_import__ : (s, o) => import(s, o)`) rather than the bare helper identifier: functions handed to puppeteer (`tab.evaluate`, `page.evaluate`, ...) are serialized with `Function.prototype.toString()` and re-evaluated inside the browser page, where the worker-injected helper does not exist, so the shim falls back to native dynamic import there
|
||||
- Module cache is busted for **local** imports between cells so edits to source files are picked up without restarting the runtime. `__omp_import__` deletes `require.cache[absPath]` before re-importing whenever the original specifier is a filesystem path: relative (`./x`, `../x`, `.`, `..`), POSIX-absolute (`/...`), home-prefixed (`~/...`), or Windows drive-letter (`C:\...` / `C:/...`). Bare specifiers (`react`, `lodash/x`) and URL/scheme specifiers (`node:fs`, `file://...`, `https://...`) are left in cache so package identity stays stable across cells. The cache-bust only fires when the resolved target is an absolute path — unresolved bare-package fallbacks (`resolveImportSpecifier()` returning the original specifier) skip it.
|
||||
- The prelude installs globals:
|
||||
- `display`, `print`
|
||||
|
||||
+260
-310
@@ -1,295 +1,264 @@
|
||||
# TUI core renderer — invariants & failure modes
|
||||
# TUI core renderer — the append-only contract
|
||||
|
||||
What you are dealing with before you touch the rendering engine. This is the
|
||||
companion to [`tui-runtime-internals.md`](./tui-runtime-internals.md): that doc
|
||||
maps the *flow* (input → component tree → render); this doc explains what
|
||||
**does not work, why it keeps breaking, and the invariants you must not
|
||||
maps the *flow* (input → component tree → render); this doc explains the
|
||||
**render contract, why it is shaped this way, and the invariants you must not
|
||||
violate**. Scope is the core engine only:
|
||||
|
||||
- [`packages/tui/src/tui.ts`](../packages/tui/src/tui.ts) — render planner, intent emitters, native-scrollback bookkeeping, cursor placement.
|
||||
- [`packages/tui/src/tui.ts`](../packages/tui/src/tui.ts) — frame pipeline, commit ledger, window math, emitters, cursor placement.
|
||||
- [`packages/tui/src/terminal.ts`](../packages/tui/src/terminal.ts) — `ProcessTerminal`, capability probes, private-CSI reassembly.
|
||||
- [`packages/tui/src/terminal-capabilities.ts`](../packages/tui/src/terminal-capabilities.ts) — `TERMINAL` profile, ED3 risk / sync-output / DECCARA / image detection.
|
||||
- [`packages/tui/src/terminal-capabilities.ts`](../packages/tui/src/terminal-capabilities.ts) — `TERMINAL` profile, sync-output / DECCARA / image detection.
|
||||
- [`packages/tui/src/stdin-buffer.ts`](../packages/tui/src/stdin-buffer.ts) — escape-sequence reassembly.
|
||||
- [`packages/tui/src/utils.ts`](../packages/tui/src/utils.ts) — width/slice/wrap (the width model).
|
||||
- [`packages/tui/src/kitty-graphics.ts`](../packages/tui/src/kitty-graphics.ts) + [`components/image.ts`](../packages/tui/src/components/image.ts) — inline images.
|
||||
- [`packages/tui/src/deccara.ts`](../packages/tui/src/deccara.ts) — rectangular-fill optimizer.
|
||||
|
||||
Application-layer renderers (transcript, tool calls, session tree, editor,
|
||||
widgets) are **out of scope** — they live in `packages/coding-agent`.
|
||||
widgets) are **out of scope** — they live in `packages/coding-agent`. The one
|
||||
app-layer file that is load-bearing for this contract is
|
||||
[`transcript-container.ts`](../packages/coding-agent/src/modes/components/transcript-container.ts),
|
||||
which implements the commit-boundary seam described below.
|
||||
|
||||
---
|
||||
|
||||
## 1. The one thing to understand first
|
||||
|
||||
> **The renderer cannot observe the terminal's scroll position on most hosts it
|
||||
> runs on.** Every decision about rewriting native scrollback is therefore a
|
||||
> *guess*, and the guess has two opposite failure modes that cannot both be
|
||||
> avoided by a single policy.
|
||||
> **The renderer cannot observe the terminal's scroll position** (ConPTY's
|
||||
> probe lies; POSIX has no API at all). The previous engine tried to *guess*
|
||||
> when it was safe to rewrite native scrollback, and every policy choice over
|
||||
> that unobservable variable traded one failure family for another (yank ↔
|
||||
> flash ↔ corruption ↔ invisible-until-resize — see the git history of this
|
||||
> file for the full war journal). The current engine removes the guess
|
||||
> entirely: **native scrollback is append-only.**
|
||||
|
||||
We keep our transcript on the **normal screen**. We deliberately have not moved
|
||||
the engine to the alternate screen: alt-screen would make the terminal handle
|
||||
viewport isolation, but the transcript/resume affordances would disappear with
|
||||
the alternate buffer. Keeping the normal screen means
|
||||
*we* own native scrollback, which means we must decide, per frame, whether it is
|
||||
safe to rebuild it. To rebuild history we emit xterm **ED3** (`CSI 3 J`, erase
|
||||
saved lines). Deciding when ED3 is safe requires knowing whether the user has
|
||||
scrolled up — and we usually can't:
|
||||
We keep the transcript on the **normal screen** (native scrollback, native
|
||||
selection, transcript persists after exit). The engine maintains one ledger:
|
||||
|
||||
- **ConPTY hosts** (Windows Terminal, Tabby, Hyper, VS Code, conhost): the
|
||||
pseudo-console buffer is pinned to the visible grid, so any "am I at the
|
||||
bottom?" console query answers "yes" even when the reader scrolled up. The
|
||||
probe *lies*.
|
||||
- **POSIX terminals**: there is no scroll-position API at all. The probe is
|
||||
*absent*.
|
||||
- **`committedRows` (C)** — frame rows `[0, C)` have been physically scrolled
|
||||
into terminal history. They are **immutable**: the engine never rewrites
|
||||
them, and components must never change them.
|
||||
- **`windowTopRow` (W)** — the frame row mapped to grid row 0. The visible
|
||||
window is frame rows `[W, W + height)`, repainted in place with relative
|
||||
cursor moves.
|
||||
- **commit boundary (B)** — reported by the component tree per frame
|
||||
(`NativeScrollbackLiveRegion`): `B = commitSafeEnd ?? liveRegionStart ??
|
||||
frame.length`. Rows below B may still re-layout and must not enter history.
|
||||
|
||||
So `Terminal.isNativeViewportAtBottom()` returns `true` / `false` / **`undefined`**,
|
||||
and `undefined` ("unknown") is the common case. The whole renderer is built
|
||||
around not trusting `undefined`.
|
||||
Per ordinary frame: `W = max(C, L − height)`, `C' = max(C, min(B, W))`, and the
|
||||
only bytes that ever touch history are the **chunk** `frame[C, C')` written at
|
||||
the scrollback seam. Scrollback therefore equals `frame[0..C)` — every row
|
||||
exactly once, in order, with its content at commit time. There is nothing to
|
||||
guess, nothing to defer, and nothing to reconcile: the scroll position is
|
||||
irrelevant because ordinary updates never rewrite anything a scrolled reader
|
||||
could be looking at.
|
||||
|
||||
### The two-way bind
|
||||
### What this costs (the accepted tradeoffs)
|
||||
|
||||
| If you guess… | …and you're wrong | Symptom |
|
||||
- A block that has scrolled past the window top cannot reflow in place. Blocks
|
||||
stay in the live region (below B) until they are final; a late mutation of
|
||||
committed content is ignored (the stale committed copy stays in history).
|
||||
- A component tree that reports **no seam** gets shell semantics: whatever
|
||||
scrolls off is final. Shrinking such a frame into its committed prefix
|
||||
re-anchors the window and leaves the stale copy in history (§3).
|
||||
- Inside multiplexers, a resize leaves the pane history wrapped at the old
|
||||
width (same as any shell output).
|
||||
|
||||
---
|
||||
|
||||
## 2. The frame pipeline (what you are editing)
|
||||
|
||||
`#doRender` per frame:
|
||||
|
||||
1. Compose the frame (`render(width)`), collecting `liveRegionStart` /
|
||||
`commitSafeEnd` from the root children (absolute row indices).
|
||||
2. **Audit the committed prefix** (`findCommittedPrefixResync`, skipped on
|
||||
geometry frames). Components must never re-layout rows below C, but real
|
||||
flows violate it (a TTSR rewind truncating a streamed block, an image-cap
|
||||
demotion shrinking a committed image) and the violation must not become
|
||||
content loss. The detector samples the prefix *tail* (up to 8 non-blank
|
||||
rows in the last 24, SGR-stripped): an in-place edit or restyle disturbs
|
||||
only the touched rows (≤1 mismatch ⇒ aligned ⇒ ignored — stale styling in
|
||||
history is the accepted artifact), while any insertion/deletion shifts
|
||||
every row below it including the tail (⇒ re-anchor C at the first changed
|
||||
row and recommit from there: history keeps the stale copy and gains a
|
||||
fresh one — **duplication, never loss**).
|
||||
3. Classify: **fullPaint** (first paint, `clearScrollback` session replace, or
|
||||
geometry change outside a multiplexer — all user gestures) or **update**.
|
||||
4. Window math as in §1. Two special rules:
|
||||
- **Overlays freeze commits** (`C' = C`): composited rows must never enter
|
||||
history; the hidden gap backfills via the chunk after the overlay closes.
|
||||
- **Shrink into the committed prefix** (`L ≤ C`): re-anchor
|
||||
`W = max(0, L − height)`, reset `C = min(B, W)`, keep the stale history
|
||||
above (no gesture, no erase).
|
||||
5. Extract the cursor marker (strip-first: markers never reach the terminal,
|
||||
the prefix ledger, or the audit), prepare lines (width fitting), slice the
|
||||
window, composite overlays **into the window slice only** (screen
|
||||
coordinates — an overlay never touches the frame or the ledger).
|
||||
6. Emit:
|
||||
|
||||
| Emitter | Bytes | When |
|
||||
|---|---|---|
|
||||
| **Eager** (rebuild now → emit `CSI 3 J`) | reader was scrolled up | **YANK** to top + **FLASH** on terminals that snap scroll on ED3 |
|
||||
| **Defer** (emit nothing, reconcile later) | viewport really was at the bottom | **CORRUPTION** (stale/duplicated rows) + **invisible-until-resize** |
|
||||
| `#emitFullPaint` | clears + `frame[0, C')` + window rows | gestures only. `clearScrollback` ⇒ `\x1b[2J\x1b[H\x1b[3J`; otherwise ED22 (when supported) + `\x1b[2J\x1b[H` |
|
||||
| `#emitUpdate` scroll-append | `\r\n` + new bottom rows + changed-row range | the rows leaving the screen are exactly the chunk, content untouched since painted |
|
||||
| `#emitUpdate` in-window diff | relative move + changed-row range rewrite | nothing scrolls, nothing commits (cursor-only when nothing changed) |
|
||||
| `#emitUpdate` seam rewrite | chunk rows + full window rewrite | commit advance, window re-anchor, hidden-gap backfill, mux resize |
|
||||
|
||||
Yank, flash, and buffer corruption are **the same bug wearing three masks.**
|
||||
Historically, every fix that suppressed one mask for one terminal class
|
||||
re-enabled the opposite mask for a neighbouring class, and the follow-on
|
||||
complaint landed within a day. If you "fix flashing" by making rebuilds more
|
||||
eager, you will reintroduce yank. If you "fix yank" by deferring more, you will
|
||||
reintroduce corruption / invisibility. **Do not move this lever without the
|
||||
fidelity harness (§9) green.**
|
||||
**ED3 (`CSI 3 J`) is emitted in exactly one place** — `#emitFullPaint` with
|
||||
`clearScrollback: true` — and is reached only by user gestures: session
|
||||
replace/branch/resume (`requestRender(true, { clearScrollback: true })`),
|
||||
resize outside a multiplexer, `resetDisplay()` (Ctrl+L). A gesture pins the
|
||||
user to the tail, so the snap is acceptable; multiplexers never get ED3 (it is
|
||||
a no-op there and a replay would duplicate pane history).
|
||||
|
||||
The ordinary update path never emits ED2/ED3 or an absolute cursor home —
|
||||
several terminal families snap a scrolled reader to the bottom on those.
|
||||
|
||||
### The commit-boundary seam (the load-bearing app contract)
|
||||
|
||||
`NativeScrollbackLiveRegion` (tui.ts) is how a component keeps mutable rows out
|
||||
of history:
|
||||
|
||||
- `getNativeScrollbackLiveRegionStart()` — first row that may still mutate
|
||||
(everything below it, including root chrome rendered after it, stays in the
|
||||
window).
|
||||
- `getNativeScrollbackCommitSafeEnd()` — optional deeper boundary: the
|
||||
append-only prefix of the live region (a streaming assistant message's
|
||||
settled rows). Without it, a single live block taller than the window would
|
||||
hold its head out of history until it finalizes.
|
||||
|
||||
`TranscriptContainer` implements this for the coding agent: finalized blocks
|
||||
freeze (their render is snapshotted, so their content can never drift after
|
||||
the engine may have committed it), still-mutating blocks
|
||||
(`isTranscriptBlockFinalized?.() === false`) anchor the live region, and
|
||||
`deriveLiveCommitState` derives the commit-safe end of the first live block
|
||||
from two independent signals:
|
||||
|
||||
- **append-only detection** — a block observed growing without visibly
|
||||
rewriting an interior row commits its full body; a rewrite suspends this
|
||||
for `VOLATILE_REARM_FRAMES` clean frames.
|
||||
- **stable-prefix ratchet** — rows that stayed visibly identical for a full
|
||||
`STABLE_PREFIX_COMMIT_FRAMES` window commit even while the block's tail
|
||||
keeps rewriting (a task tool's static prompt above a ticking progress
|
||||
tree). Without it, one perpetually animating row holds the whole block out
|
||||
of history, so a block taller than the window reads as cut off (head
|
||||
neither committed nor on screen) for the entire run. The ratchet tracks the
|
||||
window-minimum common prefix; a rewrite above the promoted run retreats it
|
||||
to the divergence, and rows that already committed are the engine audit's
|
||||
problem (recommit → duplication, never loss). That retreat also arms a
|
||||
permanent **rewrite floor** at the divergence: a row that mutates *after*
|
||||
surviving a full promotion window is a slow ticker (an agent row's tool/cost
|
||||
counter updating every few seconds), not settling content — without the
|
||||
floor, every quiet stretch re-promoted it and every later tick forced an
|
||||
audit recommit, spraying stale snapshots of the block into scrollback for
|
||||
the whole run. Rows at/after the floor never re-promote while the block
|
||||
lives (the floor index travels with append-shaped insertions above it);
|
||||
one-off re-layouts before any promotion never arm it, and the append-only
|
||||
path commits the full block regardless.
|
||||
|
||||
Freezing is unconditional — it is the engine's required guarantee, not a
|
||||
per-terminal optimization.
|
||||
|
||||
---
|
||||
|
||||
## 2. The render-intent planner (what you are editing)
|
||||
## 3. Invariants — MUST / NEVER
|
||||
|
||||
`#doRender` is split into a **planner** (`#planRender`) that classifies a frame
|
||||
into exactly one `RenderIntent`, and one `#emit*` method per intent that owns
|
||||
the bytes written and the state update. All state flows through a single
|
||||
`#commit` checkpoint at the end of every emitter. The intent union
|
||||
(`tui.ts`, search `type RenderIntent`):
|
||||
|
||||
| Intent | Emits | When |
|
||||
|---|---|---|
|
||||
| `noop` | cursor only | nothing visible changed |
|
||||
| `initial` | clear viewport, paint transcript, **keep** prior shell scrollback | first paint after `start()` |
|
||||
| `sessionReplace` | clear viewport **+ ED3** (outside multiplexers) | caller forced `{ clearScrollback: true }` (switch/branch/reload/resume) |
|
||||
| `historyRebuild` | clear viewport **+ ED3** (outside multiplexers) | geometry change rewrapped history, or a proven-at-tail rebuild |
|
||||
| `overlayRebuild` | rebuild viewport with overlay composite | overlay visibility changed |
|
||||
| `liveRegionPinned` | relative moves + per-row rewrite/suffix-clear + `\r\n` | foreground streaming on an ED3-risk host, commit-as-you-go |
|
||||
| `viewportRepaint` | rewrite the visible viewport in place (optional `appendFrom` tail first) | safe non-destructive repaint |
|
||||
| `deferredShrink` | padded viewport repaint, history left dirty | bottom-anchored shrink, viewport unobservable |
|
||||
| `deferredMutation` | **zero bytes**, history left dirty | row-reindexing edit while possibly scrolled |
|
||||
| `shrink` / `diff` | trailing-row clear / changed-line diff | ordinary in-place updates |
|
||||
|
||||
**ED3 (`CSI 3 J`) is emitted in exactly one place** — `#emitFullPaint` when
|
||||
`clearScrollback: true` (`\x1b[2J\x1b[H\x1b[3J`). The ordinary clear is
|
||||
**non-destructive**: `\x1b[22J` (copy-screen-to-scrollback, only when
|
||||
`TERMINAL.supportsScreenToScrollback`) then `\x1b[2J\x1b[H`, **no `3J`**. ED3 is
|
||||
reached only by `sessionReplace`/`historyRebuild`/`overlayRebuild`, and those
|
||||
suppress the scrollback clear inside multiplexers (`isMultiplexerSession()` =
|
||||
`TMUX || STY || ZELLIJ`).
|
||||
|
||||
### The predicate gates
|
||||
|
||||
Three private predicates encode the guessing policy. Do not "simplify" them —
|
||||
each branch is load-bearing:
|
||||
|
||||
- `#canReplayNativeScrollbackAtCheckpoint(atBottom)` → `atBottom === true`. A
|
||||
rebuild at a **keystroke checkpoint** (prompt submit) is allowed only with a
|
||||
*positive* at-tail proof. A prompt submit is **no longer** treated as implicit
|
||||
proof for an unobservable host.
|
||||
- `#canRebuildNativeScrollbackLive(atBottom, allowUnknown)` → `true` iff
|
||||
`atBottom === true`, **or** (`atBottom === undefined && allowUnknown &&
|
||||
platform !== "win32"`). i.e. live ED3 during streaming requires either proof
|
||||
or an explicit direct-user-input opt-in, and **never** on win32.
|
||||
- `#nativeViewportIsScrolled(atBottom, allowUnknown)` → `true` if
|
||||
`atBottom === false`, or (`undefined && win32 && !allowUnknown`). Used to
|
||||
decide deferral.
|
||||
|
||||
`allowUnknownViewportMutation` is the **direct-user-input opt-in** (autocomplete
|
||||
/ IME / a keystroke the user just typed). A keystroke pins the host viewport to
|
||||
the bottom, so it is safe to repaint live then. It is **not** set by passive
|
||||
streaming. `setEagerNativeScrollbackRebuild(true)` is the streaming opt-in; on
|
||||
ED3-risk hosts it is downgraded so it never promotes to a live ED3 clear.
|
||||
|
||||
### Deferral + checkpoint discipline
|
||||
|
||||
When the viewport is unobservable during **passive streaming**, the planner
|
||||
defers (`deferredMutation`/`deferredShrink`/`viewportRepaint`) and marks native
|
||||
scrollback dirty (`#markNativeScrollbackDirty()`). Reconciliation happens later
|
||||
at a checkpoint via `refreshNativeScrollbackIfDirty()` — and only if
|
||||
`#canReplayNativeScrollbackAtCheckpoint` proves at-tail. The streaming-defer +
|
||||
live-region-pin seam (`NativeScrollbackLiveRegion`,
|
||||
`getNativeScrollbackLiveRegionStart` / `getNativeScrollbackCommitSafeEnd`) is the
|
||||
**actively-churning** part of the engine; if you change how transient rows are
|
||||
committed, every structural-mutation branch (shrink **and** grow/offscreen-edit)
|
||||
must defer **symmetrically**, or you reopen the corruption family.
|
||||
|
||||
---
|
||||
|
||||
## 3. The five fault families
|
||||
|
||||
### YANK — viewport snapped to top — NOT fully converged
|
||||
- **Mechanism:** a live `historyRebuild` fires `CSI 3 J` while the reader is
|
||||
scrolled up; ED3-snap terminals reset the visible viewport to the top of the
|
||||
(now-erased) scrollback.
|
||||
- **Trigger to avoid:** treating an unobservable probe as "at bottom" during
|
||||
*passive* streaming, or OR-ing an eager-streaming flag into the live ED3 path.
|
||||
- **Current stance:** never emit ED3 on an unobservable host during passive
|
||||
streaming; defer and reconcile at a keystroke checkpoint. ConPTY/win32 never
|
||||
trust the probe at all.
|
||||
|
||||
### CORRUPTION — duplicated / stale rows — NOT fully converged
|
||||
- **Mechanism:** the flip side of the yank fix. A deferred/repainted frame
|
||||
leaves rows already committed to native scrollback out of sync with the live
|
||||
viewport; the scrollback↔viewport seam duplicates (e.g. a 2-row dup, a
|
||||
streaming-tail dup, or an async-expansion dup).
|
||||
- **Trigger to avoid:** repainting the viewport over scrollback that still holds
|
||||
the old copy; a frozen/deferred block whose snapshot no longer matches after
|
||||
the region above it reflowed; one mutation branch deferring while its mirror
|
||||
branch repaints.
|
||||
- **Current stance:** commit only the **stable prefix** line-count to native
|
||||
history; keep unstable rows out; reconcile drift at the checkpoint; park the
|
||||
hardware cursor at real content bottom, not padded bottom.
|
||||
|
||||
### FLASH (and invisible-until-resize) — NOT fully converged
|
||||
- **Two distinct causes, one symptom:**
|
||||
- *Flash* = eager ED3 rebuild wrapped in DEC 2026 BSU/ESU fired per streaming
|
||||
frame on a terminal that clamps scroll on ED3 (VTE/GNOME family).
|
||||
- *Invisible-until-resize* = the defer fix over-firing, so a structural frame
|
||||
emits **zero bytes** (`deferredMutation` returns nothing) until a resize
|
||||
forces a repaint.
|
||||
- **Trigger to avoid:** env-detection that misses a flashing terminal (SSH
|
||||
strips `VTE_VERSION`; some hosts set no distinguishing var); collapsing an
|
||||
`undefined` probe into a definite scrolled/at-bottom verdict.
|
||||
- **Current stance:** confine ED3 to the destructive path; auto-disable DEC 2026
|
||||
at runtime when the terminal reports it unsupported (DECRQM), with
|
||||
`PI_NO_SYNC_OUTPUT` as a manual hatch; keep autowrap discipline regardless.
|
||||
|
||||
### WIDTH — measurement crashes / fidelity — crash class dead, accuracy unproven
|
||||
- **Mechanism:** the measured column width of a line disagreed with the
|
||||
terminal's painted cells (emoji, wide graphemes, combining marks, Hangul
|
||||
jamo), and the old render loop **threw** on any mismatch — a 1-cell cosmetic
|
||||
error became a fatal whole-agent crash.
|
||||
- **Current stance:** **never throw in the render hot path — clamp.** The loop
|
||||
truncates over-wide lines with `truncateToWidth`/`sliceByColumn` and logs
|
||||
(under debug) instead of dying. Width is owned end-to-end by one native UAX#11
|
||||
engine shared by measure/slice/wrap (see §6). Accuracy across all scripts
|
||||
(e.g. RTL/combining marks) is still not proven by a green gate.
|
||||
|
||||
### PROBE — stray bytes injected as keystrokes — RESOLVED
|
||||
- **Mechanism:** a private-CSI probe reply (DA1 / kitty / mode 2031) split
|
||||
across a stdin flush; the unmatched prefix was dropped and the continuation
|
||||
bytes were forwarded as keystrokes.
|
||||
- **Current stance:** buffer-and-reassemble partial CSI responses; give each
|
||||
probe a typed sentinel owner. This is the **one cleanly-closed family** —
|
||||
because its contract is *bounded and observable* (bytes in = bytes out),
|
||||
unlike the unobservable-viewport families. See §7.
|
||||
|
||||
---
|
||||
|
||||
## 4. Invariants — MUST / NEVER
|
||||
|
||||
These are the rules the recurrence taught us. Treat them as load-bearing.
|
||||
|
||||
1. **NEVER add a new `CSI 3 J` (ED3) callsite.** ED3 must flow only through
|
||||
`#emitFullPaint({ clearScrollback: true })`, for the existing destructive
|
||||
intents (`sessionReplace`, proven/safe `historyRebuild`, `overlayRebuild`).
|
||||
Ordinary redraws use the non-destructive `\x1b[22J` + `\x1b[2J\x1b[H` clear.
|
||||
2. **NEVER trust an unobservable viewport probe (`undefined`) for *passive*
|
||||
streaming.** Only a positive at-tail proof, or a direct-user-input opt-in
|
||||
(`allowUnknownViewportMutation`), authorizes a live rebuild — and never on
|
||||
win32/ConPTY.
|
||||
3. **NEVER throw in the render hot path.** Clamp over-wide lines; a width
|
||||
mismatch is cosmetic, not fatal.
|
||||
4. **NEVER let a defer path emit a structurally-changed frame as zero bytes
|
||||
while at the bottom** — that is invisible-until-resize. `deferredMutation`/
|
||||
`deferredShrink` are only safe when the viewport is (or may be) scrolled.
|
||||
5. **Defer symmetrically.** If one structural-mutation branch (shrink) defers on
|
||||
an unobservable ED3-risk host, the mirror branch (grow / offscreen-edit) must
|
||||
too. Asymmetry reopens corruption.
|
||||
6. **Commit only the stable prefix to native history.** Transient/unsettled rows
|
||||
stay out of scrollback until a checkpoint; reconcile drift at the checkpoint.
|
||||
7. **Park the hardware cursor at real content bottom**, not the padded viewport
|
||||
bottom, or height shrinks scroll live rows into scrollback and duplicate them
|
||||
1. **NEVER add a new `CSI 3 J` (ED3) callsite.** ED3 flows only through
|
||||
`#emitFullPaint({ clearScrollback: true })`, only for gestures, never inside
|
||||
multiplexers.
|
||||
2. **NEVER rewrite a committed row.** No emitter may touch frame rows `< C`,
|
||||
and `W ≥ C` always (re-showing a committed row on the grid duplicates it
|
||||
for a scrolling reader — the historical corruption family). When a
|
||||
*component* violates immutability, the audit (§2) degrades to duplication —
|
||||
never silently skip rows, never erase history.
|
||||
3. **Commits are exactly the chunk.** Any byte shape that scrolls the screen
|
||||
must scroll *only* rows accounted for by `C' − C` — that is what makes
|
||||
scrollback provably `frame[0..C)`.
|
||||
4. **NEVER probe the viewport position or fork on platform in the update
|
||||
path.** win32 behaves like POSIX. The probe APIs are gone; do not
|
||||
reintroduce them.
|
||||
5. **Mutable content stays below the commit boundary.** App-layer renderers
|
||||
must finalize-before-commit; the engine trusts B and clamps, it does not
|
||||
verify content.
|
||||
6. **Park the hardware cursor at real content bottom**, not the padded window
|
||||
bottom, or height shrinks scroll live rows into history and duplicate them
|
||||
per resize step.
|
||||
8. **Cursor writes live *inside* the synchronized-output frame**, before ESU —
|
||||
never as a second frame after it (that teleports/blinks the caret).
|
||||
9. **Detect terminal *risk*, not terminal *brand*, and default unknown to
|
||||
risky.** Env sniffing is necessarily incomplete (see §5); never assume an
|
||||
un-enumerated host is safe.
|
||||
10. **Multiplexers (tmux/screen/zellij) get no destructive scrollback clear and
|
||||
no viewport probe.** ED3 is a no-op there and a full replay duplicates the
|
||||
transcript; repaint in place and rely on the pinned/commit-as-you-go path.
|
||||
11. **Any change to the eager/defer lever, the predicates, or the live-region
|
||||
seam must be validated by the render-stress fidelity harness (§9)** across
|
||||
`{win32, POSIX} × {unknown, scrolled, at-bottom}`, not by a single-terminal
|
||||
smoke test.
|
||||
7. **Cursor writes live inside the synchronized-output frame**, before ESU —
|
||||
never as a second frame after it.
|
||||
8. **NEVER throw in the render hot path.** Clamp over-wide lines
|
||||
(`truncateToWidth`); a width mismatch is cosmetic, not fatal.
|
||||
9. **Multiplexers get no destructive clear and no history rewrap on resize** —
|
||||
repaint the window in place; pane history keeps its old wrap.
|
||||
10. **Any change to the ledger math, the emitters, or the seam must be
|
||||
validated by the stress harness (§6)** across its full scenario matrix,
|
||||
not by a single-terminal smoke test.
|
||||
|
||||
---
|
||||
|
||||
## 5. Terminal capability detection (and why it is fragile)
|
||||
## 4. Terminal capability detection
|
||||
|
||||
`TERMINAL` (`terminal-capabilities.ts`) is resolved once at import from
|
||||
`TERMINAL_ID` plus environment sniffing. The detection helpers are pure and
|
||||
parameterized over `(env, platform)` so they are unit-testable:
|
||||
`TERMINAL_ID` plus environment sniffing; detection helpers are pure over
|
||||
`(env, platform)` and unit-testable.
|
||||
|
||||
- `detectTerminalEagerEraseScrollbackRisk(env, platform)` → is a live ED3
|
||||
rebuild unsafe here? Current policy: `false` on win32 (dedicated ConPTY
|
||||
deferral paths handle it) and when `PI_TUI_ED3_SAFE=1`; otherwise **`true`**
|
||||
for `WT_SESSION` (WT fronting WSL), SSH/tmux/screen/zellij, known
|
||||
ED3-snap/scrollback-clearing terminals (WezTerm, kitty, ghostty, alacritty,
|
||||
VTE, iTerm2, Apple Terminal, GNOME Terminal, Ptyxis, xfce4-terminal), Linux
|
||||
truecolor, **and every other unknown POSIX terminal**. The default is *risky*
|
||||
on purpose.
|
||||
- `shouldEnableSynchronizedOutputByDefault(env, id)` → DEC 2026 default. Precedence:
|
||||
user opt-out (`PI_NO_SYNC_OUTPUT`/`PI_TUI_SYNC_OUTPUT=0`) → user force-on
|
||||
(`PI_FORCE_SYNC_OUTPUT=1`/`PI_TUI_SYNC_OUTPUT=1`) → `TERM_FEATURES` advertises
|
||||
`Sy` → `WT_SESSION` (WT/WSL) → known direct terminals
|
||||
(kitty/ghostty/wezterm/iterm2/alacritty/vscode; SSH passes through) → off for
|
||||
risky multiplexers and everything else (VTE-family, GNU screen, Apple Terminal,
|
||||
legacy conhost, unknown). Reconciled at runtime by the DECRQM mode-2026 report:
|
||||
a positive report **enables** sync (upgrading default-off muxes like
|
||||
zellij/tmux-master), a negative one disables it; a user override still wins.
|
||||
`synchronizedOutputUserOverride(env)` is the shared opt-out/force resolver.
|
||||
- `detectRectangularSgrSupport(id, env)` → DECCARA fills: **kitty only**
|
||||
(ghostty does not implement the SGR-background extension), off in multiplexers
|
||||
and under `PI_NO_DECCARA`.
|
||||
- `shouldEnableSynchronizedOutputByDefault(env, id)` → DEC 2026 default.
|
||||
Precedence: user opt-out (`PI_NO_SYNC_OUTPUT`/`PI_TUI_SYNC_OUTPUT=0`) → user
|
||||
force-on (`PI_FORCE_SYNC_OUTPUT=1`/`PI_TUI_SYNC_OUTPUT=1`) → `TERM_FEATURES`
|
||||
advertises `Sy` → `WT_SESSION` → known direct terminals → off for risky
|
||||
multiplexers and unknowns. Reconciled at runtime by the DECRQM mode-2026
|
||||
report; a user override still wins.
|
||||
- `detectRectangularSgrSupport(id, env)` → DECCARA fills: kitty only, off in
|
||||
multiplexers and under `PI_NO_DECCARA`.
|
||||
- `supportsScreenToScrollback` → kitty's ED22 (used once, on the initial
|
||||
paint, to preserve the pre-existing shell screen).
|
||||
|
||||
**Why this keeps leaking:** terminal class is inferred from env vars that are
|
||||
**not durable**. `VTE_VERSION` is stripped by `sshd` (default `AcceptEnv`);
|
||||
`COLORTERM` is also not in default `AcceptEnv`; some hosts (Tabby) set no
|
||||
distinguishing var; WSL-fronting-WT is neither pure win32 nor pure POSIX. Every
|
||||
missed env var is a missed terminal class is a new complaint. The mitigations
|
||||
are: (a) **default unknown to risky** rather than safe, and (b) detect by
|
||||
*behavior/handshake* (DECRQM) where possible rather than a host allow-list. When
|
||||
you add a terminal, add it to the pure detector and add the **SSH-stripped env
|
||||
shape** to the test, not just the env-present shape.
|
||||
The old ED3-risk classifier (`eagerEraseScrollbackRisk`, `PI_TUI_ED3_SAFE`,
|
||||
`submitPinsViewportToTail`) is gone: behavior no longer depends on which
|
||||
terminal is rendering, so there is no risk class to detect. Env sniffing now
|
||||
only selects *optimizations* (sync output, DECCARA, images), where a miss is
|
||||
cosmetic, not corrupting.
|
||||
|
||||
---
|
||||
|
||||
## 6. Width model
|
||||
## 5. Width model
|
||||
|
||||
`visibleWidth` / `truncateToWidth` / `sliceByColumn` / `wrapTextWithAnsi`
|
||||
(`utils.ts`) all route through **one native UAX#11 engine** (`@oh-my-pi/pi-natives`,
|
||||
Rust `unicode-width`). We deliberately dropped `Bun.stringWidth` because it
|
||||
disagreed with the engine on combining marks and jamo, and mixing two width
|
||||
models in measure-vs-slice produced the crashes.
|
||||
(`utils.ts`) all route through **one native UAX#11 engine**
|
||||
(`@oh-my-pi/pi-natives`, Rust `unicode-width`). `Bun.stringWidth` was dropped
|
||||
deliberately — mixing two width models in measure-vs-slice produced crashes.
|
||||
|
||||
- Fast path: printable ASCII is one cell per code unit.
|
||||
- ZWJ pictographic emoji take the `visibleWidthByGrapheme` override (ANSI spans
|
||||
excised first, then `Intl.Segmenter`), because the native scanner double-counts
|
||||
SGR bytes when a sequence is split by the segmenter.
|
||||
- OSC 66 sized text (`\x1b]66;…`) takes the native path.
|
||||
- ZWJ pictographic emoji take the `visibleWidthByGrapheme` override.
|
||||
- OSC 66 sized text takes the native path.
|
||||
|
||||
**Rule:** if you add a code path that measures width, route it through these
|
||||
helpers. Never reintroduce `Bun.stringWidth` or a parallel width table — the
|
||||
measure model and the slice/wrap model must agree, or you get over-wide lines
|
||||
that the hot-path clamp silently truncates (cosmetic loss) or, worse, seam
|
||||
duplication.
|
||||
**Rule:** any new measuring code routes through these helpers, and the hot
|
||||
path clamps instead of throwing. Known residual: combining-heavy scripts
|
||||
(Arabic harakat) survive painting verbatim, but ghostty-web's cell readback can
|
||||
migrate non-spacing marks across cells — the stress harness compares those rows
|
||||
with marks stripped (`sameLinesAllowingMarkDrift`).
|
||||
|
||||
---
|
||||
|
||||
## 6. The fidelity gate (use it)
|
||||
|
||||
`packages/tui/test/render-stress-harness.ts` drives the renderer's **real
|
||||
emitted ANSI** into a ghostty-web `VirtualTerminal` across randomized op
|
||||
sequences and parameterized terminal shapes, and validates the contract with a
|
||||
**shadow commit ledger**: an independent reimplementation of §1's math, fed
|
||||
only by observed frames (a `render` wrap) and observed bytes (a `write` wrap).
|
||||
Per op it asserts:
|
||||
|
||||
- the whole tape (scrollback + grid) equals `shadowTape + window slice`, row
|
||||
for row, including across resizes;
|
||||
- scrolled readers stay pinned and visible history rows are never rewritten;
|
||||
- multiplexer pane history grows by exactly the committed chunk;
|
||||
- sync-output/autowrap bracket discipline, cursor parking, background columns,
|
||||
duplicate accounting.
|
||||
|
||||
Run it — plus `render-regressions.test.ts`,
|
||||
`streaming-scrollback-defer.test.ts`, and the `issue-*-repro.test.ts` files —
|
||||
before changing ledger math, emitters, or the seam. A change that passes one
|
||||
terminal and one seed is not verified.
|
||||
|
||||
---
|
||||
|
||||
@@ -300,90 +269,71 @@ a non-answering terminal is detected when DA1 returns first. Replies can arrive
|
||||
**split across a stdin flush**, so:
|
||||
|
||||
- `#privateCsiResponseBuffer` accumulates `\x1b[?…` partials while a sentinel is
|
||||
outstanding, rejoins on the terminator byte (0x40–0x7e), then runs the
|
||||
DA1/kitty/mode-2031 handlers on the **complete** reply. A new `\x1b`
|
||||
mid-reassembly or >256 bytes abandons the partial so real keys (e.g. arrow
|
||||
`\x1b[A`) still reach input.
|
||||
- `#da1SentinelOwners` is a **typed FIFO** discriminated by `kind` (`keyboard`,
|
||||
`osc11`, `privateMode`, `kittyGraphicsProbe`, `osc99Probe`) so a keyboard DA1
|
||||
cannot be mistaken for an OSC 11 / DECRQM / graphics-probe sentinel.
|
||||
- DECRQM probes (`#queryPrivateMode(2026/2048/2031)`) record support via DECRPM
|
||||
and drive runtime feature gating (e.g. auto-disabling DEC 2026 sync output).
|
||||
outstanding, rejoins on the terminator byte, then runs the handlers on the
|
||||
**complete** reply. A new `\x1b` mid-reassembly or >256 bytes abandons the
|
||||
partial so real keys still reach input.
|
||||
- `#da1SentinelOwners` is a **typed FIFO** discriminated by `kind` so a
|
||||
keyboard DA1 cannot be mistaken for an OSC 11 / DECRQM / graphics-probe
|
||||
sentinel.
|
||||
- DECRQM probes (2026/2048/2031) drive runtime feature gating.
|
||||
|
||||
**Rule:** any new probe must own a typed sentinel and survive a split reply. The
|
||||
contract is bytes-in = bytes-out; it is testable, so test it (feed the reply
|
||||
byte-by-byte and assert nothing leaks to the input handler).
|
||||
**Rule:** any new probe must own a typed sentinel and survive a split reply
|
||||
(feed the reply byte-by-byte in a test and assert nothing leaks to input).
|
||||
|
||||
---
|
||||
|
||||
## 8. Inline images & memory
|
||||
|
||||
Kitty images are **transmit-once, place-many** (`kitty-graphics.ts`):
|
||||
`encodeKittyTransmit` (`a=t`, keyed by a stable `i=`) writes the base64 a single
|
||||
time; repaints emit only `encodeKittyPlacement` (`a=p`). Text clears
|
||||
(`CSI 2 J` / `CSI 3 J`) do **not** purge the terminal's image store — only
|
||||
`encodeKittyDeleteImage` (`a=d,d=I`) does. `ImageBudget` (`components/image.ts`)
|
||||
keeps only the most-recent N images live; demoted images render their text
|
||||
fallback and are explicitly purged.
|
||||
Kitty images are **transmit-once, place-many** (`kitty-graphics.ts`).
|
||||
`ImageBudget` keeps only the most-recent N images live; when the cap is
|
||||
exceeded the demoted image's pixels are deleted by id (`a=d,d=I`) and its
|
||||
visible rows re-render as the text fallback through the ordinary window diff —
|
||||
**no destructive replay**. A demoted placement already committed to history
|
||||
simply loses its pixels (committed rows are immutable), and the text fallback
|
||||
is **height-preserving** once a graphic has rendered (reserved rows + fallback
|
||||
line), so demotion never shrinks the block and never shifts committed content
|
||||
below it.
|
||||
|
||||
**Rule:** never re-emit full base64 per frame (it pegged RAM and pinned the UI
|
||||
thread). Kitty Unicode placeholders are default-on only for kitty/ghostty
|
||||
(`PI_NO_KITTY_PLACEHOLDERS` / `PI_KITTY_PLACEHOLDERS`); other Kitty-protocol
|
||||
hosts render placeholder cells as literal PUA glyphs, so they fall back to
|
||||
direct `a=p` placement.
|
||||
**Rule:** never re-emit full base64 per frame. Kitty Unicode placeholders are
|
||||
default-on only for kitty/ghostty (`PI_NO_KITTY_PLACEHOLDERS` /
|
||||
`PI_KITTY_PLACEHOLDERS`).
|
||||
|
||||
---
|
||||
|
||||
## 9. The fidelity gate (use it)
|
||||
|
||||
`packages/tui/test/render-stress-harness.ts` renders the renderer's **real emitted ANSI** into
|
||||
a ghostty-web `VirtualTerminal` and asserts viewport fidelity (a scrolled reader
|
||||
stays put), background-column fidelity, and scrollback-buffer fidelity, across
|
||||
parameterized terminal shapes and randomized op sequences.
|
||||
|
||||
This harness is the structural fix for the whole recurrence: every guess-flip and
|
||||
sniffing-gap regression historically **shipped blind and was caught by a user**,
|
||||
because no automated "a scrolled-up reader stays pinned across kitty/WT/WSL/
|
||||
ConPTY" assertion gated CI. **Before you change the eager/defer lever, a
|
||||
predicate, the live-region seam, or width math, run the stress harness and the
|
||||
targeted repro tests** (`packages/tui/test/render-regressions.test.ts`,
|
||||
`packages/tui/test/streaming-scrollback-defer.test.ts`, the `issue-*-repro.test.ts` files).
|
||||
A change that passes one terminal and one seed is not verified.
|
||||
|
||||
---
|
||||
|
||||
## 10. Escape hatches (env vars)
|
||||
## 9. Escape hatches (env vars)
|
||||
|
||||
| Var | Effect |
|
||||
|---|---|
|
||||
| `PI_NO_SYNC_OUTPUT=1` | Disable DEC 2026 BSU/ESU wrappers (autowrap discipline stays on). For terminals that advertise but mishandle mode 2026. |
|
||||
| `PI_NO_SYNC_OUTPUT=1` | Disable DEC 2026 BSU/ESU wrappers (autowrap discipline stays on). |
|
||||
| `PI_TUI_SYNC_OUTPUT=0\|1` / `PI_FORCE_SYNC_OUTPUT=1` | Force sync output off / on. |
|
||||
| `PI_TUI_ED3_SAFE=1` | Declare the terminal safe for live ED3 (disables `eagerEraseScrollbackRisk`). |
|
||||
| `PI_NO_DECCARA` | Disable Kitty DECCARA rectangular-fill optimization (force padded-string fills). |
|
||||
| `PI_NO_DECCARA` | Disable Kitty DECCARA rectangular-fill optimization. |
|
||||
| `PI_FORCE_IMAGE_PROTOCOL=kitty\|iterm2\|sixel\|off` | Override image protocol detection. |
|
||||
| `PI_NO_KITTY_PLACEHOLDERS=1` / `PI_KITTY_PLACEHOLDERS=1` | Force Kitty Unicode placeholders off / on. |
|
||||
| `PI_CLEAR_ON_SHRINK=1` | Clear empty rows when content shrinks (default off). |
|
||||
| `PI_HARDWARE_CURSOR=1` | Show the real hardware cursor instead of a rendered one. |
|
||||
| `PI_NOTIFICATIONS=off\|0\|false` | Suppress terminal notifications. |
|
||||
| `PI_DEBUG_REDRAW=1` | Log the chosen render intent per frame to the debug log. |
|
||||
| `PI_TUI_DEBUG=1` | Dump per-render diff state under `/tmp/tui`. |
|
||||
| `PI_DEBUG_REDRAW=1` | Log the chosen render intent + ledger state per frame to the debug log. |
|
||||
|
||||
Removed with the old engine: `PI_TUI_ED3_SAFE` (no ED3-risk lever exists),
|
||||
`PI_CLEAR_ON_SHRINK` (shrinks always clear exactly), `PI_TUI_DEBUG` (per-render
|
||||
dump superseded by `PI_DEBUG_REDRAW` ledger logging and the stress harness
|
||||
replay/reduce tooling).
|
||||
|
||||
---
|
||||
|
||||
## 11. Before you touch the render core — checklist
|
||||
## 10. Before you touch the render core — checklist
|
||||
|
||||
- [ ] Are you about to emit `CSI 3 J` anywhere other than the destructive
|
||||
`clearScrollback` path? **Stop.**
|
||||
- [ ] Does your change trust `isNativeViewportAtBottom() === undefined` as
|
||||
"at bottom" during passive streaming? **Stop.**
|
||||
- [ ] Did you change one structural-mutation branch without mirroring its
|
||||
sibling (shrink ↔ grow)? **Defer symmetrically.**
|
||||
- [ ] Could any frame now emit zero bytes while the viewport is at the bottom?
|
||||
That's invisible-until-resize.
|
||||
- [ ] Did you add a terminal by brand instead of by behavior, or skip the
|
||||
SSH-stripped env shape in the test?
|
||||
- [ ] Did you run `packages/tui/test/render-stress-harness.ts` + the repro suite across
|
||||
win32/POSIX × unknown/scrolled/at-bottom — not just one terminal?
|
||||
- [ ] Are you about to emit `CSI 3 J` anywhere other than the gesture-driven
|
||||
`clearScrollback` full paint? **Stop.**
|
||||
- [ ] Could any code path rewrite, or re-show on the grid, a frame row below
|
||||
`committedRows`? **Stop.**
|
||||
- [ ] Does your byte shape scroll rows that are not the commit chunk? That
|
||||
breaks `scrollback == frame[0..C)`.
|
||||
- [ ] Are you adding a viewport probe, a platform fork, or a terminal-brand
|
||||
branch to the update path? The contract exists so none are needed.
|
||||
- [ ] New mutable UI above the editor? It must report (or live inside) the
|
||||
live-region seam, or it will freeze at first commit.
|
||||
- [ ] Did you run the stress harness and the repro suite across the full
|
||||
scenario matrix — not just one terminal and one seed?
|
||||
- [ ] New probe? Typed sentinel owner + split-reply test.
|
||||
- [ ] New width path? Routed through the shared native engine, clamped (never
|
||||
thrown) in the hot path.
|
||||
|
||||
@@ -29,7 +29,7 @@ Boundary rule: the TUI engine is message-agnostic. It only knows `Component.rend
|
||||
|
||||
## Boot and component tree assembly
|
||||
|
||||
`InteractiveMode` constructs `TUI(new ProcessTerminal(), settings.get("showHardwareCursor"))`, applies `clearOnShrink`, `tui.maxInlineImages`, and Kitty text-sizing settings, then creates persistent containers:
|
||||
`InteractiveMode` constructs `TUI(new ProcessTerminal(), settings.get("showHardwareCursor"))`, applies `tui.maxInlineImages` and Kitty text-sizing settings, then creates persistent containers:
|
||||
|
||||
- `chatContainer`
|
||||
- `pendingMessagesContainer`
|
||||
@@ -97,29 +97,22 @@ Routing details:
|
||||
|
||||
This keeps key parsing/editor mechanics in `packages/tui` and mode semantics in coding-agent controllers.
|
||||
|
||||
## Render loop and diffing strategy
|
||||
## Render loop and the append-only contract
|
||||
|
||||
`TUI.requestRender()` coalesces render requests and rate-limits ordinary frames:
|
||||
|
||||
- forced renders (`requestRender(true, ...)`) schedule an immediate frame and set `#forceViewportRepaintOnNextRender`; with `clearScrollback`, they also queue `sessionReplace`
|
||||
- forced renders (`requestRender(true, ...)`) schedule an immediate frame and force a full window rewrite; with `clearScrollback`, they trigger a destructive full paint (ED3 outside multiplexers)
|
||||
- ordinary renders schedule through `#scheduleRender()` and respect `TUI.#MIN_RENDER_INTERVAL_MS`
|
||||
- repeated requests while a render is pending collapse into the same scheduled frame
|
||||
|
||||
`#doRender()` pipeline:
|
||||
|
||||
1. Render root component tree to `newLines`.
|
||||
2. Composite visible overlays (if any).
|
||||
3. Extract and strip `CURSOR_MARKER` from the visible viewport.
|
||||
4. Normalize non-image lines and append reset/hyperlink terminators.
|
||||
5. Classify the frame into a render intent:
|
||||
- initial paint / forced viewport repaint
|
||||
- explicit session replacement or native scrollback rebuild
|
||||
- viewport repaint for width/height/offscreen mutations
|
||||
- deferred mutation/shrink when native scrollback is scrolled
|
||||
- trailing shrink
|
||||
- changed-line diff
|
||||
- noop
|
||||
6. Emit only the bytes required by the intent and commit cached frame/cursor/viewport state.
|
||||
1. Render root component tree, collecting the commit-boundary seam (`NativeScrollbackLiveRegion`) from the children.
|
||||
2. Advance the append-only ledger: `windowTop = max(committedRows, frame.length - height)`, commit chunk = settled rows crossing the window top (never past the seam).
|
||||
3. Extract and strip `CURSOR_MARKER`, normalize lines, slice the visible window, composite overlays into the window slice (screen coordinates; overlays freeze commits).
|
||||
4. Emit one of: gesture-driven full paint (initial / session replace / resize), scroll-append (chunk rows only), in-window row diff, or seam rewrite (chunk + full window).
|
||||
|
||||
Native scrollback always equals the committed frame prefix — rows enter history exactly once, in order, when the seam says they are final. There are no viewport probes and no deferred reconciliation; see [`tui-core-renderer.md`](./tui-core-renderer.md).
|
||||
|
||||
Render writes use synchronized output mode (`CSI ? 2026 h/l`) when enabled; capability detection, DECRQM, or `PI_NO_SYNC_OUTPUT` can disable the wrappers while leaving autowrap discipline on.
|
||||
|
||||
@@ -145,9 +138,8 @@ Resize events are event-driven from `ProcessTerminal` to `TUI.requestRender()`.
|
||||
|
||||
Effects:
|
||||
|
||||
- Width or height changes repaint or rebuild because terminal reflow invalidates wrapping, viewport, and cursor anchors.
|
||||
- Inside terminal multiplexers, resize uses viewport repaint instead of destructive native-scrollback replay; pane history cannot be erased safely and a full replay duplicates transcript rows.
|
||||
- Viewport/top tracking (`#viewportTopRow`, `#maxLinesRendered`, scrollback high-water state) avoids invalid relative cursor math and defers destructive native scrollback rewrites while the user is scrolled into history.
|
||||
- A resize is an explicit user gesture: outside multiplexers the engine erases and replays (`ED3` + full paint) so history rewraps at the new geometry; the commit ledger restarts from the replayed frame.
|
||||
- Inside terminal multiplexers, resize repaints the visible window in place after a settle debounce (issue #2088); pane history keeps its old wrap, like any shell output, because pane scrollback cannot be erased safely.
|
||||
- Overlay visibility can depend on terminal dimensions (`OverlayOptions.visible`); focus is corrected when overlays become non-visible after resize.
|
||||
|
||||
## Streaming and incremental UI updates
|
||||
|
||||
+9
-9
@@ -20,15 +20,15 @@
|
||||
"@huggingface/transformers": "^4.2.0",
|
||||
"@mozilla/readability": "^0.6.0",
|
||||
"@napi-rs/cli": "3.7.0",
|
||||
"@oh-my-pi/hashline": "15.10.9",
|
||||
"@oh-my-pi/omp-stats": "15.10.9",
|
||||
"@oh-my-pi/pi-agent-core": "15.10.9",
|
||||
"@oh-my-pi/pi-ai": "15.10.9",
|
||||
"@oh-my-pi/pi-coding-agent": "15.10.9",
|
||||
"@oh-my-pi/pi-mnemopi": "15.10.9",
|
||||
"@oh-my-pi/pi-natives": "15.10.9",
|
||||
"@oh-my-pi/pi-tui": "15.10.9",
|
||||
"@oh-my-pi/pi-utils": "15.10.9",
|
||||
"@oh-my-pi/hashline": "15.10.10",
|
||||
"@oh-my-pi/omp-stats": "15.10.10",
|
||||
"@oh-my-pi/pi-agent-core": "15.10.10",
|
||||
"@oh-my-pi/pi-ai": "15.10.10",
|
||||
"@oh-my-pi/pi-coding-agent": "15.10.10",
|
||||
"@oh-my-pi/pi-mnemopi": "15.10.10",
|
||||
"@oh-my-pi/pi-natives": "15.10.10",
|
||||
"@oh-my-pi/pi-tui": "15.10.10",
|
||||
"@oh-my-pi/pi-utils": "15.10.10",
|
||||
"@opentelemetry/api": "^1.9.1",
|
||||
"@opentelemetry/context-async-hooks": "^2.7.1",
|
||||
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
|
||||
|
||||
@@ -2,6 +2,10 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
|
||||
- Editorial pass over the compaction prompts: fixed garbled grammar and missing articles, RFC-keyed prohibitions, deduped restated instructions; parsed markers (`<read-files>`/`<modified-files>`/`<previous-summary>`) and all output-format headings left byte-identical
|
||||
|
||||
## [15.10.8] - 2026-06-09
|
||||
### Added
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-agent-core",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -4,7 +4,7 @@ You MUST use EXACT format:
|
||||
|
||||
## Goal
|
||||
|
||||
[What user trying to accomplish in this branch?]
|
||||
[What is the user trying to accomplish in this branch?]
|
||||
|
||||
## Constraints & Preferences
|
||||
- [Constraints, preferences, requirements mentioned]
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
Another language model started to solve this problem and produced a summary of its thinking process. You also have access to the state of the tools that were used by that language model. You MUST use this to build on the work that has already been done and NEVER duplicate work. Here is the summary produced by the other language model; you MUST use the information in this summary to assist with your own analysis:
|
||||
Another language model started to solve this problem and produced a summary of its thinking process. You also have access to the state of the tools that model used. You MUST build on the work already done and NEVER duplicate it. Here is that summary:
|
||||
|
||||
<summary>
|
||||
{{summary}}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
You MUST summarize the conversation above into a structured context checkpoint handoff summary for another LLM to resume task.
|
||||
You MUST summarize the conversation above into a structured handoff summary for another LLM to resume the task.
|
||||
|
||||
IMPORTANT: If conversation ends with unanswered question to user or imperative/request awaiting user response (e.g., "Please run command and paste output"), you MUST preserve that exact question/request.
|
||||
IMPORTANT: If the conversation ends with an unanswered question or a request awaiting user response (e.g., "Please run command and paste output"), you MUST preserve that exact question/request.
|
||||
|
||||
You MUST use this format (sections can be omitted if not applicable):
|
||||
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
You MUST incorporate new messages above into the existing handoff summary in <previous-summary> tags, used by another LLM to resume task.
|
||||
You MUST incorporate the new messages above into the existing handoff summary in <previous-summary> tags, used by another LLM to resume the task.
|
||||
RULES:
|
||||
- MUST preserve all information from previous summary
|
||||
- MUST preserve all information from the previous summary
|
||||
- MUST add new progress, decisions, and context from new messages
|
||||
- MUST update Progress: move items from "In Progress" to "Done" when completed
|
||||
- MUST update "Next Steps" based on what was accomplished
|
||||
- MUST preserve exact file paths, function names, and error messages
|
||||
- You MAY remove anything no longer relevant
|
||||
|
||||
IMPORTANT: If new messages end with unanswered question or request to user, you MUST add it to Critical Context (replacing any previous pending question if answered).
|
||||
IMPORTANT: If the new messages end with an unanswered question or request to the user, you MUST add it to Critical Context (replacing any previous pending question if answered).
|
||||
|
||||
You MUST use this format (omit sections if not applicable):
|
||||
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
Summarize conversations between users and AI coding assistants. Produce structured summaries in the exact specified format.
|
||||
|
||||
Do NOT continue the conversation. Do NOT respond to questions in the conversation. Output ONLY the structured summary.
|
||||
NEVER continue the conversation. NEVER respond to questions in it. Output ONLY the structured summary.
|
||||
|
||||
@@ -2,6 +2,99 @@
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
|
||||
- Reduced idle-watchdog churn on the token hot path: the abort promise/listener is created once per stream instead of per yielded item, the deadline uses a persistent re-armed timer instead of a `setTimeout` create/destroy pair per delta, and the persistent race promises are re-minted every 1024 items so per-race reaction records cannot accumulate for the stream's whole life.
|
||||
- Memoized Anthropic many-image downscaling by content-block identity, so long sessions with stable message objects no longer re-decode and re-encode every oversized image on each request and retry.
|
||||
- Tool-argument validation errors now truncate embedded argument strings at 256 chars per field — a failed `write`-class call no longer echoes hundreds of KB of payload back to the model as the error message.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed Gemini streaming silently presenting truncated or blocked output as a successful `stop`: in-band `{"error":{...}}` events and `promptFeedback.blockReason` chunks were never inspected, and a stream ending without any `finishReason` kept the initialized `stop` — all three now surface as errors (both the API-key and gemini-cli/Antigravity consumers), and the `toolUse` stop-reason override no longer masks `SAFETY`/`MALFORMED_FUNCTION_CALL` finishes that arrive after a valid tool call.
|
||||
- Fixed Gemini/Bedrock error finishes reporting "An unknown error occurred": the raw finish/stop reason (`MALFORMED_FUNCTION_CALL`, `RECITATION`, `guardrail_intervened`, …) is now recorded into the surfaced error message.
|
||||
- Fixed the Anthropic provider retry loop ignoring server `retry-after` on 429/529 — it now waits `max(headerDelay, backoff)` instead of hammering a rate-limited endpoint three times within ~14s of guaranteed failures.
|
||||
- Fixed in-stream Anthropic SSE `error` events being thrown as raw JSON envelopes; the structured `error.type`/`message` is parsed out, keeping retry classification on the typed token instead of accidental regex hits.
|
||||
- Fixed transparent-reconnect tolerance duplicating content behind replaying proxies: after a duplicate `message_start`, replayed `content_block_start` events for already-closed indexes are now consumed silently instead of appending duplicate text/tool calls.
|
||||
- Fixed the Anthropic gateway accepting malformed known-type content blocks (e.g. `{type:"text", text:123}`) through the unknown-block catch-all, corrupting history and surfacing later as an opaque TypeError — they now fail validation with a clean 400. The gateway's encode stream also emits `ping` keepalives every 15s and a complete `message_start`/`message_delta`/`message_stop` envelope when the inner stream ends without a terminal event, so strict clients no longer classify slow or empty streams as protocol errors.
|
||||
- Fixed the Mistral `requiresThinkingAsText` replay path calling `.unshift()` on string assistant content — an unconditional TypeError that failed any same-model history turn carrying both thinking and text.
|
||||
- Fixed the Responses gateway stripping `encrypted_content` from inbound reasoning items (strip-mode schema), which broke codex-style stateless replay; the schema is now loose, restoring the symmetry the outbound encoder already preserved. Composite internal `callId|itemId` ids are also split before hitting the wire so third-party clients that validate `call_id` charsets no longer reject them.
|
||||
- Ported the shared unfinished-tool-call sweep to the codex `response.completed` handler, so a lost `output_item.done` can no longer persist a tool call with stale `{}` arguments and transient parser fields into session history.
|
||||
- Fixed live text freezing until item completion when a lossy proxy drops `content_part.added`: the missing part is now synthesized on the first `output_text`/`refusal` delta (shared and codex decoders).
|
||||
- Fixed interleaved `content`/`tool_calls` deltas fragmenting a tool call into a truncated call plus a nameless phantom: text/thinking transitions no longer finish open tool-call blocks, so index-only continuation deltas re-find them.
|
||||
- Fixed the Azure chat-completions path ignoring `AZURE_OPENAI_DEPLOYMENT_NAME_MAP` (only the Responses provider honored it), producing opaque 404s when deployment names differ from catalog model ids.
|
||||
- Fixed the chat gateway discarding inbound assistant `reasoning_content`, which fed DeepSeek/Kimi exact-replay upstreams a placeholder instead of the model's actual reasoning; it now round-trips as a thinking block, and `toolcall_end` emits a corrective id/name chunk when the streamed start carried empty values.
|
||||
- Fixed the auth retry loop minting OAuth tokens and firing a doomed request after the caller aborted, and stopped masking resolver failures (broker/network/refresh errors) as "No API key" — the actual cause is preserved.
|
||||
- Fixed `EventStream.end()` without a terminal result leaving `.result()` pending forever (reachable via extension streams and the lazy wrapper); it now rejects with a synthesized error.
|
||||
- Fixed the Copilot retry wrapper blind-retrying every retryable error with fixed 400ms delays: 429/5xx now honor `Retry-After` (capped at 30s) and other statuses are not retried, while status-less transport blips keep the linear retry.
|
||||
- Fixed the OpenAI completions error path ending the stream without closing open text/thinking/tool-call blocks, leaving consumers with orphaned block lifecycles on every stream error or idle-timeout abort.
|
||||
- Fixed DSML hold-back freezing display on any bare `<` in model output for up to 256 chars: idle-state holding now only triggers on a strict DSML section-open prefix, and blowing the 1MB parameter cap no longer leaks the closing envelope tags as visible text; a capped parameter value also carries an explicit `…[parameter truncated]` marker instead of executing the tool with silently corrupted input.
|
||||
- Fixed schema normalization blanking DAG-shared subtrees to `{}`: the visited-set cycle guard treated a subschema object reused across two properties as a cycle; path-tracking `enter`/`exit` now allows sharing while still short-circuiting true cycles, frozen input schemas no longer throw, and the path counter no longer leaks depth on the cycle branch (which made every later normalization of the same object misreport a cycle).
|
||||
- Fixed shared in-flight Google token refreshes being bound to the first caller's `AbortSignal`, failing every concurrent waiter when one parallel Vertex call was cancelled; callers now race their own signal against a detached refresh, which is bounded by its own 30s timeout so a hung fetch cannot pin the in-flight slot until process restart.
|
||||
- Fixed Gemini <3 multimodal tool results breaking the single-function-response-turn invariant for parallel tool calls (image turns are buffered and flushed after the merged functionResponse turn), and the gemini-cli consumer now defaults missing `functionCall.args` to `{}` like the shared consumer.
|
||||
- Fixed Bedrock dropping `toolConfig` entirely when `toolChoice` is `"none"` while history still contains tool blocks — the Converse API rejects such requests, so tool specs are kept and only the choice is omitted.
|
||||
- Fixed AWS credential handling serving expired credentials until process restart: cache entries are invalidated on 401/403, file-sourced session-token credentials get a 5-minute TTL, and concurrent first requests single-flight instead of spawning duplicate `credential_process`/SSO fetches — the shared resolution is detached from the first caller's abort signal (one cancelled request no longer fails every waiter) and bounded by its own 30s timeout. The eventstream reader also cancels the response body on abnormal exit instead of leaving the HTTP connection draining.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the dead `iterateUntilAbort` helper (superseded by `iterateWithIdleTimeout`); it leaked the upstream iterator when the consumer abandoned mid-yield and had no production call sites.
|
||||
|
||||
## [15.10.10] - 2026-06-09
|
||||
|
||||
### Added
|
||||
|
||||
- Exported `wrapFetchForCch` so non-streaming OAuth callers (e.g. the web-search provider) can patch the Claude Code billing-header `cch` attestation into their request bodies instead of shipping the `cch=00000` placeholder.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed an unbounded, zero-backoff Codex WebSocket reconnect loop on `websocket_connection_limit_reached`: the no-content reconnect path never consulted the retry budget and never waited, hammering the endpoint forever when the limit is account-scoped. Reconnects are now budgeted and delayed like every other WS retry path, falling back to a single SSE replay when exhausted.
|
||||
- Fixed the Codex whitespace-loop breaker not observing degenerate frames that arrive after their item closed (or before it opened) — those frames count as stream progress, so the idle watchdogs never fired and the turn hung forever, which is exactly the failure mode the breaker exists for. Whitespace-loop recovery now also refuses to replay the turn once a `toolcall_end` was delivered, surfacing the error instead of re-emitting the same tool calls.
|
||||
- Fixed the two remaining Codex retry paths (WS mid-stream reconnect and the empty-content SSE fallback) leaking blockless native output items (e.g. `web_search_call`) from the failed attempt into the replayed turn's `providerPayload` and append baseline.
|
||||
- Fixed Codex WebSocket failure handling closing whatever connection currently occupies the session slot — including a concurrent caller's in-flight CONNECTING handshake, whose rejection (`websocket closed before open`) is classified fatal and disabled WebSockets for the whole session. Failure cleanup now skips CONNECTING sockets and the pool re-joins replacement handshakes (bounded).
|
||||
- Fixed the Codex request transformer not repairing orphan `custom_tool_call_output` items (only `function_call_output` was folded into an assistant note) — a compaction splice that dropped an `apply_patch` call while keeping its result produced a hard 400 on the default GPT-5 Codex toolset.
|
||||
- Fixed `processResponsesStream` finalizing reasoning items via a bare `itemId` content scan instead of the routed entry: with id-less reasoning items (local hosts), every `output_item.done` matched the FIRST thinking block — the second item's text clobbered it and the second block was never finalized or signed.
|
||||
- Fixed `processResponsesStream` dropping tool calls and message text whose `output_item.added` event was lost (lossy proxies): `toolcall_end` was emitted with a dangling contentIndex while the call never entered `message.content`, so the agent loop silently never executed it. The done handler now synthesizes the missing block; still-open tool-call blocks are also final-parsed at `response.completed` so the `toolUse` override cannot hand the agent stale `{}` arguments.
|
||||
- Fixed `response.incomplete` with `incomplete_details.reason: "content_filter"` being reported as a token-cap truncation (`stopReason: "length"`) — the agent loop's length recovery then asked the model to "shorten" a filtered prompt. Content-filtered turns now surface as errors; usage is also populated from `response.failed` events, and an unknown terminal status degrades to `"stop"` with a logged anomaly instead of throwing away a fully-streamed response.
|
||||
- Fixed Copilot `premiumRequests` accounting being dropped from failed/cancelled responses: `populateResponsesUsageFromResponse` replaced `usage` wholesale and the error path threw before the success-path re-apply. The populate now preserves the field.
|
||||
- Fixed `deduplicateToolCallIds` suffixing the whole composite Responses id (`callId|itemId`) — `normalizeResponsesToolCallId` extracts the first segment as the wire `call_id` at encode time, so both copies collapsed back onto one `call_id` and the request carried duplicate call/output pairs. The suffix and length budget now apply per segment.
|
||||
- Gated native history payload replay on api + model id in both Responses providers: after a mid-session model switch, reasoning items carrying encrypted content minted by the previous model were replayed verbatim under the new model. Replay now falls back to block re-encode (which already strips foreign signatures), matching `transformMessages`' same-model trust rule.
|
||||
- Fixed Azure OpenAI Responses requests omitting `store: false` while requesting `reasoning.encrypted_content` (stateless-only per OpenAI), replaying custom tool calls paired with mismatched `function_call_output` items (customCallIds was never threaded through), letting the SDK's internal retries (maxRetries 5) silently re-POST inside the explicit first-event deadline, and sending a `prompt_cache_key` when the caller opted out via `cacheRetention: "none"`.
|
||||
- Fixed strict-pairing Responses backends (Azure, Copilot) silently discarding tool results whose call is absent from history — the result is now folded into an assistant note (same shape as orphan-output repair) so the model keeps the information.
|
||||
- Fixed the OpenAI Responses first-event watchdog staying armed across the `onResponse` notification callback (a slow callback aborted an already-connected stream), Copilot transient-model retries re-attempting on an already-aborted signal (instant dead retry surfacing the scheduler's AbortError), Codex `reasoningSummary: null` being coerced to `"auto"` (the documented omit-summary contract was unreachable), nested Codex error codes (`response.error.code`) being invisible to the connection-limit/previous-response recovery matchers, and the session id leaking unredacted into `PI_CODEX_DEBUG` logs via the `x-client-request-id` header.
|
||||
- Fixed `processResponsesStream` (shared by `openai-responses` and `azure-openai-responses`) ignoring the terminal `response.incomplete` event: a max-output-tokens-truncated response ended with `stopReason: "stop"`, zero usage, and no cost instead of `"length"` with the reported token counts. `response.incomplete` is now handled alongside `response.completed` and counts as stream progress for the idle watchdogs.
|
||||
- Fixed custom tool-call content blocks keeping the transient `partialJson` accumulation buffer (and a potentially stale `arguments.input`) after `response.output_item.done` in the shared Responses stream processor — the function_call branch already cleaned these up.
|
||||
- Fixed two OpenAI Codex stream-retry paths (whitespace-loop recovery and retryable provider errors) leaking native output items from the abandoned attempt into the replayed turn's `providerPayload` — stale reasoning items completed before the failure were re-sent as history input on subsequent requests alongside the retry's own items.
|
||||
- Fixed the Codex WebSocket queue wiping already-received frames when a transport error arrived: a `response.completed` queued just before an eager server close was discarded, turning a finished response into a spurious `websocket closed` failure and a full request replay. Errors now append behind pending data frames.
|
||||
- Fixed concurrent `getOrCreateCodexWebSocketConnection` callers (prewarm racing the first request) tearing down each other's in-flight handshake — closing a CONNECTING socket rejected the other caller with a fatal `websocket closed before open`, disabling WebSockets for the entire session. Callers now join the pending handshake.
|
||||
- Stopped the Codex connection-limit recovery from replaying a turn over SSE after a `toolcall_end` had already been delivered to the consumer (`canSafelyReplayWebsocketOverSse` guard was bypassed, re-emitting the same tool calls); the error now surfaces instead.
|
||||
- Extended the Codex whitespace-only argument-delta circuit breaker to `custom_tool_call_input.delta` frames, which counted as stream progress and could keep a degenerate response alive forever with no cap on buffer growth.
|
||||
- Fixed Codex stream failures during transport open reporting a synthetic request dump (empty URL/body) instead of the real request, and a `response.created` event resetting the recorded time-to-first-token.
|
||||
- Fixed the Codex WebSocket connect watchdog timer leaking (pinning the event loop for up to 10s) when the request signal aborted before or during the handshake.
|
||||
- Fixed OpenRouter-hosted Anthropic adaptive reasoning models (Claude Fable/Mythos 5 and Opus 4.6+) so the catalog exposes `xhigh`; Fable/Mythos and Opus 4.7+ requests now map user `high`/`xhigh` onto OpenRouter's Anthropic `xhigh`/`max` effort scale.
|
||||
- Fixed an unknown Anthropic `stop_reason` failing the whole turn after the response had fully streamed. `mapStopReason` threw on unrecognized values, and since the reason arrives on the trailing `message_delta` the error was unretryable — the live `model_context_window_exceeded` stop reason (default on Sonnet 4.5+) hit this path. It now maps to `length`, and any future unknown reason degrades to a logged anomaly plus a normal `stop` instead of an error.
|
||||
- Stopped clamping API-key Anthropic requests to Claude Code's 64k output cap. The `CLAUDE_CODE_MAX_OUTPUT_TOKENS` clamp exists to match the OAuth wire fingerprint, but `buildParams` applied it unconditionally, silently halving the output budget of 128k-output models (e.g. Opus 4.8) for API-key callers. OAuth requests keep the clamp.
|
||||
- Stopped a successful strict-tools fallback from shipping `errorMessage` on a `stopReason: "stop"` assistant message. After a grammar-too-large 400 triggered the non-strict retry, the original 400 text was kept on the final message even when the retry succeeded — consumers that treat `errorMessage` presence as failure (e.g. balance probes) misclassified the turn, and the stale text suppressed later refusal explanations. The fallback is now logged instead.
|
||||
- Fixed model-supplied `User-Agent` headers being silently dropped on non-OAuth Anthropic requests. `enforcedHeaderKeys` filtered the header out of `modelHeaders` in every branch but only the OAuth branch set one back; the Cloudflare-gateway, bearer-gateway, and `X-Api-Key` branches now forward the caller's value verbatim.
|
||||
- Stopped sending the `fast-mode-2026-02-01` beta header once a session has learned the endpoint+model rejects fast mode (`fastModeDisabled` provider state), matching the already-dropped `speed` param.
|
||||
- Stopped `buildAnthropicHeaders` defaulting API-key requests onto the full Claude Code OAuth beta list (`oauth-2025-04-20`, `claude-code-20250219`, …). The `claudeCodeBetas` default is now OAuth-gated, matching the streaming path — the web-search header builder was the only caller hitting the default, so API-key search requests now carry just their own betas (e.g. `web-search-2025-03-05`). An empty `anthropic-beta` header is omitted entirely instead of being sent as an empty string.
|
||||
- Fixed image-bearing `developer` messages being upgraded to mid-conversation `system` turns on Opus 4.8+/Fable/Mythos 5. System content is text-only on the wire, so a developer turn carrying image blocks in an upgrade-eligible position produced a 400; it now stays a `user` message.
|
||||
- Fixed a spliced reconnect's second envelope overwriting the completed Anthropic message: `message_delta` was not gated by the terminal-stop flag (content events and duplicate `message_start` were), so the splice's `stop_reason`/usage replaced the finished turn's — a `tool_use` turn could be relabeled `stop`, and the harness then never executed the streamed tool calls. Post-terminal deltas are now logged as envelope anomalies and skipped.
|
||||
- Fixed a `ping` arriving before `message_start` consuming the Anthropic first-event watchdog: the stall was then classified as a terminal mid-stream idle timeout instead of a retryable first-event timeout. Pings no longer count as the first item but still refresh the idle deadline once content is flowing.
|
||||
- Fixed Anthropic-compatible proxies that omit `usage`/`delta` objects from `message_start`/`message_delta`/`content_block_*` envelopes crashing the turn with an unretryable `TypeError`; the missing payloads now degrade to logged envelope anomalies like every other malformed-frame case.
|
||||
- Fixed `applyPromptCaching` placing `cache_control` on `thinking`/`redacted_thinking` blocks — Anthropic rejects that with a 400. A thinking-only assistant turn inside the trailing cache window (e.g. followed by the synthetic `Continue.` pad) no longer receives a breakpoint.
|
||||
- Fixed consecutive `assistant` params reaching the wire when an empty user/developer turn between two assistant turns was dropped by the converter (e.g. an empty "nudge" submission after a length-truncated reply); Anthropic 400s on non-alternating assistant turns, and the broken triple replayed on every subsequent request. A `user: "Continue."` separator is now inserted, mirroring the trailing-prefill fallback.
|
||||
- Fixed `supportsAdaptiveThinkingDisplay` misparsing bare dated Opus ids: `claude-opus-4-20250514` (Opus 4.0) parsed as minor `20250514` ≥ 4.7, which silently dropped the `interleaved-thinking-2025-05-14` beta for API-key Opus 4.0 requests.
|
||||
- Fixed `output_config.effort` shipping without the `effort-2025-11-24` beta on thinking-off requests against adaptive-only Claude models (the effort:"low" pin), and the mid-conversation `system` role shipping without `mid-conversation-system-2026-04-07` on API-key and OAuth-utility requests; both betas are now added whenever the request can carry the corresponding field.
|
||||
- Fixed GitHub Copilot anthropic-messages requests going out with no `Content-Type` and no `anthropic-version` header — the copilot branch builds its headers from scratch and Bun's fetch does not default `Content-Type` for string bodies. Both headers are now pinned to match every other branch.
|
||||
- Fixed Anthropic client/provider retry multiplication: with the first-event watchdog disabled (`PI_STREAM_FIRST_EVENT_TIMEOUT_MS=0`), the client's internal `maxRetries: 5` reactivated and stacked with the provider loop's 3 retries — up to 24 wire attempts with double backoff. The provider now pins per-request `maxRetries: 0` unconditionally.
|
||||
- Fixed `AnthropicMessagesClient` spreading `fetchOptions` after the core request fields, letting a caller-supplied `signal`/`method`/`body` silently disconnect the timeout controller or corrupt the request. Transport extras (TLS) still pass through; core fields now always win.
|
||||
- Fixed Foundry mTLS/CA material being cached for the process lifetime when the env vars point at files: the cache key now folds in the file mtime so on-disk certificate rotation takes effect.
|
||||
- Fixed the Claude Code fingerprint version drifting across surfaces: the usage endpoint (`claude-cli/2.1.160`) and OAuth bootstrap (`claude-code/2.1.160`) pinned a stale version while `/v1/messages` reported 2.1.165; both now derive from `claudeCodeVersion`.
|
||||
- Fixed a system prompt that merely *mentions* `x-anthropic-billing-header:` mid-text suppressing the entire Claude Code system-block injection (billing header, instruction, and cch attestation); the resumed-session guard now anchors with `startsWith`.
|
||||
- Fixed lone surrogates in cross-API tool-call arguments reaching Anthropic's strict UTF-8 validation: replayed OpenAI/Google-origin `tool_use.input` string leaves are now deep-sanitized with `toWellFormed()`, while same-API Anthropic arguments stay byte-identical to keep prompt-cache prefixes stable.
|
||||
- Bounded the many-image resize fan-out to 4 concurrent decodes (it previously decoded every oversized image at once, two encode pipelines each — multi-GB transient memory at the 20+-image threshold that activates the feature).
|
||||
- Fixed `mergeHeaders` merging case-sensitively on the Copilot/client-options path, where a miscased user-configured header (e.g. `authorization` next to the synthesized `Authorization`) survived as two keys that the `Headers` constructor joins comma-separated on the wire.
|
||||
- Hardened the Anthropic stream lifecycle: prologue failures (e.g. a malformed Copilot credential in `buildCopilotDynamicHeaders`) and error-finalization failures now surface as an `error` event instead of an unhandled rejection that left `stream.result()` hanging forever; the spurious "cch billing placeholder not patched" warning no longer fires when the placeholder only appears in user content.
|
||||
|
||||
## [15.10.9] - 2026-06-09
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-ai",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -4002,8 +4002,8 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
||||
return;
|
||||
}
|
||||
|
||||
const schemaVersion = this.#readAuthSchemaVersion() ?? this.#inferAuthSchemaVersion();
|
||||
const shouldWriteSchemaVersion = schemaVersion <= AUTH_SCHEMA_VERSION;
|
||||
const recordedVersion = this.#readAuthSchemaVersion();
|
||||
const schemaVersion = recordedVersion ?? this.#inferAuthSchemaVersion();
|
||||
if (schemaVersion > AUTH_SCHEMA_VERSION) {
|
||||
logger.warn("SqliteAuthCredentialStore schema version mismatch", {
|
||||
current: schemaVersion,
|
||||
@@ -4015,7 +4015,9 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
||||
|
||||
this.#createAuthCredentialIndexes();
|
||||
this.#backfillCredentialIdentityKeys();
|
||||
if (shouldWriteSchemaVersion) {
|
||||
// Rewriting an already-current version row is a no-op write transaction
|
||||
// on every boot; only persist when the recorded version actually changes.
|
||||
if (recordedVersion !== AUTH_SCHEMA_VERSION && schemaVersion <= AUTH_SCHEMA_VERSION) {
|
||||
this.#writeAuthSchemaVersion(AUTH_SCHEMA_VERSION);
|
||||
}
|
||||
}
|
||||
@@ -4171,9 +4173,13 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
|
||||
.all() as AuthRow[];
|
||||
if (rows.length === 0) return;
|
||||
|
||||
const updateIdentity = this.#db.prepare("UPDATE auth_credentials SET identity_key = ? WHERE id = ?");
|
||||
let updateIdentity: Statement | null = null;
|
||||
for (const row of rows) {
|
||||
const identityKey = resolveRowCredentialIdentityKey(row.provider, row);
|
||||
// Rows whose identity cannot be derived stay NULL; writing NULL over
|
||||
// NULL would just burn a write transaction on every boot.
|
||||
if (identityKey === null) continue;
|
||||
updateIdentity ??= this.#db.prepare("UPDATE auth_credentials SET identity_key = ? WHERE id = ?");
|
||||
updateIdentity.run(identityKey, row.id);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -374,6 +374,15 @@ function isFableOrMythos(kind: AnthropicKind): boolean {
|
||||
return kind === "fable" || kind === "mythos";
|
||||
}
|
||||
|
||||
function isOpenRouterAnthropicAdaptiveReasoningModel<TApi extends Api>(
|
||||
parsedModel: AnthropicModel,
|
||||
model: ApiModel<TApi>,
|
||||
): boolean {
|
||||
if (model.api !== "openai-completions") return false;
|
||||
if (model.provider !== "openrouter" && !model.baseUrl.includes("openrouter.ai")) return false;
|
||||
return isFableOrMythos(parsedModel.kind) || (parsedModel.kind === "opus" && semverGte(parsedModel.version, "4.6"));
|
||||
}
|
||||
|
||||
function anthropicModelHasRealXHighEffort<TApi extends Api>(model: ApiModel<TApi>): boolean {
|
||||
if (model.api !== "anthropic-messages") return false;
|
||||
const parsedModel = parseKnownModel(model.id);
|
||||
@@ -591,6 +600,9 @@ function inferAnthropicSupportedEfforts<TApi extends Api>(
|
||||
? DEFAULT_REASONING_EFFORTS_WITH_XHIGH
|
||||
: DEFAULT_REASONING_EFFORTS;
|
||||
}
|
||||
if (isOpenRouterAnthropicAdaptiveReasoningModel(parsedModel, model)) {
|
||||
return DEFAULT_REASONING_EFFORTS_WITH_XHIGH;
|
||||
}
|
||||
return inferFallbackEfforts(model);
|
||||
}
|
||||
|
||||
|
||||
@@ -48898,7 +48898,7 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"anthropic/claude-haiku-4.5": {
|
||||
@@ -49018,7 +49018,7 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"anthropic/claude-opus-4.6-fast": {
|
||||
@@ -49043,7 +49043,7 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"anthropic/claude-opus-4.7": {
|
||||
@@ -49068,7 +49068,7 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"anthropic/claude-opus-4.7-fast": {
|
||||
@@ -49093,7 +49093,7 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"anthropic/claude-opus-4.8": {
|
||||
@@ -49118,7 +49118,7 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"anthropic/claude-opus-4.8-fast": {
|
||||
@@ -49143,7 +49143,7 @@
|
||||
"thinking": {
|
||||
"mode": "effort",
|
||||
"minLevel": "minimal",
|
||||
"maxLevel": "high"
|
||||
"maxLevel": "xhigh"
|
||||
}
|
||||
},
|
||||
"anthropic/claude-sonnet-4": {
|
||||
|
||||
@@ -10,28 +10,36 @@ import type { Api, KnownProvider, Model, Usage } from "./types";
|
||||
*
|
||||
* For runtime-aware resolution, use `createModelManager()` / `resolveProviderModels()`.
|
||||
*/
|
||||
const modelRegistry: Map<string, Map<string, Model<Api>>> = new Map();
|
||||
for (const [provider, models] of Object.entries(MODELS)) {
|
||||
const providerModels = new Map<string, Model<Api>>();
|
||||
for (const [id, model] of Object.entries(models)) {
|
||||
providerModels.set(id, enrichModelThinking(model as Model<Api>));
|
||||
let modelRegistry: Map<string, Map<string, Model<Api>>> | undefined;
|
||||
|
||||
/** Build (once) and return the enriched bundled-model registry. Lazy: enrichment of ~12K models is deferred off module load. */
|
||||
function getModelRegistry(): Map<string, Map<string, Model<Api>>> {
|
||||
if (modelRegistry === undefined) {
|
||||
modelRegistry = new Map();
|
||||
for (const [provider, models] of Object.entries(MODELS)) {
|
||||
const providerModels = new Map<string, Model<Api>>();
|
||||
for (const [id, model] of Object.entries(models)) {
|
||||
providerModels.set(id, enrichModelThinking(model as Model<Api>));
|
||||
}
|
||||
modelRegistry.set(provider, providerModels);
|
||||
}
|
||||
}
|
||||
modelRegistry.set(provider, providerModels);
|
||||
return modelRegistry;
|
||||
}
|
||||
|
||||
export type GeneratedProvider = keyof typeof MODELS;
|
||||
|
||||
export function getBundledModel<TApi extends Api = Api>(provider: GeneratedProvider, modelId: string): Model<TApi> {
|
||||
const providerModels = modelRegistry.get(provider);
|
||||
const providerModels = getModelRegistry().get(provider);
|
||||
return providerModels?.get(modelId) as Model<TApi>;
|
||||
}
|
||||
|
||||
export function getBundledProviders(): KnownProvider[] {
|
||||
return Array.from(modelRegistry.keys()) as KnownProvider[];
|
||||
return Object.keys(MODELS) as KnownProvider[];
|
||||
}
|
||||
|
||||
export function getBundledModels(provider: GeneratedProvider): Model<Api>[] {
|
||||
const models = modelRegistry.get(provider);
|
||||
const models = getModelRegistry().get(provider);
|
||||
return models ? (Array.from(models.values()) as Model<Api>[]) : [];
|
||||
}
|
||||
|
||||
|
||||
@@ -32,7 +32,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
|
||||
import { appendRawHttpRequestDumpFor400, type RawHttpRequestDump, withHttpStatus } from "../utils/http-inspector";
|
||||
import { parseStreamingJson, parseStreamingJsonThrottled } from "../utils/json-parse";
|
||||
import { toolWireSchema } from "../utils/schema/wire";
|
||||
import { resolveAwsCredentials } from "./aws-credentials";
|
||||
import { invalidateAwsCredentialCache, resolveAwsCredentials } from "./aws-credentials";
|
||||
import { decodeEventStream } from "./aws-eventstream";
|
||||
import { signRequest } from "./aws-sigv4";
|
||||
import { transformMessages } from "./transform-messages";
|
||||
@@ -203,7 +203,10 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
||||
|
||||
try {
|
||||
const cacheRetention = resolveCacheRetention(options.cacheRetention);
|
||||
const toolConfig = convertToolConfig(context.tools, options.toolChoice);
|
||||
const historyHasToolBlocks = context.messages.some(
|
||||
m => m.role === "toolResult" || (m.role === "assistant" && m.content.some(b => b.type === "toolCall")),
|
||||
);
|
||||
const toolConfig = convertToolConfig(context.tools, options.toolChoice, historyHasToolBlocks);
|
||||
let additionalModelRequestFields = buildAdditionalModelRequestFields(model, options);
|
||||
|
||||
// Bedrock rejects thinking + forced tool_choice ("any" or specific tool).
|
||||
@@ -282,6 +285,11 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
if (!bearerToken && (response.status === 401 || response.status === 403)) {
|
||||
// Stale cached credentials (e.g. rotated session keys in ~/.aws/credentials) —
|
||||
// drop the cache entry so the next attempt re-resolves from scratch.
|
||||
invalidateAwsCredentialCache({ profile: options.profile, region });
|
||||
}
|
||||
const errBody = await response.text().catch(() => "");
|
||||
throw withHttpStatus(
|
||||
new Error(`Bedrock HTTP ${response.status}: ${errBody.slice(0, 1000)}`),
|
||||
@@ -340,6 +348,9 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
||||
case "messageStop": {
|
||||
const ev = payload as MessageStopEvent;
|
||||
output.stopReason = mapStopReason(ev.stopReason);
|
||||
if (output.stopReason === "error") {
|
||||
output.errorMessage = `Generation failed with stop reason: ${ev.stopReason ?? "unknown"}`;
|
||||
}
|
||||
break;
|
||||
}
|
||||
case "metadata": {
|
||||
@@ -740,8 +751,9 @@ function convertMessages(
|
||||
function convertToolConfig(
|
||||
tools: Tool[] | undefined,
|
||||
toolChoice: BedrockOptions["toolChoice"],
|
||||
historyHasToolBlocks: boolean,
|
||||
): WireToolConfig | undefined {
|
||||
if (!tools?.length || toolChoice === "none") return undefined;
|
||||
if (!tools?.length) return undefined;
|
||||
|
||||
const bedrockTools: WireToolSpec[] = tools.map(tool => ({
|
||||
toolSpec: {
|
||||
@@ -751,6 +763,13 @@ function convertToolConfig(
|
||||
},
|
||||
}));
|
||||
|
||||
// Bedrock rejects requests whose history contains toolUse/toolResult blocks without a
|
||||
// toolConfig. With prior tool use we must keep the tool specs and merely omit the choice
|
||||
// (there is no "none" choice on Converse); dropping toolConfig entirely would 400.
|
||||
if (toolChoice === "none") {
|
||||
return historyHasToolBlocks ? { tools: bedrockTools } : undefined;
|
||||
}
|
||||
|
||||
let bedrockToolChoice: WireToolChoice | undefined;
|
||||
switch (toolChoice) {
|
||||
case "auto":
|
||||
|
||||
@@ -43,7 +43,9 @@ export interface AnthropicRequestOptions {
|
||||
/**
|
||||
* Extra `RequestInit` fields merged into every fetch call. Bun extends
|
||||
* `RequestInit` with a `tls` option used for the Claude Code TLS profile and
|
||||
* Foundry mTLS.
|
||||
* Foundry mTLS. Core request fields (`method`, `headers`, `body`, `signal`)
|
||||
* are owned by the client and cannot be overridden from here — the timeout
|
||||
* controller's signal in particular must always win.
|
||||
*/
|
||||
export type AnthropicFetchOptions = RequestInit & {
|
||||
tls?: {
|
||||
@@ -121,7 +123,7 @@ function shouldRetryResponse(response: Response): boolean {
|
||||
}
|
||||
|
||||
/** Server-suggested delay (`retry-after-ms`, then `retry-after` seconds or HTTP date). */
|
||||
function retryDelayFromHeaders(headers: Headers | undefined): number | undefined {
|
||||
export function retryDelayFromHeaders(headers: Headers | undefined): number | undefined {
|
||||
if (!headers) return undefined;
|
||||
const retryAfterMs = headers.get("retry-after-ms");
|
||||
if (retryAfterMs) {
|
||||
@@ -288,11 +290,11 @@ export class AnthropicMessagesClient implements AnthropicMessagesClientLike {
|
||||
callerSignal?.addEventListener("abort", onAbort, { once: true });
|
||||
try {
|
||||
return await fetchFn(url, {
|
||||
...(this.#options.fetchOptions ?? {}),
|
||||
method: "POST",
|
||||
headers,
|
||||
body,
|
||||
signal: controller.signal,
|
||||
...(this.#options.fetchOptions ?? {}),
|
||||
});
|
||||
} catch (error) {
|
||||
if (timedOut && !callerSignal?.aborted) throw new AnthropicConnectionTimeoutError();
|
||||
|
||||
@@ -102,7 +102,17 @@ const toolResultBlockSchema = z.object({
|
||||
// natively understand (server_tool_use, web_search_tool_result, mcp_*,
|
||||
// container_upload, code_execution_*, document, …). The walker flattens these
|
||||
// to a text placeholder so legitimate Anthropic clients don't get rejected.
|
||||
const unknownContentBlockSchema = z.object({ type: z.string() }).loose();
|
||||
// Known `type` values are excluded so a malformed known block (e.g.
|
||||
// `{type:"text", text: 123}`) fails validation with a clean 400 instead of
|
||||
// slipping past the discriminated union and throwing a TypeError downstream.
|
||||
function unknownContentBlockSchema(knownTypes: readonly string[]) {
|
||||
const known = new Set(knownTypes);
|
||||
return z
|
||||
.object({
|
||||
type: z.string().refine(t => !known.has(t), { message: "malformed known content block" }),
|
||||
})
|
||||
.loose();
|
||||
}
|
||||
|
||||
// ─── System ────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -118,7 +128,7 @@ export const systemSchema = z.union([z.string(), z.array(systemBlockSchema)]).op
|
||||
|
||||
const userContentBlockSchema = z.union([
|
||||
z.discriminatedUnion("type", [textBlockSchema, imageBlockSchema, toolResultBlockSchema]),
|
||||
unknownContentBlockSchema,
|
||||
unknownContentBlockSchema(["text", "image", "tool_result"]),
|
||||
]);
|
||||
|
||||
const assistantContentBlockSchema = z.union([
|
||||
@@ -128,7 +138,7 @@ const assistantContentBlockSchema = z.union([
|
||||
redactedThinkingBlockSchema,
|
||||
toolUseBlockSchema,
|
||||
]),
|
||||
unknownContentBlockSchema,
|
||||
unknownContentBlockSchema(["text", "thinking", "redacted_thinking", "tool_use"]),
|
||||
]);
|
||||
|
||||
export const userMessageSchema = z.object({
|
||||
|
||||
@@ -488,17 +488,37 @@ interface OpenBlock {
|
||||
kind: BlockKind;
|
||||
}
|
||||
|
||||
// Keepalive cadence for the SSE encoder. Anthropic's API pings periodically;
|
||||
// without frames between message_start and the first content block (slow first
|
||||
// token) SDK first-event/idle watchdogs classify the stream as stalled.
|
||||
const STREAM_PING_INTERVAL_MS = 15_000;
|
||||
|
||||
const ZERO_WIRE_USAGE: Record<string, unknown> = {
|
||||
input_tokens: 0,
|
||||
output_tokens: 0,
|
||||
cache_read_input_tokens: 0,
|
||||
cache_creation_input_tokens: 0,
|
||||
};
|
||||
|
||||
export function encodeStream(
|
||||
events: AssistantMessageEventStream,
|
||||
requestedModelId: string,
|
||||
): ReadableStream<Uint8Array> {
|
||||
let pingTimer: NodeJS.Timeout | undefined;
|
||||
const stopPings = () => {
|
||||
if (pingTimer !== undefined) {
|
||||
clearInterval(pingTimer);
|
||||
pingTimer = undefined;
|
||||
}
|
||||
};
|
||||
return new ReadableStream<Uint8Array>({
|
||||
async start(controller) {
|
||||
const messageId = newMessageId();
|
||||
let started = false;
|
||||
let lastPartial: AssistantMessage | undefined;
|
||||
const open = new Map<number, OpenBlock>();
|
||||
|
||||
const ensureStart = (partial: AssistantMessage) => {
|
||||
const ensureStart = (partial: AssistantMessage | undefined) => {
|
||||
if (started) return;
|
||||
started = true;
|
||||
controller.enqueue(
|
||||
@@ -514,7 +534,7 @@ export function encodeStream(
|
||||
// TODO: same as encodeResponse — surface matched stop sequence
|
||||
// once pi-ai propagates it.
|
||||
stop_sequence: null,
|
||||
usage: encodeUsage(partial),
|
||||
usage: partial ? encodeUsage(partial) : ZERO_WIRE_USAGE,
|
||||
},
|
||||
}),
|
||||
);
|
||||
@@ -526,8 +546,18 @@ export function encodeStream(
|
||||
open.delete(index);
|
||||
};
|
||||
|
||||
pingTimer = setInterval(() => {
|
||||
try {
|
||||
controller.enqueue(sseFrame("ping", { type: "ping" }));
|
||||
} catch {
|
||||
// Controller already closed/errored (client gone); stop the timer.
|
||||
stopPings();
|
||||
}
|
||||
}, STREAM_PING_INTERVAL_MS);
|
||||
|
||||
try {
|
||||
for await (const ev of events) {
|
||||
if ("partial" in ev) lastPartial = ev.partial;
|
||||
switch (ev.type) {
|
||||
case "start":
|
||||
ensureStart(ev.partial);
|
||||
@@ -646,8 +676,18 @@ export function encodeStream(
|
||||
}
|
||||
}
|
||||
}
|
||||
// stream ended without explicit done; close gracefully
|
||||
// Stream ended without an explicit done: emit a complete envelope
|
||||
// (message_start + message_delta carrying a stop_reason) so strict
|
||||
// clients don't reject the response as a protocol error.
|
||||
ensureStart(lastPartial);
|
||||
for (const idx of [...open.keys()]) closeBlock(idx);
|
||||
controller.enqueue(
|
||||
sseFrame("message_delta", {
|
||||
type: "message_delta",
|
||||
delta: { stop_reason: "end_turn", stop_sequence: null },
|
||||
usage: lastPartial ? encodeUsage(lastPartial) : ZERO_WIRE_USAGE,
|
||||
}),
|
||||
);
|
||||
controller.enqueue(sseFrame("message_stop", { type: "message_stop" }));
|
||||
controller.close();
|
||||
} catch (err) {
|
||||
@@ -658,8 +698,13 @@ export function encodeStream(
|
||||
}),
|
||||
);
|
||||
controller.close();
|
||||
} finally {
|
||||
stopPings();
|
||||
}
|
||||
},
|
||||
cancel() {
|
||||
stopPings();
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -188,7 +188,8 @@ export type StopReason =
|
||||
| "tool_use"
|
||||
| "pause_turn"
|
||||
| "refusal"
|
||||
| "sensitive";
|
||||
| "sensitive"
|
||||
| "model_context_window_exceeded";
|
||||
|
||||
export type CacheCreation = {
|
||||
ephemeral_5m_input_tokens?: number | null;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -23,6 +23,7 @@ import * as fs from "node:fs";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { $env, isEnoent, logger } from "@oh-my-pi/pi-utils";
|
||||
import { raceWithSignal } from "../utils/abort";
|
||||
import type { AwsCredentials } from "./aws-sigv4";
|
||||
|
||||
export interface ResolvedCredentials extends AwsCredentials {
|
||||
@@ -39,6 +40,17 @@ export interface CredentialResolveOptions {
|
||||
}
|
||||
|
||||
const REFRESH_SKEW_MS = 60_000;
|
||||
/**
|
||||
* TTL for file-sourced credentials that carry a session token but no expiry.
|
||||
* Tools like aws-vault/saml2aws rewrite ~/.aws/credentials with short-lived STS
|
||||
* session keys; caching them forever serves stale creds after rotation.
|
||||
*/
|
||||
const FILE_SESSION_CREDS_TTL_MS = 5 * 60_000;
|
||||
/**
|
||||
* Bound for the detached (signal-free) shared resolution: a hung
|
||||
* credential_process/SSO/IMDS fetch must not pin the inflight slot forever.
|
||||
*/
|
||||
const SHARED_RESOLVE_TIMEOUT_MS = 30_000;
|
||||
|
||||
interface CacheEntry {
|
||||
creds: ResolvedCredentials;
|
||||
@@ -46,6 +58,7 @@ interface CacheEntry {
|
||||
}
|
||||
|
||||
const cache: Map<string, CacheEntry> = new Map();
|
||||
const inflight: Map<string, Promise<ResolvedCredentials>> = new Map();
|
||||
|
||||
export async function resolveAwsCredentials(opts: CredentialResolveOptions = {}): Promise<ResolvedCredentials> {
|
||||
const profile = opts.profile || $env.AWS_PROFILE || "default";
|
||||
@@ -55,9 +68,24 @@ export async function resolveAwsCredentials(opts: CredentialResolveOptions = {})
|
||||
const hit = cache.get(cacheKey);
|
||||
if (hit && hit.expiresAt - REFRESH_SKEW_MS > Date.now()) return hit.creds;
|
||||
|
||||
const creds = await resolveFresh(profile, region, opts.signal);
|
||||
cache.set(cacheKey, { creds, expiresAt: creds.expiresAt ?? Number.POSITIVE_INFINITY });
|
||||
return creds;
|
||||
// Single-flight: N concurrent cold calls must not each spawn credential_process/SSO/IMDS fetches.
|
||||
// The shared resolution is deliberately detached from any caller's signal — aborting one
|
||||
// request must not fail every waiter — and bounded by its own timeout instead; each caller
|
||||
// races its own signal against the shared promise.
|
||||
const existing = inflight.get(cacheKey);
|
||||
if (existing) return raceWithSignal(existing, opts.signal);
|
||||
|
||||
const promise = (async () => {
|
||||
try {
|
||||
const creds = await resolveFresh(profile, region, AbortSignal.timeout(SHARED_RESOLVE_TIMEOUT_MS));
|
||||
cache.set(cacheKey, { creds, expiresAt: creds.expiresAt ?? Number.POSITIVE_INFINITY });
|
||||
return creds;
|
||||
} finally {
|
||||
inflight.delete(cacheKey);
|
||||
}
|
||||
})();
|
||||
inflight.set(cacheKey, promise);
|
||||
return raceWithSignal(promise, opts.signal);
|
||||
}
|
||||
|
||||
async function resolveFresh(profile: string, region: string, signal?: AbortSignal): Promise<ResolvedCredentials> {
|
||||
@@ -157,7 +185,12 @@ async function readProfileCredentials(
|
||||
accessKeyId: merged.aws_access_key_id,
|
||||
secretAccessKey: merged.aws_secret_access_key,
|
||||
};
|
||||
if (merged.aws_session_token) out.sessionToken = merged.aws_session_token;
|
||||
if (merged.aws_session_token) {
|
||||
out.sessionToken = merged.aws_session_token;
|
||||
// Session-token creds in the credentials file are short-lived STS keys that
|
||||
// external tools rotate in place; cap the cache so rotations are picked up.
|
||||
out.expiresAt = Date.now() + FILE_SESSION_CREDS_TTL_MS;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -499,3 +532,13 @@ async function readImdsCredentials(parentSignal: AbortSignal | undefined): Promi
|
||||
export function clearAwsCredentialCache(): void {
|
||||
cache.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop the cache entry for one profile/region. Called by the Bedrock provider on
|
||||
* 401/403 responses so stale credentials are re-resolved instead of served until restart.
|
||||
*/
|
||||
export function invalidateAwsCredentialCache(opts: { profile?: string; region?: string } = {}): void {
|
||||
const profile = opts.profile || $env.AWS_PROFILE || "default";
|
||||
const region = opts.region || $env.AWS_REGION || $env.AWS_DEFAULT_REGION || "us-east-1";
|
||||
cache.delete(`${profile}\x00${region}`);
|
||||
}
|
||||
|
||||
@@ -161,6 +161,7 @@ export async function* decodeEventStream(source: ReadableStream<Uint8Array>): As
|
||||
// Single growable buffer; we slide a read cursor along it and compact when a
|
||||
// complete prefix has been consumed. Avoids per-message Uint8Array copies.
|
||||
let buf: Uint8Array<ArrayBufferLike> = new Uint8Array(0);
|
||||
let completed = false;
|
||||
try {
|
||||
while (true) {
|
||||
const { value, done } = await reader.read();
|
||||
@@ -179,7 +180,11 @@ export async function* decodeEventStream(source: ReadableStream<Uint8Array>): As
|
||||
if (done) break;
|
||||
}
|
||||
if (buf.length > 0) throw new Error("eventstream: truncated message at end of stream");
|
||||
completed = true;
|
||||
} finally {
|
||||
// On abnormal exit (consumer threw/broke, decode error) cancel the body so the
|
||||
// HTTP connection is released instead of draining until GC.
|
||||
if (!completed) await reader.cancel().catch(() => {});
|
||||
reader.releaseLock();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -28,9 +28,10 @@ import {
|
||||
iterateWithIdleTimeout,
|
||||
} from "../utils/idle-iterator";
|
||||
import { sanitizeSchemaForOpenAIResponses, toolWireSchema } from "../utils/schema";
|
||||
import { createSdkStreamRequestOptions } from "../utils/sdk-stream-timeout";
|
||||
import { notifyRawSseEvent } from "../utils/sse-debug";
|
||||
import { mapToOpenAIResponsesToolChoice } from "../utils/tool-choice";
|
||||
import { normalizeOpenAIResponsesPromptCacheKey, supportsDeveloperRole } from "./openai-responses";
|
||||
import { getOpenAIResponsesCacheSessionId, supportsDeveloperRole } from "./openai-responses";
|
||||
import {
|
||||
appendResponsesToolResultMessages,
|
||||
applyCommonResponsesSamplingParams,
|
||||
@@ -49,7 +50,7 @@ const DEFAULT_AZURE_API_VERSION = "v1";
|
||||
const AZURE_OPENAI_RESPONSES_FIRST_EVENT_TIMEOUT_MESSAGE =
|
||||
"Azure OpenAI responses stream timed out while waiting for the first event";
|
||||
|
||||
function parseDeploymentNameMap(value: string | undefined): Map<string, string> {
|
||||
export function parseAzureDeploymentNameMap(value: string | undefined): Map<string, string> {
|
||||
const map = new Map<string, string>();
|
||||
if (!value) return map;
|
||||
for (const entry of value.split(",")) {
|
||||
@@ -66,7 +67,7 @@ function resolveDeploymentName(model: Model<"azure-openai-responses">, options?:
|
||||
if (options?.azureDeploymentName) {
|
||||
return options.azureDeploymentName;
|
||||
}
|
||||
const mappedDeployment = parseDeploymentNameMap($env.AZURE_OPENAI_DEPLOYMENT_NAME_MAP).get(model.id);
|
||||
const mappedDeployment = parseAzureDeploymentNameMap($env.AZURE_OPENAI_DEPLOYMENT_NAME_MAP).get(model.id);
|
||||
return mappedDeployment ?? model.id;
|
||||
}
|
||||
|
||||
@@ -156,10 +157,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
||||
}
|
||||
let openaiStream: AsyncIterable<ResponseStreamEvent>;
|
||||
try {
|
||||
const requestOptions =
|
||||
requestTimeoutMs === undefined
|
||||
? { signal: requestSignal }
|
||||
: { signal: requestSignal, timeout: requestTimeoutMs };
|
||||
const requestOptions = createSdkStreamRequestOptions(requestSignal, requestTimeoutMs);
|
||||
openaiStream = await client.responses.create(params, requestOptions);
|
||||
} catch (error) {
|
||||
if (error instanceof OpenAIConnectionTimeoutError && !abortTracker.wasCallerAbort()) {
|
||||
@@ -306,7 +304,10 @@ function buildParams(
|
||||
model: deploymentName,
|
||||
input: messages,
|
||||
stream: true,
|
||||
prompt_cache_key: normalizeOpenAIResponsesPromptCacheKey(options?.promptCacheKey ?? options?.sessionId),
|
||||
prompt_cache_key: getOpenAIResponsesCacheSessionId(options),
|
||||
// Encrypted reasoning replay (applyResponsesReasoningParams) requires
|
||||
// stateless responses, matching the openai provider.
|
||||
store: false,
|
||||
};
|
||||
|
||||
applyCommonResponsesSamplingParams(params, options, model);
|
||||
@@ -332,6 +333,7 @@ function convertMessages(
|
||||
const messages: ResponseInput = [];
|
||||
const transformedMessages = transformMessages(context.messages, model, normalizeResponsesToolCallIdForTransform);
|
||||
const knownCallIds = new Set<string>();
|
||||
const customCallIds = new Set<string>();
|
||||
|
||||
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
||||
if (systemPrompts.length > 0) {
|
||||
@@ -351,11 +353,18 @@ function convertMessages(
|
||||
content: msg.role === "developer" && typeof msg.content === "string" ? msg.content.toWellFormed() : content,
|
||||
});
|
||||
} else if (msg.role === "assistant") {
|
||||
const outputItems = convertResponsesAssistantMessage(msg as AssistantMessage, model, msgIndex, knownCallIds);
|
||||
const outputItems = convertResponsesAssistantMessage(
|
||||
msg as AssistantMessage,
|
||||
model,
|
||||
msgIndex,
|
||||
knownCallIds,
|
||||
true,
|
||||
customCallIds,
|
||||
);
|
||||
if (outputItems.length === 0) continue;
|
||||
messages.push(...outputItems);
|
||||
} else if (msg.role === "toolResult") {
|
||||
appendResponsesToolResultMessages(messages, msg, model, strictResponsesPairing, knownCallIds);
|
||||
appendResponsesToolResultMessages(messages, msg, model, strictResponsesPairing, knownCallIds, customCallIds);
|
||||
}
|
||||
msgIndex++;
|
||||
}
|
||||
|
||||
@@ -17,6 +17,7 @@ import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { $envpos, isEnoent, logger } from "@oh-my-pi/pi-utils";
|
||||
import type { FetchImpl } from "../types";
|
||||
import { raceWithSignal } from "../utils/abort";
|
||||
|
||||
const OAUTH_TOKEN_URL = "https://oauth2.googleapis.com/token";
|
||||
const METADATA_TOKEN_URL = "http://metadata.google.internal/computeMetadata/v1/instance/service-accounts/default/token";
|
||||
@@ -258,6 +259,13 @@ async function resolveAccessTokenUncached(
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bound for the detached (signal-free) shared token resolution: a hung OAuth
|
||||
* exchange or metadata fetch must not pin the inflight slot forever — every
|
||||
* later call would await the stuck promise until process restart.
|
||||
*/
|
||||
const SHARED_TOKEN_RESOLVE_TIMEOUT_MS = 30_000;
|
||||
|
||||
/**
|
||||
* Returns a Bearer access token suitable for the `Authorization` header on Vertex AI calls.
|
||||
* The token is cached in module scope and refreshed `GOOGLE_VERTEX_REFRESH_SKEW_MS` ms before it expires.
|
||||
@@ -277,11 +285,17 @@ export async function getVertexAccessToken(options?: { signal?: AbortSignal; fet
|
||||
|
||||
const cacheKey = "vertex-adc";
|
||||
const existing = inflight.get(cacheKey);
|
||||
if (existing) return existing;
|
||||
if (existing) return raceWithSignal(existing, options?.signal);
|
||||
|
||||
// Deliberately resolve without any caller's signal: the in-flight promise is shared
|
||||
// by every concurrent caller, so aborting one request must not fail the whole batch.
|
||||
// Each caller races its own signal against the shared promise instead.
|
||||
const promise = (async () => {
|
||||
try {
|
||||
const { source, token } = await resolveAccessTokenUncached(options?.signal, fetchImpl);
|
||||
const { source, token } = await resolveAccessTokenUncached(
|
||||
AbortSignal.timeout(SHARED_TOKEN_RESOLVE_TIMEOUT_MS),
|
||||
fetchImpl,
|
||||
);
|
||||
const expiresAtMs = Date.now() + Math.max(0, token.expires_in * 1000);
|
||||
tokenCache.set(source, { token: token.access_token, expiresAtMs });
|
||||
logger.debug("vertex.adc acquired access token", { source, expiresInSec: token.expires_in });
|
||||
@@ -291,7 +305,7 @@ export async function getVertexAccessToken(options?: { signal?: AbortSignal; fet
|
||||
}
|
||||
})();
|
||||
inflight.set(cacheKey, promise);
|
||||
return promise;
|
||||
return raceWithSignal(promise, options?.signal);
|
||||
}
|
||||
|
||||
/** Test seam: clears every cached token. */
|
||||
|
||||
@@ -253,7 +253,10 @@ interface CloudCodeAssistResponseChunk {
|
||||
};
|
||||
modelVersion?: string;
|
||||
responseId?: string;
|
||||
promptFeedback?: { blockReason?: string; blockReasonMessage?: string };
|
||||
};
|
||||
/** In-band stream failure (quota, internal error) delivered as a final JSON event. */
|
||||
error?: { code?: number; message?: string; status?: string };
|
||||
traceId?: string;
|
||||
}
|
||||
|
||||
@@ -362,6 +365,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
const requestUrl = response.url;
|
||||
|
||||
let started = false;
|
||||
let sawFinishReason = false;
|
||||
const ensureStarted = () => {
|
||||
if (!started) {
|
||||
if (!firstTokenTime) firstTokenTime = Date.now();
|
||||
@@ -384,6 +388,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
output.errorMessage = undefined;
|
||||
output.timestamp = Date.now();
|
||||
started = false;
|
||||
sawFinishReason = false;
|
||||
};
|
||||
|
||||
const streamResponse = async (activeResponse: Response): Promise<boolean> => {
|
||||
@@ -401,8 +406,21 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
options?.signal,
|
||||
event => options?.onSseEvent?.({ event: event.event, data: event.data, raw: [...event.raw] }, model),
|
||||
)) {
|
||||
if (chunk.error) {
|
||||
const detail = chunk.error.message || chunk.error.status || "unknown error";
|
||||
const err = new Error(`Cloud Code Assist stream error: ${detail}`);
|
||||
throw typeof chunk.error.code === "number" && chunk.error.code >= 400
|
||||
? withHttpStatus(err, chunk.error.code)
|
||||
: err;
|
||||
}
|
||||
const responseData = chunk.response;
|
||||
if (!responseData) continue;
|
||||
if (!responseData.candidates?.length && responseData.promptFeedback?.blockReason) {
|
||||
const detail = responseData.promptFeedback.blockReasonMessage;
|
||||
throw new Error(
|
||||
`Request blocked by Google (${responseData.promptFeedback.blockReason})${detail ? `: ${detail}` : ""}`,
|
||||
);
|
||||
}
|
||||
|
||||
const candidate = responseData.candidates?.[0];
|
||||
if (candidate?.content?.parts) {
|
||||
@@ -463,7 +481,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
type: "toolCall",
|
||||
id: toolCallId,
|
||||
name: part.functionCall.name || "",
|
||||
arguments: part.functionCall.args as Record<string, unknown>,
|
||||
arguments: (part.functionCall.args ?? {}) as Record<string, unknown>,
|
||||
...(part.thoughtSignature && { thoughtSignature: part.thoughtSignature }),
|
||||
};
|
||||
|
||||
@@ -475,9 +493,17 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
}
|
||||
|
||||
if (candidate?.finishReason) {
|
||||
output.stopReason = mapStopReasonString(candidate.finishReason);
|
||||
if (output.content.some(b => b.type === "toolCall")) {
|
||||
sawFinishReason = true;
|
||||
const mapped = mapStopReasonString(candidate.finishReason);
|
||||
// Only let a trailing tool call upgrade benign finishes; error finishes
|
||||
// (SAFETY, MALFORMED_FUNCTION_CALL, ...) must surface even with tool calls present.
|
||||
if ((mapped === "stop" || mapped === "length") && output.content.some(b => b.type === "toolCall")) {
|
||||
output.stopReason = "toolUse";
|
||||
} else {
|
||||
output.stopReason = mapped;
|
||||
if (mapped === "error") {
|
||||
output.errorMessage = `Generation failed with finish reason: ${candidate.finishReason}`;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -568,6 +594,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
||||
throw new Error("Request was aborted");
|
||||
}
|
||||
|
||||
if (!sawFinishReason) {
|
||||
throw new Error(
|
||||
"Cloud Code Assist stream ended without a finish reason (connection dropped or response truncated)",
|
||||
);
|
||||
}
|
||||
|
||||
if (output.stopReason === "aborted" || output.stopReason === "error") {
|
||||
throw new Error(output.errorMessage ?? "An unknown error occurred");
|
||||
}
|
||||
|
||||
@@ -160,7 +160,19 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
|
||||
const transformedMessages = transformMessages(context.messages, model, normalizeToolCallId);
|
||||
|
||||
// Gemini < 3 image tool results go in a separate user turn, but parallel tool results must
|
||||
// stay a single contiguous functionResponse turn ("number of function response parts is not
|
||||
// equal to number of function call parts"). Buffer image turns and flush them only after the
|
||||
// merged functionResponse turn is complete.
|
||||
let pendingToolImageParts: Part[] = [];
|
||||
const flushPendingToolImages = () => {
|
||||
if (pendingToolImageParts.length === 0) return;
|
||||
contents.push({ role: "user", parts: pendingToolImageParts });
|
||||
pendingToolImageParts = [];
|
||||
};
|
||||
|
||||
for (const msg of transformedMessages) {
|
||||
if (msg.role !== "toolResult") flushPendingToolImages();
|
||||
if (msg.role === "user" || msg.role === "developer") {
|
||||
if (typeof msg.content === "string") {
|
||||
// Skip empty user messages
|
||||
@@ -314,15 +326,13 @@ export function convertMessages<T extends GoogleApiType>(model: Model<T>, contex
|
||||
});
|
||||
}
|
||||
|
||||
// For Gemini < 3, add images in a separate user message
|
||||
// For Gemini < 3, buffer images for a separate user message after the functionResponse turn
|
||||
if (hasImages && !modelSupportsMultimodalFunctionResponse) {
|
||||
contents.push({
|
||||
role: "user",
|
||||
parts: [{ text: "Tool result image:" }, ...imageParts],
|
||||
});
|
||||
pendingToolImageParts.push({ text: "Tool result image:" }, ...imageParts);
|
||||
}
|
||||
}
|
||||
}
|
||||
flushPendingToolImages();
|
||||
|
||||
return contents;
|
||||
}
|
||||
@@ -527,6 +537,7 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
|
||||
const blockIndex = () => blocks.length - 1;
|
||||
let currentBlock: TextContent | ThinkingContent | null = null;
|
||||
let firstTokenSeen = false;
|
||||
let sawFinishReason = false;
|
||||
|
||||
const flushCurrent = () => {
|
||||
if (!currentBlock) return;
|
||||
@@ -534,6 +545,19 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
|
||||
};
|
||||
|
||||
for await (const chunk of googleStream) {
|
||||
if (chunk.error) {
|
||||
const detail = chunk.error.message || chunk.error.status || "unknown error";
|
||||
const err = new Error(`Google API stream error: ${detail}`);
|
||||
throw typeof chunk.error.code === "number" && chunk.error.code >= 400
|
||||
? withHttpStatus(err, chunk.error.code)
|
||||
: err;
|
||||
}
|
||||
if (!chunk.candidates?.length && chunk.promptFeedback?.blockReason) {
|
||||
const detail = chunk.promptFeedback.blockReasonMessage;
|
||||
throw new Error(
|
||||
`Request blocked by Google (${chunk.promptFeedback.blockReason})${detail ? `: ${detail}` : ""}`,
|
||||
);
|
||||
}
|
||||
const candidate = chunk.candidates?.[0];
|
||||
if (candidate?.content?.parts) {
|
||||
for (const part of candidate.content.parts) {
|
||||
@@ -606,9 +630,17 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
|
||||
}
|
||||
|
||||
if (candidate?.finishReason) {
|
||||
output.stopReason = mapStopReason(candidate.finishReason);
|
||||
if (output.content.some(b => b.type === "toolCall")) {
|
||||
sawFinishReason = true;
|
||||
const mapped = mapStopReason(candidate.finishReason);
|
||||
// Only let a trailing tool call upgrade benign finishes; SAFETY/MALFORMED_FUNCTION_CALL
|
||||
// and friends must surface as errors even when earlier chunks carried valid tool calls.
|
||||
if ((mapped === "stop" || mapped === "length") && output.content.some(b => b.type === "toolCall")) {
|
||||
output.stopReason = "toolUse";
|
||||
} else {
|
||||
output.stopReason = mapped;
|
||||
if (mapped === "error") {
|
||||
output.errorMessage = `Generation failed with finish reason: ${candidate.finishReason}`;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -645,6 +677,10 @@ export async function consumeGoogleStream<T extends GoogleApiType>(args: {
|
||||
throw new Error("Request was aborted");
|
||||
}
|
||||
|
||||
if (!sawFinishReason) {
|
||||
throw new Error("Google API stream ended without a finish reason (connection dropped or response truncated)");
|
||||
}
|
||||
|
||||
if (output.stopReason === "aborted" || output.stopReason === "error") {
|
||||
throw new Error(output.errorMessage ?? "An unknown error occurred");
|
||||
}
|
||||
|
||||
@@ -157,11 +157,20 @@ export interface UsageMetadata {
|
||||
cachedContentTokenCount?: number;
|
||||
}
|
||||
|
||||
/** Prompt-level safety feedback; `blockReason` is set (with no candidates) when the prompt is blocked. */
|
||||
export interface PromptFeedback {
|
||||
blockReason?: string;
|
||||
blockReasonMessage?: string;
|
||||
[key: string]: unknown;
|
||||
}
|
||||
|
||||
/** Single SSE chunk's parsed JSON body. */
|
||||
export interface GenerateContentResponse {
|
||||
candidates?: Candidate[];
|
||||
usageMetadata?: UsageMetadata;
|
||||
modelVersion?: string;
|
||||
responseId?: string;
|
||||
promptFeedback?: Record<string, unknown>;
|
||||
promptFeedback?: PromptFeedback;
|
||||
/** In-band stream failure (quota, internal error) delivered as a final JSON event. */
|
||||
error?: { code?: number; message?: string; status?: string };
|
||||
}
|
||||
|
||||
@@ -145,6 +145,11 @@ export const assistantMessageSchema = z.object({
|
||||
role: z.literal("assistant"),
|
||||
content: baseContent.optional(),
|
||||
tool_calls: z.array(toolCallSchema).optional(),
|
||||
// DeepSeek-style reasoning channel. The gateway emits it on the way out
|
||||
// (encodeResponse/encodeStream); accept it back so thinking-mode
|
||||
// continuations replay the model's actual reasoning instead of a
|
||||
// synthesized placeholder.
|
||||
reasoning_content: z.string().nullish(),
|
||||
});
|
||||
|
||||
export const toolMessageSchema = z.object({
|
||||
|
||||
@@ -91,6 +91,7 @@ export function parseRequest(body: unknown, headers?: Headers): ParsedRequest {
|
||||
buildAssistantMessage(
|
||||
(m.content ?? undefined) as string | OpenAIChatContentPart[] | undefined,
|
||||
m.tool_calls,
|
||||
(m as { reasoning_content?: string | null }).reasoning_content ?? undefined,
|
||||
data.model,
|
||||
now,
|
||||
),
|
||||
@@ -227,10 +228,17 @@ function decodeDataUri(url: string): { data: string; mimeType: string } | undefi
|
||||
function buildAssistantMessage(
|
||||
content: string | OpenAIChatContentPart[] | undefined,
|
||||
toolCalls: OpenAIChatToolCall[] | undefined,
|
||||
reasoningContent: string | undefined,
|
||||
modelId: string,
|
||||
now: number,
|
||||
): AssistantMessage {
|
||||
const parts: AssistantMessage["content"] = [];
|
||||
if (reasoningContent !== undefined && reasoningContent.length > 0) {
|
||||
// Replayed reasoning channel. The signature names the wire field so
|
||||
// completions providers that demand exact `reasoning_content` replay
|
||||
// (DeepSeek/Kimi) echo the model's actual reasoning back verbatim.
|
||||
parts.push({ type: "thinking", thinking: reasoningContent, thinkingSignature: "reasoning_content" });
|
||||
}
|
||||
const text = stringifyContent(content);
|
||||
if (text.length > 0) parts.push({ type: "text", text });
|
||||
if (toolCalls) {
|
||||
@@ -529,6 +537,9 @@ export function encodeStream(
|
||||
async start(controller) {
|
||||
// contentIndex (from pi-ai events) -> tool_calls index on the wire.
|
||||
const toolIndexByContentIndex = new Map<number, number>();
|
||||
// wire index -> id/name emitted on the start chunk, to detect late-arriving
|
||||
// upstream id/name that needs a corrective chunk before the finish.
|
||||
const sentToolMeta = new Map<number, { id: string; name: string }>();
|
||||
let nextToolIndex = 0;
|
||||
let hasToolCalls = false;
|
||||
let finishReason: string = "stop";
|
||||
@@ -559,6 +570,7 @@ export function encodeStream(
|
||||
toolIndexByContentIndex.set(event.contentIndex, idx);
|
||||
const partial = event.partial.content[event.contentIndex];
|
||||
const call = partial && partial.type === "toolCall" ? partial : undefined;
|
||||
sentToolMeta.set(idx, { id: call?.id ?? "", name: call?.name ?? "" });
|
||||
writeSse(
|
||||
controller,
|
||||
baseChunk(
|
||||
@@ -588,6 +600,38 @@ export function encodeStream(
|
||||
break;
|
||||
}
|
||||
|
||||
case "toolcall_end": {
|
||||
const idx = toolIndexByContentIndex.get(event.contentIndex);
|
||||
if (idx === undefined) break;
|
||||
const sent = sentToolMeta.get(idx);
|
||||
if (sent === undefined) break;
|
||||
// Upstream completions providers can receive the real id/name in a
|
||||
// later chunk than toolcall_start. Emit a corrective chunk only when
|
||||
// the streamed value was empty: accumulating clients concatenate
|
||||
// string fields, so "" + value is the only safe correction.
|
||||
const correctId = sent.id === "" && event.toolCall.id !== "" ? event.toolCall.id : undefined;
|
||||
const correctName =
|
||||
sent.name === "" && event.toolCall.name !== "" ? event.toolCall.name : undefined;
|
||||
if (correctId !== undefined || correctName !== undefined) {
|
||||
writeSse(
|
||||
controller,
|
||||
baseChunk(
|
||||
{
|
||||
tool_calls: [
|
||||
{
|
||||
index: idx,
|
||||
...(correctId !== undefined ? { id: correctId } : {}),
|
||||
...(correctName !== undefined ? { function: { name: correctName } } : {}),
|
||||
},
|
||||
],
|
||||
},
|
||||
null,
|
||||
),
|
||||
);
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case "done":
|
||||
finishReason =
|
||||
event.reason === "toolUse"
|
||||
@@ -610,8 +654,8 @@ export function encodeStream(
|
||||
return;
|
||||
}
|
||||
|
||||
// Drop start / *_start / *_end — chat-completions wire only
|
||||
// surfaces deltas and the terminal finish_reason.
|
||||
// Drop start / *_start and text/thinking *_end — chat-completions
|
||||
// wire only surfaces deltas and the terminal finish_reason.
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -136,7 +136,6 @@ const CODEX_RETRYABLE_EVENT_MESSAGE =
|
||||
const CODEX_PROVIDER_SESSION_STATE_KEY = "openai-codex-responses";
|
||||
const X_CODEX_TURN_STATE_HEADER = "x-codex-turn-state";
|
||||
const X_MODELS_ETAG_HEADER = "x-models-etag";
|
||||
const X_REASONING_INCLUDED_HEADER = "x-reasoning-included";
|
||||
/** Connection-level websocket failures that should immediately fall back to SSE without retrying. */
|
||||
const CODEX_WEBSOCKET_FATAL_PATTERNS = ["websocket error:", "websocket closed before open", "connection timeout"];
|
||||
/** Max total time to spend retrying 429s with server-provided delays (5 minutes). */
|
||||
@@ -196,7 +195,6 @@ type CodexWebSocketSessionState = {
|
||||
canAppend: boolean;
|
||||
turnState?: string;
|
||||
modelsEtag?: string;
|
||||
reasoningIncluded?: boolean;
|
||||
connection?: CodexWebSocketConnection;
|
||||
lastTransport?: CodexTransport;
|
||||
fallbackCount: number;
|
||||
@@ -383,6 +381,7 @@ function isCodexWebSocketRetryableStreamError(error: unknown): boolean {
|
||||
message.includes("websocket ping failed") ||
|
||||
message.includes("websocket pong timeout") ||
|
||||
message.includes("websocket message queue exceeded") ||
|
||||
message.includes("websocket request already in progress") ||
|
||||
message.includes("idle timeout waiting for websocket") ||
|
||||
message.includes("timeout waiting for first websocket event") ||
|
||||
message.includes("syntaxerror") ||
|
||||
@@ -434,11 +433,6 @@ function updateCodexSessionMetadataFromHeaders(
|
||||
if (modelsEtag && modelsEtag.length > 0) {
|
||||
state.modelsEtag = modelsEtag;
|
||||
}
|
||||
const reasoningIncluded = resolvedHeaders.get(X_REASONING_INCLUDED_HEADER);
|
||||
if (reasoningIncluded !== null) {
|
||||
const normalized = reasoningIncluded.trim().toLowerCase();
|
||||
state.reasoningIncluded = normalized.length === 0 ? true : normalized !== "false";
|
||||
}
|
||||
}
|
||||
|
||||
function extractCodexWebSocketHandshakeHeaders(socket: Bun.WebSocket, openEvent?: Event): Headers | undefined {
|
||||
@@ -709,14 +703,14 @@ async function buildTransformedCodexRequestBody(
|
||||
): Promise<RequestBody> {
|
||||
const params: RequestBody = {
|
||||
model: model.id,
|
||||
input: [...convertMessages(model, context)],
|
||||
input: convertMessages(model, context),
|
||||
stream: true,
|
||||
prompt_cache_key: promptCacheKey,
|
||||
};
|
||||
|
||||
if (options?.maxTokens) {
|
||||
params.max_output_tokens = options.maxTokens;
|
||||
}
|
||||
// `maxTokens` is intentionally not forwarded: transformRequestBody strips
|
||||
// `max_output_tokens`/`max_completion_tokens` (the Codex backend rejects
|
||||
// caller-supplied output caps).
|
||||
if (options?.temperature !== undefined) {
|
||||
params.temperature = options.temperature;
|
||||
}
|
||||
@@ -766,7 +760,7 @@ async function buildTransformedCodexRequestBody(
|
||||
const developerMessages = systemPrompts.slice(1);
|
||||
const codexOptions: CodexRequestOptions = {
|
||||
reasoningEffort: options?.reasoning,
|
||||
reasoningSummary: options?.reasoningSummary ?? "auto",
|
||||
reasoningSummary: options?.reasoningSummary === undefined ? "auto" : options.reasoningSummary,
|
||||
textVerbosity: options?.textVerbosity,
|
||||
include: options?.include,
|
||||
};
|
||||
@@ -1065,12 +1059,7 @@ async function processCodexResponseStream(
|
||||
try {
|
||||
let firstTokenTime = context.firstTokenTime;
|
||||
for await (const rawEvent of runtime.eventStream) {
|
||||
firstTokenTime = handleCodexStreamEvent({
|
||||
...context,
|
||||
runtime,
|
||||
rawEvent,
|
||||
firstTokenTime,
|
||||
});
|
||||
firstTokenTime = handleCodexStreamEvent(context, runtime, rawEvent, firstTokenTime);
|
||||
if (runtime.sawTerminalEvent) break;
|
||||
}
|
||||
return { firstTokenTime };
|
||||
@@ -1083,21 +1072,15 @@ async function processCodexResponseStream(
|
||||
}
|
||||
}
|
||||
|
||||
function handleCodexStreamEvent(args: {
|
||||
model: Model<"openai-codex-responses">;
|
||||
output: AssistantMessage;
|
||||
stream: AssistantMessageEventStream;
|
||||
runtime: CodexStreamRuntime;
|
||||
rawEvent: Record<string, unknown>;
|
||||
firstTokenTime?: number;
|
||||
}): number | undefined {
|
||||
const { model, output, stream, runtime, rawEvent } = args;
|
||||
function handleCodexStreamEvent(
|
||||
context: CodexStreamProcessingContext,
|
||||
runtime: CodexStreamRuntime,
|
||||
rawEvent: Record<string, unknown>,
|
||||
firstTokenTime: number | undefined,
|
||||
): number | undefined {
|
||||
const { model, output, stream } = context;
|
||||
const eventType = typeof rawEvent.type === "string" ? rawEvent.type : "";
|
||||
if (!eventType) return args.firstTokenTime;
|
||||
|
||||
const blocks = output.content;
|
||||
const blockIndex = () => blocks.length - 1;
|
||||
let firstTokenTime = args.firstTokenTime;
|
||||
if (!eventType) return firstTokenTime;
|
||||
|
||||
if (eventType === "response.output_item.added") {
|
||||
resetWhitespaceToolCallArgumentsDelta(runtime);
|
||||
@@ -1109,7 +1092,7 @@ function handleCodexStreamEvent(args: {
|
||||
output.content.push(runtime.currentBlock);
|
||||
stream.push({
|
||||
type: getOutputBlockStartEventType(runtime.currentBlock),
|
||||
contentIndex: blockIndex(),
|
||||
contentIndex: output.content.length - 1,
|
||||
partial: output,
|
||||
});
|
||||
return firstTokenTime;
|
||||
@@ -1121,12 +1104,12 @@ function handleCodexStreamEvent(args: {
|
||||
}
|
||||
|
||||
if (eventType === "response.reasoning_summary_text.delta") {
|
||||
handleReasoningSummaryTextDelta(runtime.currentItem, runtime.currentBlock, rawEvent, stream, output, blockIndex);
|
||||
handleReasoningSummaryTextDelta(runtime.currentItem, runtime.currentBlock, rawEvent, stream, output);
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.reasoning_summary_part.done") {
|
||||
handleReasoningSummaryPartDone(runtime.currentItem, runtime.currentBlock, stream, output, blockIndex);
|
||||
handleReasoningSummaryPartDone(runtime.currentItem, runtime.currentBlock, stream, output);
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
@@ -1136,33 +1119,17 @@ function handleCodexStreamEvent(args: {
|
||||
}
|
||||
|
||||
if (eventType === "response.output_text.delta") {
|
||||
handleMessageTextDelta(
|
||||
runtime.currentItem,
|
||||
runtime.currentBlock,
|
||||
rawEvent,
|
||||
stream,
|
||||
output,
|
||||
blockIndex,
|
||||
"output_text",
|
||||
);
|
||||
handleMessageTextDelta(runtime.currentItem, runtime.currentBlock, rawEvent, stream, output, "output_text");
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.refusal.delta") {
|
||||
handleMessageTextDelta(
|
||||
runtime.currentItem,
|
||||
runtime.currentBlock,
|
||||
rawEvent,
|
||||
stream,
|
||||
output,
|
||||
blockIndex,
|
||||
"refusal",
|
||||
);
|
||||
handleMessageTextDelta(runtime.currentItem, runtime.currentBlock, rawEvent, stream, output, "refusal");
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.function_call_arguments.delta") {
|
||||
const interruption = handleToolCallArgumentsDelta(runtime, rawEvent, stream, output, blockIndex);
|
||||
const interruption = handleToolCallArgumentsDelta(runtime, rawEvent, stream, output);
|
||||
if (interruption) interruptWhitespaceToolCallArgumentsDelta(runtime, interruption);
|
||||
return firstTokenTime;
|
||||
}
|
||||
@@ -1174,23 +1141,26 @@ function handleCodexStreamEvent(args: {
|
||||
}
|
||||
|
||||
if (eventType === "response.custom_tool_call_input.delta") {
|
||||
handleCustomToolCallInputDelta(runtime.currentItem, runtime.currentBlock, rawEvent, stream, output, blockIndex);
|
||||
const interruption = handleCustomToolCallInputDelta(runtime, rawEvent, stream, output);
|
||||
if (interruption) interruptWhitespaceToolCallArgumentsDelta(runtime, interruption);
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.custom_tool_call_input.done") {
|
||||
resetWhitespaceToolCallArgumentsDelta(runtime);
|
||||
handleCustomToolCallInputDone(runtime.currentItem, runtime.currentBlock, rawEvent);
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.output_item.done") {
|
||||
resetWhitespaceToolCallArgumentsDelta(runtime);
|
||||
handleOutputItemDone(model, output, stream, runtime, rawEvent, blockIndex);
|
||||
handleOutputItemDone(model, output, stream, runtime, rawEvent);
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.created") {
|
||||
return handleResponseCreated(runtime, rawEvent);
|
||||
handleResponseCreated(runtime, rawEvent);
|
||||
return firstTokenTime;
|
||||
}
|
||||
|
||||
if (eventType === "response.completed" || eventType === "response.done" || eventType === "response.incomplete") {
|
||||
@@ -1255,7 +1225,6 @@ function handleReasoningSummaryTextDelta(
|
||||
rawEvent: Record<string, unknown>,
|
||||
stream: AssistantMessageEventStream,
|
||||
output: AssistantMessage,
|
||||
blockIndex: () => number,
|
||||
): void {
|
||||
if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
|
||||
currentItem.summary = currentItem.summary || [];
|
||||
@@ -1264,7 +1233,7 @@ function handleReasoningSummaryTextDelta(
|
||||
const delta = (rawEvent as { delta?: string }).delta || "";
|
||||
currentBlock.thinking += delta;
|
||||
lastPart.text += delta;
|
||||
stream.push({ type: "thinking_delta", contentIndex: blockIndex(), delta, partial: output });
|
||||
stream.push({ type: "thinking_delta", contentIndex: output.content.length - 1, delta, partial: output });
|
||||
}
|
||||
|
||||
function handleReasoningSummaryPartDone(
|
||||
@@ -1272,7 +1241,6 @@ function handleReasoningSummaryPartDone(
|
||||
currentBlock: CodexOutputBlock | null,
|
||||
stream: AssistantMessageEventStream,
|
||||
output: AssistantMessage,
|
||||
blockIndex: () => number,
|
||||
): void {
|
||||
if (currentItem?.type !== "reasoning" || currentBlock?.type !== "thinking") return;
|
||||
currentItem.summary = currentItem.summary || [];
|
||||
@@ -1280,7 +1248,7 @@ function handleReasoningSummaryPartDone(
|
||||
if (!lastPart) return;
|
||||
currentBlock.thinking += "\n\n";
|
||||
lastPart.text += "\n\n";
|
||||
stream.push({ type: "thinking_delta", contentIndex: blockIndex(), delta: "\n\n", partial: output });
|
||||
stream.push({ type: "thinking_delta", contentIndex: output.content.length - 1, delta: "\n\n", partial: output });
|
||||
}
|
||||
|
||||
function handleContentPartAdded(currentItem: CodexEventItem | null, rawEvent: Record<string, unknown>): void {
|
||||
@@ -1298,13 +1266,20 @@ function handleMessageTextDelta(
|
||||
rawEvent: Record<string, unknown>,
|
||||
stream: AssistantMessageEventStream,
|
||||
output: AssistantMessage,
|
||||
blockIndex: () => number,
|
||||
partType: "output_text" | "refusal",
|
||||
): void {
|
||||
if (currentItem?.type !== "message" || currentBlock?.type !== "text") return;
|
||||
if (!currentItem.content || currentItem.content.length === 0) return;
|
||||
const lastPart = currentItem.content[currentItem.content.length - 1];
|
||||
if (!lastPart || lastPart.type !== partType) return;
|
||||
currentItem.content = currentItem.content || [];
|
||||
let lastPart = currentItem.content[currentItem.content.length - 1];
|
||||
if (lastPart?.type !== partType) {
|
||||
// `content_part.added` never arrived (lossy proxy) — synthesize the part
|
||||
// so live text still streams instead of freezing until output_item.done.
|
||||
lastPart =
|
||||
partType === "output_text"
|
||||
? { type: "output_text", text: "", annotations: [] }
|
||||
: { type: "refusal", refusal: "" };
|
||||
currentItem.content.push(lastPart);
|
||||
}
|
||||
const delta = (rawEvent as { delta?: string }).delta || "";
|
||||
currentBlock.text += delta;
|
||||
if (lastPart.type === "output_text") {
|
||||
@@ -1312,7 +1287,7 @@ function handleMessageTextDelta(
|
||||
} else {
|
||||
lastPart.refusal += delta;
|
||||
}
|
||||
stream.push({ type: "text_delta", contentIndex: blockIndex(), delta, partial: output });
|
||||
stream.push({ type: "text_delta", contentIndex: output.content.length - 1, delta, partial: output });
|
||||
}
|
||||
|
||||
function handleToolCallArgumentsDelta(
|
||||
@@ -1320,21 +1295,24 @@ function handleToolCallArgumentsDelta(
|
||||
rawEvent: Record<string, unknown>,
|
||||
stream: AssistantMessageEventStream,
|
||||
output: AssistantMessage,
|
||||
blockIndex: () => number,
|
||||
): CodexWhitespaceToolCallArgumentsDeltaInterruption | undefined {
|
||||
const delta = (rawEvent as { delta?: string }).delta || "";
|
||||
// Observe BEFORE the item/block guard: degenerate whitespace frames can keep
|
||||
// arriving after the item closed (currentBlock detached) and still count as
|
||||
// progress for the idle watchdogs — dropping them unobserved would reopen
|
||||
// the infinite-loop hole the breaker exists for.
|
||||
const interruption = observeWhitespaceToolCallArgumentsDelta(runtime, rawEvent, delta);
|
||||
if (interruption) return interruption;
|
||||
const currentItem = runtime.currentItem;
|
||||
const currentBlock = runtime.currentBlock;
|
||||
if (currentItem?.type !== "function_call" || currentBlock?.type !== "toolCall") return undefined;
|
||||
const delta = (rawEvent as { delta?: string }).delta || "";
|
||||
const interruption = observeWhitespaceToolCallArgumentsDelta(runtime, rawEvent, delta);
|
||||
if (interruption) return interruption;
|
||||
currentBlock.partialJson += delta;
|
||||
const throttled = parseStreamingJsonThrottled(currentBlock.partialJson, currentBlock.lastParseLen ?? 0);
|
||||
if (throttled) {
|
||||
currentBlock.arguments = throttled.value;
|
||||
currentBlock.lastParseLen = throttled.parsedLen;
|
||||
}
|
||||
stream.push({ type: "toolcall_delta", contentIndex: blockIndex(), delta, partial: output });
|
||||
stream.push({ type: "toolcall_delta", contentIndex: output.content.length - 1, delta, partial: output });
|
||||
return undefined;
|
||||
}
|
||||
|
||||
@@ -1354,18 +1332,22 @@ function handleToolCallArgumentsDone(
|
||||
}
|
||||
|
||||
function handleCustomToolCallInputDelta(
|
||||
currentItem: CodexEventItem | null,
|
||||
currentBlock: CodexOutputBlock | null,
|
||||
runtime: CodexStreamRuntime,
|
||||
rawEvent: Record<string, unknown>,
|
||||
stream: AssistantMessageEventStream,
|
||||
output: AssistantMessage,
|
||||
blockIndex: () => number,
|
||||
): void {
|
||||
if (currentItem?.type !== "custom_tool_call" || currentBlock?.type !== "toolCall") return;
|
||||
): CodexWhitespaceToolCallArgumentsDeltaInterruption | undefined {
|
||||
const delta = (rawEvent as { delta?: string }).delta || "";
|
||||
// Observe BEFORE the item/block guard — see handleToolCallArgumentsDelta.
|
||||
const interruption = observeWhitespaceToolCallArgumentsDelta(runtime, rawEvent, delta);
|
||||
if (interruption) return interruption;
|
||||
const currentItem = runtime.currentItem;
|
||||
const currentBlock = runtime.currentBlock;
|
||||
if (currentItem?.type !== "custom_tool_call" || currentBlock?.type !== "toolCall") return undefined;
|
||||
currentBlock.partialJson += delta;
|
||||
currentBlock.arguments = { input: currentBlock.partialJson };
|
||||
stream.push({ type: "toolcall_delta", contentIndex: blockIndex(), delta, partial: output });
|
||||
(currentBlock.arguments as { input?: string }).input = currentBlock.partialJson;
|
||||
stream.push({ type: "toolcall_delta", contentIndex: output.content.length - 1, delta, partial: output });
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function handleCustomToolCallInputDone(
|
||||
@@ -1387,9 +1369,10 @@ function handleOutputItemDone(
|
||||
stream: AssistantMessageEventStream,
|
||||
runtime: CodexStreamRuntime,
|
||||
rawEvent: Record<string, unknown>,
|
||||
blockIndex: () => number,
|
||||
): void {
|
||||
const item = structuredCloneJSON(rawEvent.item) as CodexEventItem;
|
||||
const rawItem = rawEvent.item;
|
||||
if (!rawItem || typeof rawItem !== "object") return;
|
||||
const item = structuredCloneJSON(rawItem) as CodexEventItem;
|
||||
runtime.nativeOutputItems.push(item as unknown as Record<string, unknown>);
|
||||
|
||||
if (item.type === "reasoning" && runtime.currentBlock?.type === "thinking") {
|
||||
@@ -1397,7 +1380,7 @@ function handleOutputItemDone(
|
||||
runtime.currentBlock.thinkingSignature = JSON.stringify(item);
|
||||
stream.push({
|
||||
type: "thinking_end",
|
||||
contentIndex: blockIndex(),
|
||||
contentIndex: output.content.length - 1,
|
||||
content: runtime.currentBlock.thinking,
|
||||
partial: output,
|
||||
});
|
||||
@@ -1413,7 +1396,7 @@ function handleOutputItemDone(
|
||||
runtime.currentBlock.textSignature = encodeTextSignatureV1(item.id, phase);
|
||||
stream.push({
|
||||
type: "text_end",
|
||||
contentIndex: blockIndex(),
|
||||
contentIndex: output.content.length - 1,
|
||||
content: runtime.currentBlock.text,
|
||||
partial: output,
|
||||
});
|
||||
@@ -1434,9 +1417,12 @@ function handleOutputItemDone(
|
||||
runtime.currentBlock.arguments = toolCall.arguments;
|
||||
delete (runtime.currentBlock as { partialJson?: string }).partialJson;
|
||||
delete (runtime.currentBlock as { lastParseLen?: number }).lastParseLen;
|
||||
// Detach so a late/duplicate arguments.delta cannot append to the
|
||||
// finished block or trip the whitespace-loop guard against it.
|
||||
runtime.currentBlock = null;
|
||||
}
|
||||
runtime.canSafelyReplayWebsocketOverSse = false;
|
||||
stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output });
|
||||
stream.push({ type: "toolcall_end", contentIndex: output.content.length - 1, toolCall, partial: output });
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1452,21 +1438,25 @@ function handleOutputItemDone(
|
||||
arguments: { input: rawInput },
|
||||
customWireName: item.name,
|
||||
};
|
||||
if (runtime.currentBlock?.type === "toolCall") {
|
||||
runtime.currentBlock.arguments = { input: rawInput };
|
||||
delete (runtime.currentBlock as { partialJson?: string }).partialJson;
|
||||
runtime.currentBlock = null;
|
||||
}
|
||||
runtime.canSafelyReplayWebsocketOverSse = false;
|
||||
stream.push({ type: "toolcall_end", contentIndex: blockIndex(), toolCall, partial: output });
|
||||
stream.push({ type: "toolcall_end", contentIndex: output.content.length - 1, toolCall, partial: output });
|
||||
return;
|
||||
}
|
||||
|
||||
void model;
|
||||
}
|
||||
|
||||
function handleResponseCreated(runtime: CodexStreamRuntime, rawEvent: Record<string, unknown>): number | undefined {
|
||||
function handleResponseCreated(runtime: CodexStreamRuntime, rawEvent: Record<string, unknown>): void {
|
||||
const response = (rawEvent as { response?: { id?: string } }).response;
|
||||
const state = runtime.websocketState;
|
||||
if (runtime.transport === "websocket" && state && typeof response?.id === "string" && response.id.length > 0) {
|
||||
state.lastResponseId = response.id;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function handleResponseCompleted(
|
||||
@@ -1504,8 +1494,29 @@ function handleResponseCompleted(
|
||||
if (typeof response?.id === "string" && response.id.length > 0) {
|
||||
state.lastResponseId = response.id;
|
||||
state.lastResponseItems = stripInputItemIds(structuredCloneJSON(runtime.nativeOutputItems));
|
||||
state.canAppend = rawEvent.type === "response.done" || rawEvent.type === "response.completed";
|
||||
} else {
|
||||
// Without a response id the append baseline cannot be trusted.
|
||||
state.canAppend = false;
|
||||
}
|
||||
state.canAppend = rawEvent.type === "response.done" || rawEvent.type === "response.completed";
|
||||
}
|
||||
|
||||
// Finalize any toolCall block whose output_item.done never arrived: the
|
||||
// throttled delta parser may have left block.arguments stale, and the
|
||||
// toolUse promotion below would hand the agent incomplete arguments.
|
||||
// Mirrors the shared decoder's response.completed sweep; also strips the
|
||||
// transient partialJson/lastParseLen fields so they never persist.
|
||||
for (const block of output.content) {
|
||||
if (block.type !== "toolCall") continue;
|
||||
const pending = block as ToolCall & { partialJson?: string; lastParseLen?: number };
|
||||
if (pending.partialJson) {
|
||||
pending.arguments =
|
||||
pending.customWireName !== undefined
|
||||
? { input: pending.partialJson }
|
||||
: parseStreamingJson(pending.partialJson);
|
||||
}
|
||||
delete pending.partialJson;
|
||||
delete pending.lastParseLen;
|
||||
}
|
||||
|
||||
calculateCost(model, output.usage);
|
||||
@@ -1560,7 +1571,9 @@ function dropTrailingDegenerateToolCall(output: AssistantMessage, runtime: Codex
|
||||
* scratch — bounded by {@link CODEX_WHITESPACE_LOOP_RETRY_LIMIT}. Sampling
|
||||
* nondeterminism usually breaks the loop on a fresh attempt; once the budget is
|
||||
* exhausted the original error is surfaced (now without the junk tool call
|
||||
* polluting the message).
|
||||
* polluting the message). Replay is refused once a toolcall_end was already
|
||||
* delivered to the consumer (`canSafelyReplayWebsocketOverSse`) — it would
|
||||
* re-emit the same tool calls.
|
||||
*/
|
||||
async function tryRecoverCodexWhitespaceToolCallLoop(
|
||||
context: CodexStreamProcessingContext,
|
||||
@@ -1573,7 +1586,11 @@ async function tryRecoverCodexWhitespaceToolCallLoop(
|
||||
// Drop the half-built degenerate tool call whether or not we retry, so it
|
||||
// never reaches the caller's message.
|
||||
dropTrailingDegenerateToolCall(context.output, runtime);
|
||||
if (runtime.whitespaceLoopRetries >= CODEX_WHITESPACE_LOOP_RETRY_LIMIT || context.options?.signal?.aborted) {
|
||||
if (
|
||||
runtime.whitespaceLoopRetries >= CODEX_WHITESPACE_LOOP_RETRY_LIMIT ||
|
||||
!runtime.canSafelyReplayWebsocketOverSse ||
|
||||
context.options?.signal?.aborted
|
||||
) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1593,6 +1610,7 @@ async function tryRecoverCodexWhitespaceToolCallLoop(
|
||||
runtime.currentItem = null;
|
||||
runtime.currentBlock = null;
|
||||
runtime.sawTerminalEvent = false;
|
||||
runtime.nativeOutputItems.length = 0;
|
||||
resetWhitespaceToolCallArgumentsDelta(runtime);
|
||||
resetOutputState(context.output);
|
||||
context.firstTokenTime = undefined;
|
||||
@@ -1613,7 +1631,9 @@ async function tryRecoverCodexWhitespaceToolCallLoop(
|
||||
* Handles `websocket_connection_limit_reached` errors by closing the stale connection
|
||||
* and opening a fresh websocket. If content has already been emitted to the caller,
|
||||
* falls back to SSE replay (same as other WS failures) since we cannot safely
|
||||
* continue a partial response on a new connection.
|
||||
* continue a partial response on a new connection. If a tool call was already
|
||||
* delivered (`canSafelyReplayWebsocketOverSse` is false), the error surfaces
|
||||
* instead — replaying would re-emit the same tool calls.
|
||||
*/
|
||||
async function tryReconnectCodexWebSocketOnConnectionLimit(
|
||||
context: CodexStreamProcessingContext,
|
||||
@@ -1633,6 +1653,12 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
|
||||
websocketState.connection = undefined;
|
||||
resetCodexWebSocketAppendState(websocketState);
|
||||
|
||||
if (context.output.content.length > 0 && !runtime.canSafelyReplayWebsocketOverSse) {
|
||||
// A toolcall_end already reached the consumer; a full replay would emit
|
||||
// the same tool calls a second time. Let the error surface instead.
|
||||
return false;
|
||||
}
|
||||
|
||||
logCodexDebug("codex websocket connection limit reached, reconnecting", {
|
||||
hadContent: context.output.content.length > 0,
|
||||
retry: runtime.websocketStreamRetries,
|
||||
@@ -1641,7 +1667,6 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
|
||||
if (context.output.content.length > 0) {
|
||||
// Content already emitted to the caller — cannot safely continue on a new WS.
|
||||
// Reset and replay the full request over SSE.
|
||||
runtime.canSafelyReplayWebsocketOverSse = true;
|
||||
runtime.currentItem = null;
|
||||
runtime.currentBlock = null;
|
||||
runtime.nativeOutputItems.length = 0;
|
||||
@@ -1652,8 +1677,24 @@ async function tryReconnectCodexWebSocketOnConnectionLimit(
|
||||
return true;
|
||||
}
|
||||
|
||||
// No content emitted yet — reconnect over websocket.
|
||||
// No content emitted yet — clear accumulator state from the failed attempt
|
||||
// (blockless native items can exist even with empty content) and reconnect
|
||||
// over websocket, bounded by the shared retry budget: an account-scoped
|
||||
// limit can reject every fresh connection, and an unbounded loop would
|
||||
// hammer the endpoint with zero backoff.
|
||||
runtime.currentItem = null;
|
||||
runtime.currentBlock = null;
|
||||
runtime.nativeOutputItems.length = 0;
|
||||
context.firstTokenTime = undefined;
|
||||
if (runtime.websocketStreamRetries >= getCodexWebSocketRetryBudget()) {
|
||||
recordCodexWebSocketFailure(websocketState, true);
|
||||
await reopenCodexSseRuntimeStream(context, runtime, websocketState);
|
||||
return true;
|
||||
}
|
||||
runtime.websocketStreamRetries += 1;
|
||||
await scheduler.wait(getCodexWebSocketRetryDelayMs(runtime.websocketStreamRetries), {
|
||||
signal: context.requestSetup.requestSignal,
|
||||
});
|
||||
await reopenCodexWebSocketRuntimeStream(context, runtime, websocketState);
|
||||
return true;
|
||||
}
|
||||
@@ -1729,6 +1770,13 @@ async function tryReplayWebsocketFailureOverSse(
|
||||
|
||||
if (!activateFallback) {
|
||||
runtime.websocketStreamRetries += 1;
|
||||
// Full re-send on a fresh socket: clear accumulator state from the failed
|
||||
// attempt. Content is empty here, but blockless native items (e.g.
|
||||
// web_search_call) may already have accumulated.
|
||||
runtime.currentItem = null;
|
||||
runtime.currentBlock = null;
|
||||
runtime.nativeOutputItems.length = 0;
|
||||
context.firstTokenTime = undefined;
|
||||
await scheduler.wait(getCodexWebSocketRetryDelayMs(runtime.websocketStreamRetries), {
|
||||
signal: context.requestSetup.requestSignal,
|
||||
});
|
||||
@@ -1736,14 +1784,11 @@ async function tryReplayWebsocketFailureOverSse(
|
||||
return true;
|
||||
}
|
||||
|
||||
if (replayingBufferedOutputOverSse) {
|
||||
runtime.canSafelyReplayWebsocketOverSse = true;
|
||||
runtime.currentItem = null;
|
||||
runtime.currentBlock = null;
|
||||
runtime.nativeOutputItems.length = 0;
|
||||
resetOutputState(context.output);
|
||||
context.firstTokenTime = undefined;
|
||||
}
|
||||
runtime.currentItem = null;
|
||||
runtime.currentBlock = null;
|
||||
runtime.nativeOutputItems.length = 0;
|
||||
resetOutputState(context.output);
|
||||
context.firstTokenTime = undefined;
|
||||
|
||||
await reopenCodexSseRuntimeStream(context, runtime, state);
|
||||
return true;
|
||||
@@ -1780,6 +1825,7 @@ async function tryRetryCodexProviderError(
|
||||
runtime.currentItem = null;
|
||||
runtime.currentBlock = null;
|
||||
runtime.sawTerminalEvent = false;
|
||||
runtime.nativeOutputItems.length = 0;
|
||||
resetOutputState(context.output);
|
||||
context.firstTokenTime = undefined;
|
||||
await scheduler.wait(CODEX_RETRY_DELAY_MS * runtime.providerRetryAttempt, {
|
||||
@@ -1862,9 +1908,10 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
|
||||
const output = createAssistantOutput(model);
|
||||
const requestSetup = createRequestSetup(options);
|
||||
let processingContext: CodexStreamProcessingContext | undefined;
|
||||
let requestContext: CodexRequestContext | undefined;
|
||||
|
||||
try {
|
||||
const requestContext = await buildCodexRequestContext(model, context, options, output);
|
||||
requestContext = await buildCodexRequestContext(model, context, options, output);
|
||||
const initialTransport = await openInitialCodexEventStream(model, options, requestSetup, requestContext);
|
||||
const runtime = createCodexStreamRuntime({
|
||||
...initialTransport,
|
||||
@@ -1898,7 +1945,7 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
|
||||
stream,
|
||||
options,
|
||||
requestSetup,
|
||||
requestContext: {
|
||||
requestContext: requestContext ?? {
|
||||
apiKey: "",
|
||||
accountId: "",
|
||||
baseUrl: model.baseUrl || CODEX_BASE_URL,
|
||||
@@ -1916,8 +1963,19 @@ export const streamOpenAICodexResponses: StreamFunction<"openai-codex-responses"
|
||||
},
|
||||
startTime,
|
||||
} satisfies CodexStreamProcessingContext);
|
||||
const failure = await handleCodexStreamFailure(failureContext, error);
|
||||
stream.push({ type: "error", reason: failure.stopReason as "error" | "aborted", error: failure });
|
||||
try {
|
||||
const failure = await handleCodexStreamFailure(failureContext, error);
|
||||
stream.push({ type: "error", reason: failure.stopReason as "error" | "aborted", error: failure });
|
||||
} catch (failureError) {
|
||||
// Last resort — the failure handler itself threw (exotic error object or
|
||||
// request-dump formatting). Never leave the stream un-ended.
|
||||
logger.error("Codex stream failure handler threw", {
|
||||
error: failureError instanceof Error ? failureError.message : String(failureError),
|
||||
});
|
||||
output.stopReason = "error";
|
||||
output.errorMessage ??= error instanceof Error ? error.message : String(error);
|
||||
stream.push({ type: "error", reason: "error", error: output });
|
||||
}
|
||||
stream.end();
|
||||
}
|
||||
})();
|
||||
@@ -2032,13 +2090,18 @@ function resetCodexWebSocketAppendState(state: CodexWebSocketSessionState): void
|
||||
function resetCodexSessionMetadata(state: CodexWebSocketSessionState): void {
|
||||
state.turnState = undefined;
|
||||
state.modelsEtag = undefined;
|
||||
state.reasoningIncluded = undefined;
|
||||
}
|
||||
|
||||
function recordCodexWebSocketFailure(state: CodexWebSocketSessionState, activateFallback: boolean): void {
|
||||
resetCodexWebSocketAppendState(state);
|
||||
state.connection?.close("fallback");
|
||||
state.connection = undefined;
|
||||
// Never tear down a CONNECTING socket: it belongs to a concurrent caller's
|
||||
// in-flight handshake (prewarm/request race); closing it would reject that
|
||||
// caller with a fatal "websocket closed before open" and disable websockets
|
||||
// for the whole session.
|
||||
if (state.connection && !state.connection.isConnecting()) {
|
||||
state.connection.close("fallback");
|
||||
state.connection = undefined;
|
||||
}
|
||||
state.lastFallbackAt = Date.now();
|
||||
if (activateFallback && !state.disableWebsocket) {
|
||||
state.disableWebsocket = true;
|
||||
@@ -2269,6 +2332,11 @@ class CodexWebSocketConnection {
|
||||
return this.#socket?.readyState === WebSocket.OPEN;
|
||||
}
|
||||
|
||||
/** True while a handshake (possibly started by another caller) is still in flight. */
|
||||
isConnecting(): boolean {
|
||||
return this.#connectPromise !== undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Stricter variant of {@link isOpen} for the connection-pool reuse gate.
|
||||
* Refuses sockets that have been silent past {@link CODEX_WEBSOCKET_MAX_IDLE_REUSE_MS}.
|
||||
@@ -2324,10 +2392,18 @@ class CodexWebSocketConnection {
|
||||
this.#socket = socket;
|
||||
let settled = false;
|
||||
let timeout: NodeJS.Timeout | undefined;
|
||||
const clearPending = () => {
|
||||
if (timeout !== undefined) {
|
||||
clearTimeout(timeout);
|
||||
timeout = undefined;
|
||||
}
|
||||
if (signal) signal.removeEventListener("abort", onAbort);
|
||||
};
|
||||
const onAbort = () => {
|
||||
socket.close(1000, "aborted");
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
clearPending();
|
||||
reject(createCodexWebSocketTransportError("request was aborted"));
|
||||
}
|
||||
};
|
||||
@@ -2338,17 +2414,16 @@ class CodexWebSocketConnection {
|
||||
signal.addEventListener("abort", onAbort, { once: true });
|
||||
}
|
||||
}
|
||||
const clearPending = () => {
|
||||
if (timeout) clearTimeout(timeout);
|
||||
if (signal) signal.removeEventListener("abort", onAbort);
|
||||
};
|
||||
timeout = setTimeout(() => {
|
||||
socket.close(1000, "connect-timeout");
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
reject(createCodexWebSocketTransportError("connection timeout"));
|
||||
}
|
||||
}, CODEX_WEBSOCKET_CONNECT_TIMEOUT_MS);
|
||||
if (!settled) {
|
||||
timeout = setTimeout(() => {
|
||||
socket.close(1000, "connect-timeout");
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
clearPending();
|
||||
reject(createCodexWebSocketTransportError("connection timeout"));
|
||||
}
|
||||
}, CODEX_WEBSOCKET_CONNECT_TIMEOUT_MS);
|
||||
}
|
||||
|
||||
socket.onopen = event => {
|
||||
if (!settled) {
|
||||
@@ -2434,6 +2509,9 @@ class CodexWebSocketConnection {
|
||||
if (this.#activeRequest) {
|
||||
throw createCodexWebSocketTransportError("websocket request already in progress");
|
||||
}
|
||||
if (signal?.aborted) {
|
||||
throw createCodexWebSocketTransportError("request was aborted");
|
||||
}
|
||||
this.#activeRequest = true;
|
||||
this.#streamObserver = onSseEvent;
|
||||
// Drain any non-error frames left over from a prior request before sending.
|
||||
@@ -2451,13 +2529,7 @@ class CodexWebSocketConnection {
|
||||
this.close("aborted");
|
||||
this.#push(createCodexWebSocketTransportError("request was aborted"));
|
||||
};
|
||||
if (signal) {
|
||||
if (signal.aborted) {
|
||||
onAbort();
|
||||
} else {
|
||||
signal.addEventListener("abort", onAbort, { once: true });
|
||||
}
|
||||
}
|
||||
if (signal) signal.addEventListener("abort", onAbort, { once: true });
|
||||
|
||||
try {
|
||||
const debugSession = isRequestDebugEnabled()
|
||||
@@ -2475,8 +2547,13 @@ class CodexWebSocketConnection {
|
||||
|
||||
const requestPayload = JSON.stringify(request);
|
||||
notifyCodexWebSocketOutbound(onSseEvent, request, requestPayload);
|
||||
// Re-check liveness: the debug-session await above can outlive the socket.
|
||||
const socket = this.#socket;
|
||||
if (!socket || socket.readyState !== WebSocket.OPEN) {
|
||||
throw createCodexWebSocketTransportError("websocket connection is unavailable");
|
||||
}
|
||||
try {
|
||||
this.#socket.send(requestPayload);
|
||||
socket.send(requestPayload);
|
||||
} catch (error) {
|
||||
throw createCodexWebSocketTransportError(
|
||||
`websocket send failed: ${error instanceof Error ? error.message : String(error)}`,
|
||||
@@ -2695,9 +2772,11 @@ class CodexWebSocketConnection {
|
||||
|
||||
#push(item: Record<string, unknown> | Error | null): void {
|
||||
if (item instanceof Error) {
|
||||
if (!(this.#queue[0] instanceof Error)) {
|
||||
this.#queue.length = 0;
|
||||
}
|
||||
// Append after frames already received instead of wiping them: a queued
|
||||
// terminal event (e.g. `response.completed` followed by an eager server
|
||||
// close) must still reach the consumer rather than morph into a spurious
|
||||
// transport failure. `#dropStaleFrames` keeps errors across requests, so
|
||||
// the death signal still surfaces if the data frames go unconsumed.
|
||||
this.#queue.push(item);
|
||||
this.#wakeWaiters();
|
||||
return;
|
||||
@@ -2752,6 +2831,22 @@ async function getOrCreateCodexWebSocketConnection(
|
||||
signal?: AbortSignal,
|
||||
): Promise<CodexWebSocketConnection> {
|
||||
const headerRecord = headersToRecord(headers);
|
||||
// Join an in-flight handshake instead of tearing it down: closing a
|
||||
// CONNECTING socket rejects the concurrent caller (prewarm racing the first
|
||||
// request) with a fatal "websocket closed before open", which would disable
|
||||
// websockets for the entire session.
|
||||
// Bounded re-join: a fresh handshake may have been started by yet another
|
||||
// caller while we awaited the previous one.
|
||||
for (let joinAttempt = 0; joinAttempt < 3; joinAttempt += 1) {
|
||||
const pending = state.connection;
|
||||
if (!pending || pending.isOpen() || !pending.isConnecting()) break;
|
||||
try {
|
||||
await pending.connect(signal);
|
||||
} catch {
|
||||
// The handshake owner surfaces its own failure; re-evaluate below
|
||||
// (state.connection may have been replaced or cleared).
|
||||
}
|
||||
}
|
||||
if (state.connection?.isOpen()) {
|
||||
if (!state.connection.matchesAuth(headerRecord)) {
|
||||
state.connection.close("token-refresh");
|
||||
@@ -2819,7 +2914,6 @@ async function openCodexSseEventStream(
|
||||
contentType: response.headers.get("content-type") || null,
|
||||
cfRay: response.headers.get("cf-ray") || null,
|
||||
});
|
||||
updateCodexSessionMetadataFromHeaders(state, response.headers);
|
||||
if (!response.ok) {
|
||||
const info = await parseCodexError(response);
|
||||
const error = new Error(info.friendlyMessage || info.message);
|
||||
@@ -2827,6 +2921,7 @@ async function openCodexSseEventStream(
|
||||
(error as { headers?: Headers; status?: number }).status = response.status;
|
||||
throw error;
|
||||
}
|
||||
updateCodexSessionMetadataFromHeaders(state, response.headers);
|
||||
if (!response.body) {
|
||||
throw new Error("No response body");
|
||||
}
|
||||
@@ -2876,6 +2971,7 @@ function createCodexHeaders(
|
||||
} else {
|
||||
headers.delete(OPENAI_HEADERS.CONVERSATION_ID);
|
||||
headers.delete(OPENAI_HEADERS.SESSION_ID);
|
||||
headers.delete("x-client-request-id");
|
||||
}
|
||||
if (state?.turnState) {
|
||||
headers.set(X_CODEX_TURN_STATE_HEADER, state.turnState);
|
||||
@@ -2914,6 +3010,7 @@ function redactHeaders(headers: Headers): Record<string, string> {
|
||||
lower.includes("account") ||
|
||||
lower.includes("session") ||
|
||||
lower.includes("conversation") ||
|
||||
lower === "x-client-request-id" ||
|
||||
lower === "cookie"
|
||||
) {
|
||||
redacted[key] = "[redacted]";
|
||||
@@ -2993,11 +3090,13 @@ function convertMessages(model: Model<"openai-codex-responses">, context: Contex
|
||||
|
||||
if (msg.role === "assistant") {
|
||||
const assistantMsg = msg as AssistantMessage;
|
||||
const providerPayload = getOpenAIResponsesHistoryPayload(
|
||||
assistantMsg.providerPayload,
|
||||
model.provider,
|
||||
assistantMsg.provider,
|
||||
);
|
||||
// Native items are model-bound (reasoning carries encrypted content
|
||||
// minted by the producing model); after a mid-session model switch fall
|
||||
// back to block re-encode, which strips foreign signatures.
|
||||
const providerPayload =
|
||||
assistantMsg.api === model.api && assistantMsg.model === model.id
|
||||
? getOpenAIResponsesHistoryPayload(assistantMsg.providerPayload, model.provider, assistantMsg.provider)
|
||||
: undefined;
|
||||
const historyItems = providerPayload?.items as Array<ResponseInput[number]> | undefined;
|
||||
if (historyItems) {
|
||||
for (const item of historyItems) {
|
||||
@@ -3150,7 +3249,9 @@ function isRetryableCodexFailureEvent(rawEvent: Record<string, unknown>): boolea
|
||||
}
|
||||
|
||||
function createCodexProviderStreamError(rawEvent: Record<string, unknown>): CodexProviderStreamError {
|
||||
const code = getString(rawEvent.code) ?? "";
|
||||
const response = asRecord(rawEvent.response);
|
||||
const nestedError = asRecord(rawEvent.error) ?? (response ? asRecord(response.error) : null);
|
||||
const code = getString(rawEvent.code) ?? getString(nestedError?.code) ?? getString(nestedError?.type) ?? "";
|
||||
const message = getString(rawEvent.message) ?? "";
|
||||
const formattedMessage =
|
||||
typeof rawEvent.type === "string" && rawEvent.type === "error"
|
||||
|
||||
@@ -105,8 +105,8 @@ function orphanFunctionOutputToMessage(item: InputItem, callId: string): InputIt
|
||||
* Repair both halves of unpaired tool exchanges so the Responses input grammar
|
||||
* stays valid — the API rejects either orphan with a 400:
|
||||
*
|
||||
* - `function_call_output` with no matching `function_call` → folded into an
|
||||
* assistant message (`400 No tool call found for function call output …`).
|
||||
* - `function_call_output` / `custom_tool_call_output` with no matching call →
|
||||
* folded into an assistant message (`400 No tool call found for … output`).
|
||||
* Regression of #472 / #1351.
|
||||
* - `function_call` / `custom_tool_call` with no matching `*_output` → a
|
||||
* placeholder output is synthesized immediately after the call
|
||||
@@ -131,7 +131,11 @@ function repairToolCallPairs(input: InputItem[]): InputItem[] {
|
||||
for (const item of input) {
|
||||
const callId = typeof item.call_id === "string" ? item.call_id : undefined;
|
||||
|
||||
if (item.type === "function_call_output" && callId !== undefined && !callIds.has(callId)) {
|
||||
if (
|
||||
(item.type === "function_call_output" || item.type === "custom_tool_call_output") &&
|
||||
callId !== undefined &&
|
||||
!callIds.has(callId)
|
||||
) {
|
||||
repaired.push(orphanFunctionOutputToMessage(item, callId));
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -47,6 +47,35 @@ function detectStrictModeSupport(provider: string, baseUrl: string): boolean {
|
||||
);
|
||||
}
|
||||
|
||||
function getOpenRouterAnthropicReasoningEffortMap(
|
||||
modelId: string,
|
||||
): Partial<Record<OpenAIReasoningEffort, string>> | undefined {
|
||||
const match = /(?:^|\/)claude-(opus|fable|mythos)-(\d{1,2})(?:[.-](\d{1,2}))?/.exec(modelId);
|
||||
if (!match) return undefined;
|
||||
|
||||
const kind = match[1];
|
||||
const major = Number(match[2]);
|
||||
const minor = Number(match[3] ?? 0);
|
||||
const isFableOrMythos = kind === "fable" || kind === "mythos";
|
||||
const isOpusAdaptive = kind === "opus" && (major > 4 || (major === 4 && minor >= 6));
|
||||
if (!isFableOrMythos && !isOpusAdaptive) return undefined;
|
||||
|
||||
const hasRealXHigh = isFableOrMythos || major > 4 || (major === 4 && minor >= 7);
|
||||
if (hasRealXHigh) {
|
||||
return {
|
||||
minimal: "low",
|
||||
low: "medium",
|
||||
medium: "high",
|
||||
high: "xhigh",
|
||||
xhigh: "max",
|
||||
};
|
||||
}
|
||||
return {
|
||||
minimal: "low",
|
||||
xhigh: "max",
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect compatibility settings from provider and baseUrl for known providers.
|
||||
* Provider takes precedence over URL-based detection since it's explicitly configured.
|
||||
@@ -175,6 +204,9 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
isCopilotHost ||
|
||||
isZenmuxHost);
|
||||
|
||||
const openRouterAnthropicReasoningEffortMap = isOpenRouter
|
||||
? getOpenRouterAnthropicReasoningEffortMap(lowerId)
|
||||
: undefined;
|
||||
const reasoningEffortMap: NonNullable<OpenAICompat["reasoningEffortMap"]> =
|
||||
provider === "groq" && model.id === "qwen/qwen3-32b"
|
||||
? ({
|
||||
@@ -192,13 +224,15 @@ export function detectOpenAICompat(model: Model<"openai-completions">, resolvedB
|
||||
high: "high",
|
||||
xhigh: "max",
|
||||
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
||||
: isFireworks
|
||||
? ({
|
||||
// Fireworks' OpenAI-compatible endpoint rejects OpenAI's
|
||||
// `minimal` literal but accepts `none` for the lowest setting.
|
||||
minimal: "none",
|
||||
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
||||
: {};
|
||||
: openRouterAnthropicReasoningEffortMap
|
||||
? openRouterAnthropicReasoningEffortMap
|
||||
: isFireworks
|
||||
? ({
|
||||
// Fireworks' OpenAI-compatible endpoint rejects OpenAI's
|
||||
// `minimal` literal but accepts `none` for the lowest setting.
|
||||
minimal: "none",
|
||||
} satisfies Partial<Record<OpenAIReasoningEffort, string>>)
|
||||
: {};
|
||||
|
||||
return {
|
||||
supportsStore: !isNonStandard,
|
||||
|
||||
@@ -67,6 +67,7 @@ import {
|
||||
type StreamMarkupHealingEvent,
|
||||
} from "../utils/stream-markup-healing";
|
||||
import { isForcedToolChoice, mapToOpenAICompletionsToolChoice } from "../utils/tool-choice";
|
||||
import { parseAzureDeploymentNameMap } from "./azure-openai-responses";
|
||||
import {
|
||||
buildCopilotDynamicHeaders,
|
||||
hasCopilotVisionInput,
|
||||
@@ -460,6 +461,10 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
const { requestAbortController, requestSignal } = abortTracker;
|
||||
const onSseEvent = options?.onSseEvent;
|
||||
const rawSseObserver = onSseEvent ? (event: RawSseEvent) => onSseEvent(event, model) : undefined;
|
||||
// Assigned once the block helpers exist (they are scoped to the `try`);
|
||||
// the catch handler uses it to close any open blocks before emitting the
|
||||
// terminal error so both exit paths obey the same block lifecycle.
|
||||
let finishOpenBlocksOnError: () => void = () => {};
|
||||
|
||||
try {
|
||||
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
||||
@@ -634,13 +639,21 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
}
|
||||
finishToolCallBlock(block);
|
||||
};
|
||||
finishOpenBlocksOnError = () => {
|
||||
if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
|
||||
finishPendingToolCallBlocks();
|
||||
};
|
||||
const appendText = (
|
||||
message: AssistantMessage,
|
||||
eventStream: AssistantMessageEventStream,
|
||||
text: string,
|
||||
): void => {
|
||||
if (currentBlock?.type !== "text") {
|
||||
finishCurrentBlock(currentBlock);
|
||||
// Leave toolCall blocks pending across text transitions: chunks after
|
||||
// the first typically carry only `index`, so a finished (de-registered)
|
||||
// call would be reborn as a nameless phantom block when its arguments
|
||||
// resume. The stream-end sweep finalizes pending calls.
|
||||
if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
|
||||
currentBlock = { type: "text", text: "" };
|
||||
message.content.push(currentBlock);
|
||||
eventStream.push({ type: "text_start", contentIndex: blockIndex(currentBlock), partial: message });
|
||||
@@ -663,7 +676,9 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
currentBlock?.type !== "thinking" ||
|
||||
(signature !== undefined && currentBlock.thinkingSignature !== signature)
|
||||
) {
|
||||
finishCurrentBlock(currentBlock);
|
||||
// Same as appendText: leave toolCall blocks pending so index-only
|
||||
// continuation deltas can still find them.
|
||||
if (currentBlock?.type !== "toolCall") finishCurrentBlock(currentBlock);
|
||||
currentBlock = { type: "thinking", thinking: "", thinkingSignature: signature };
|
||||
message.content.push(currentBlock);
|
||||
eventStream.push({
|
||||
@@ -896,6 +911,11 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
partial: output,
|
||||
});
|
||||
} else {
|
||||
// Resuming a pending call after interleaved text/thinking:
|
||||
// close the text/thinking block we drifted into.
|
||||
if (currentBlock !== block && currentBlock && currentBlock.type !== "toolCall") {
|
||||
finishCurrentBlock(currentBlock);
|
||||
}
|
||||
currentBlock = block;
|
||||
if (streamIndex !== undefined && block.streamIndex === undefined) {
|
||||
block.streamIndex = streamIndex;
|
||||
@@ -1037,6 +1057,12 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
||||
stream.push({ type: "done", reason: output.stopReason, message: output });
|
||||
stream.end();
|
||||
} catch (error) {
|
||||
// Close open blocks first so consumers tracking text_/thinking_/toolcall_
|
||||
// lifecycles never see orphaned starts on the error path. Best-effort: a
|
||||
// throw here must not prevent the terminal error event below.
|
||||
try {
|
||||
finishOpenBlocksOnError();
|
||||
} catch {}
|
||||
for (const block of output.content) delete (block as any).index;
|
||||
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
||||
output.stopReason = abortTracker.wasCallerAbort() ? "aborted" : "error";
|
||||
@@ -1129,7 +1155,11 @@ async function createClient(
|
||||
if (baseUrl?.includes(".openai.azure.com")) {
|
||||
const apiVersion = $env.AZURE_OPENAI_API_VERSION || "2024-10-21";
|
||||
if (!baseUrl.includes("/deployments/")) {
|
||||
baseUrl = `${baseUrl}/deployments/${model.id}`;
|
||||
// Honor AZURE_OPENAI_DEPLOYMENT_NAME_MAP like the responses provider:
|
||||
// deployment names routinely differ from catalog model ids.
|
||||
const deploymentName =
|
||||
parseAzureDeploymentNameMap($env.AZURE_OPENAI_DEPLOYMENT_NAME_MAP).get(model.id) ?? model.id;
|
||||
baseUrl = `${baseUrl}/deployments/${deploymentName}`;
|
||||
}
|
||||
azureDefaultQuery = { "api-version": apiVersion };
|
||||
}
|
||||
@@ -1738,12 +1768,12 @@ export function convertMessages(
|
||||
if (compat.requiresThinkingAsText) {
|
||||
// Convert thinking blocks to plain text (no tags to avoid model mimicking them)
|
||||
const thinkingText = nonEmptyThinkingBlocks.map(b => b.thinking).join("\n\n");
|
||||
const textContent = assistantMsg.content as Array<{ type: "text"; text: string }> | null;
|
||||
if (textContent) {
|
||||
textContent.unshift({ type: "text", text: thinkingText });
|
||||
} else {
|
||||
assistantMsg.content = [{ type: "text", text: thinkingText }];
|
||||
}
|
||||
// `content` is a plain string at this point (set above) or null —
|
||||
// never an array. Prepend the thinking text to the string form.
|
||||
assistantMsg.content =
|
||||
typeof assistantMsg.content === "string" && assistantMsg.content.length > 0
|
||||
? `${thinkingText}\n\n${assistantMsg.content}`
|
||||
: thinkingText;
|
||||
} else if (compat.requiresReasoningContentForToolCalls) {
|
||||
// Use the streamed signature when the backend accepts whichever
|
||||
// recognized field name was emitted (allowsSynthetic=true). Backends
|
||||
|
||||
@@ -97,12 +97,17 @@ const assistantMessageItemSchema = z.object({
|
||||
content: z.union([z.string(), z.array(outputContentBlockSchema)]).optional(),
|
||||
});
|
||||
|
||||
const reasoningItemSchema = z.object({
|
||||
type: z.literal("reasoning"),
|
||||
id: z.string().optional(),
|
||||
summary: z.array(summaryTextSchema).optional(),
|
||||
content: z.array(reasoningTextSchema).optional(),
|
||||
});
|
||||
const reasoningItemSchema = z
|
||||
.object({
|
||||
type: z.literal("reasoning"),
|
||||
id: z.string().optional(),
|
||||
summary: z.array(summaryTextSchema).optional(),
|
||||
content: z.array(reasoningTextSchema).optional(),
|
||||
})
|
||||
// Loose: unknown keys like `encrypted_content` must survive the parse —
|
||||
// the outbound encoder replays them verbatim (buildReasoningItem spreads
|
||||
// the persisted item to preserve encrypted reasoning round-trips).
|
||||
.loose();
|
||||
|
||||
const functionCallItemSchema = z.object({
|
||||
type: z.literal("function_call"),
|
||||
|
||||
@@ -573,6 +573,17 @@ function reasoningItemId(part: ThinkingContent): string {
|
||||
return makeReasoningId();
|
||||
}
|
||||
|
||||
/**
|
||||
* pi-ai responses providers mint composite `"{call_id}|{item_id}"` tool-call
|
||||
* ids ({@link encodeResponsesToolCallId}). Only the call_id half belongs on
|
||||
* the wire: third-party clients validate the `call_id` charset
|
||||
* (`^[a-zA-Z0-9_-]+$`) or echo it to other backends, and `|` fails both.
|
||||
*/
|
||||
function wireCallId(id: string): string {
|
||||
const sep = id.indexOf("|");
|
||||
return sep >= 0 ? id.slice(0, sep) : id;
|
||||
}
|
||||
|
||||
/**
|
||||
* Walk the assistant content array and group consecutive TextContent into a
|
||||
* single message item; each ThinkingContent / ToolCall is its own item.
|
||||
@@ -609,7 +620,7 @@ function buildOutputItems(message: AssistantMessage): OutputItem[] {
|
||||
out.push({
|
||||
type: "custom_tool_call",
|
||||
id: part.thoughtSignature ?? makeCustomCallId(),
|
||||
call_id: part.id,
|
||||
call_id: wireCallId(part.id),
|
||||
name: part.customWireName,
|
||||
input: rawInput,
|
||||
status: "completed",
|
||||
@@ -618,7 +629,7 @@ function buildOutputItems(message: AssistantMessage): OutputItem[] {
|
||||
out.push({
|
||||
type: "function_call",
|
||||
id: part.thoughtSignature ?? makeFuncCallId(),
|
||||
call_id: part.id,
|
||||
call_id: wireCallId(part.id),
|
||||
name: part.name,
|
||||
arguments: JSON.stringify(part.arguments ?? {}),
|
||||
status: "completed",
|
||||
@@ -801,7 +812,7 @@ export function encodeStream(
|
||||
: undefined;
|
||||
const isCustom = customWireName !== undefined;
|
||||
const itemId = tc?.thoughtSignature ?? (isCustom ? makeCustomCallId() : makeFuncCallId());
|
||||
const callId = tc?.id ?? "";
|
||||
const callId = wireCallId(tc?.id ?? "");
|
||||
const name = customWireName ?? tc?.name ?? "";
|
||||
const item = isCustom
|
||||
? {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
||||
import { logger, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
||||
import type OpenAI from "openai";
|
||||
import type {
|
||||
ResponseCustomToolCall,
|
||||
@@ -49,6 +49,7 @@ export const OPENAI_RESPONSES_PROGRESS_EVENT_TYPES: ReadonlySet<string> = new Se
|
||||
"response.custom_tool_call_input.done",
|
||||
"response.output_item.done",
|
||||
"response.completed",
|
||||
"response.incomplete",
|
||||
"response.failed",
|
||||
"error",
|
||||
]);
|
||||
@@ -310,6 +311,7 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
||||
customCallIds?: Set<string>,
|
||||
): ResponseInput {
|
||||
const outputItems: ResponseInput = [];
|
||||
let unsignedTextBlocks = 0;
|
||||
const isDifferentModel =
|
||||
assistantMsg.model !== model.id && assistantMsg.provider === model.provider && assistantMsg.api === model.api;
|
||||
|
||||
@@ -319,7 +321,12 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
||||
continue;
|
||||
}
|
||||
if (block.thinkingSignature) {
|
||||
outputItems.push(JSON.parse(block.thinkingSignature) as ResponseReasoningItem);
|
||||
try {
|
||||
outputItems.push(JSON.parse(block.thinkingSignature) as ResponseReasoningItem);
|
||||
} catch {
|
||||
// Legacy/corrupt persisted signature — skip the reasoning item
|
||||
// rather than failing the whole request build.
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
@@ -328,7 +335,10 @@ export function convertResponsesAssistantMessage<TApi extends Api>(
|
||||
const parsedSignature = parseTextSignature(block.textSignature);
|
||||
let msgId = parsedSignature?.id;
|
||||
if (!msgId) {
|
||||
msgId = `msg_${msgIndex}`;
|
||||
// Distinct ids per unsigned block: several text blocks in one message
|
||||
// (cross-provider replay downgrades thinking → text) must not share an id.
|
||||
msgId = unsignedTextBlocks === 0 ? `msg_${msgIndex}` : `msg_${msgIndex}_${unsignedTextBlocks}`;
|
||||
unsignedTextBlocks += 1;
|
||||
} else if (msgId.length > 64) {
|
||||
msgId = `msg_${Bun.hash(msgId).toString(36)}`;
|
||||
}
|
||||
@@ -393,10 +403,6 @@ export function appendResponsesToolResultMessages<TApi extends Api>(
|
||||
const hasImages = toolResult.content.some((block): block is ImageContent => block.type === "image");
|
||||
const omittedImages = hasImages && !supportsImages;
|
||||
const normalized = normalizeResponsesToolCallId(toolResult.toolCallId);
|
||||
if (strictResponsesPairing && !knownCallIds.has(normalized.callId)) {
|
||||
return;
|
||||
}
|
||||
|
||||
const output = (
|
||||
omittedImages
|
||||
? joinTextWithImagePlaceholder(textResult, true)
|
||||
@@ -404,6 +410,19 @@ export function appendResponsesToolResultMessages<TApi extends Api>(
|
||||
? textResult
|
||||
: "(see attached image)"
|
||||
).toWellFormed();
|
||||
if (strictResponsesPairing && !knownCallIds.has(normalized.callId)) {
|
||||
// Strict backends (Azure, Copilot) reject unpaired outputs outright, but
|
||||
// silently dropping the result loses information the model needs. Fold it
|
||||
// into an assistant note instead (same shape as repairOrphanResponsesToolOutputs).
|
||||
const limit = 16_000;
|
||||
const noteText = output.length > limit ? `${output.slice(0, limit)}\n...[truncated]` : output;
|
||||
messages.push({
|
||||
type: "message",
|
||||
role: "assistant",
|
||||
content: `[Orphan ${toolResult.toolName || "tool"} result; call_id=${normalized.callId}]: ${noteText}`,
|
||||
} as ResponseInput[number]);
|
||||
return;
|
||||
}
|
||||
if (customCallIds?.has(normalized.callId)) {
|
||||
messages.push({
|
||||
type: "custom_tool_call_output",
|
||||
@@ -645,32 +664,42 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
} else if (event.type === "response.output_text.delta") {
|
||||
const entry = lookupOpenItem(event);
|
||||
if (entry?.item.type === "message" && entry.block.type === "text") {
|
||||
const lastPart = entry.item.content?.[entry.item.content.length - 1];
|
||||
if (lastPart?.type === "output_text") {
|
||||
entry.block.text += event.delta;
|
||||
lastPart.text += event.delta;
|
||||
stream.push({
|
||||
type: "text_delta",
|
||||
contentIndex: contentIndexOf(entry.block),
|
||||
delta: event.delta,
|
||||
partial: output,
|
||||
});
|
||||
entry.item.content = entry.item.content || [];
|
||||
let lastPart = entry.item.content[entry.item.content.length - 1];
|
||||
if (lastPart?.type !== "output_text") {
|
||||
// `content_part.added` never arrived (lossy proxy) — synthesize the
|
||||
// part so live text still streams instead of freezing until the
|
||||
// item's output_item.done recovers the final text.
|
||||
lastPart = { type: "output_text", text: "", annotations: [] };
|
||||
entry.item.content.push(lastPart);
|
||||
}
|
||||
entry.block.text += event.delta;
|
||||
lastPart.text += event.delta;
|
||||
stream.push({
|
||||
type: "text_delta",
|
||||
contentIndex: contentIndexOf(entry.block),
|
||||
delta: event.delta,
|
||||
partial: output,
|
||||
});
|
||||
}
|
||||
} else if (event.type === "response.refusal.delta") {
|
||||
const entry = lookupOpenItem(event);
|
||||
if (entry?.item.type === "message" && entry.block.type === "text") {
|
||||
const lastPart = entry.item.content?.[entry.item.content.length - 1];
|
||||
if (lastPart?.type === "refusal") {
|
||||
entry.block.text += event.delta;
|
||||
lastPart.refusal += event.delta;
|
||||
stream.push({
|
||||
type: "text_delta",
|
||||
contentIndex: contentIndexOf(entry.block),
|
||||
delta: event.delta,
|
||||
partial: output,
|
||||
});
|
||||
entry.item.content = entry.item.content || [];
|
||||
let lastPart = entry.item.content[entry.item.content.length - 1];
|
||||
if (lastPart?.type !== "refusal") {
|
||||
// Same lossy-proxy hardening as the output_text branch above.
|
||||
lastPart = { type: "refusal", refusal: "" };
|
||||
entry.item.content.push(lastPart);
|
||||
}
|
||||
entry.block.text += event.delta;
|
||||
lastPart.refusal += event.delta;
|
||||
stream.push({
|
||||
type: "text_delta",
|
||||
contentIndex: contentIndexOf(entry.block),
|
||||
delta: event.delta,
|
||||
partial: output,
|
||||
});
|
||||
}
|
||||
} else if (event.type === "response.function_call_arguments.delta") {
|
||||
const entry = lookupOpenFunctionCallItem(event);
|
||||
@@ -732,9 +761,15 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
: item.content?.[0]?.type === "reasoning_text"
|
||||
? (item.content[0].text ?? "")
|
||||
: "";
|
||||
const reasoningBlock = output.content.find(
|
||||
b => b.type === "thinking" && (b as ThinkingContent).itemId === item.id,
|
||||
) as ThinkingContent | undefined;
|
||||
// Prefer the routed entry; the bare itemId find misroutes when ids are
|
||||
// absent (`undefined === undefined` matches the FIRST thinking block) and
|
||||
// misses entirely when the done-event id drifts from the added-event id.
|
||||
const reasoningBlock =
|
||||
entry?.block.type === "thinking"
|
||||
? entry.block
|
||||
: (output.content.find(b => b.type === "thinking" && (b as ThinkingContent).itemId === item.id) as
|
||||
| ThinkingContent
|
||||
| undefined);
|
||||
if (reasoningBlock) {
|
||||
reasoningBlock.thinking = thinking;
|
||||
reasoningBlock.thinkingSignature = JSON.stringify(item);
|
||||
@@ -746,18 +781,25 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
});
|
||||
}
|
||||
closeOpenItem(event.output_index, item.id, entry);
|
||||
} else if (item.type === "message" && entry?.block.type === "text") {
|
||||
const block = entry.block;
|
||||
block.text = item.content
|
||||
} else if (item.type === "message") {
|
||||
const block = entry?.block.type === "text" ? entry.block : undefined;
|
||||
const text = item.content
|
||||
.map(part => (part.type === "output_text" ? (part.text ?? "") : (part.refusal ?? "")))
|
||||
.join("");
|
||||
block.textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined);
|
||||
stream.push({
|
||||
type: "text_end",
|
||||
contentIndex: contentIndexOf(block),
|
||||
content: block.text,
|
||||
partial: output,
|
||||
});
|
||||
const textSignature = encodeTextSignatureV1(item.id, item.phase ?? undefined);
|
||||
let contentIndex: number;
|
||||
if (block) {
|
||||
block.text = text;
|
||||
block.textSignature = textSignature;
|
||||
contentIndex = contentIndexOf(block);
|
||||
} else {
|
||||
// `output_item.added` never arrived (lossy proxy) — synthesize the
|
||||
// block so the final message still carries the authoritative text.
|
||||
const synthesized: TextContent = { type: "text", text, textSignature };
|
||||
output.content.push(synthesized);
|
||||
contentIndex = output.content.length - 1;
|
||||
}
|
||||
stream.push({ type: "text_end", contentIndex, content: text, partial: output });
|
||||
closeOpenItem(event.output_index, item.id, entry);
|
||||
} else if (item.type === "function_call") {
|
||||
const block = entry?.block.type === "toolCall" ? entry.block : undefined;
|
||||
@@ -772,6 +814,7 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
name: item.name,
|
||||
arguments: args,
|
||||
};
|
||||
let contentIndex: number;
|
||||
if (block) {
|
||||
// Persist the authoritative final args on the stored block. The
|
||||
// throttled delta parser may have skipped the last partial parse,
|
||||
@@ -781,8 +824,14 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
delete (block as { partialJson?: string }).partialJson;
|
||||
delete (block as { lastParseLen?: number }).lastParseLen;
|
||||
delete (block as { argumentsDone?: boolean }).argumentsDone;
|
||||
contentIndex = contentIndexOf(block);
|
||||
} else {
|
||||
// `output_item.added` never arrived (lossy proxy) — synthesize the
|
||||
// block so the final message carries the call the consumer was told
|
||||
// completed (the agent loop executes tools from message.content).
|
||||
output.content.push(toolCall);
|
||||
contentIndex = output.content.length - 1;
|
||||
}
|
||||
const contentIndex = block ? contentIndexOf(block) : output.content.length - 1;
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id);
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
} else if (item.type === "custom_tool_call") {
|
||||
@@ -795,12 +844,39 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
arguments: { input: rawInput },
|
||||
customWireName: item.name,
|
||||
};
|
||||
const contentIndex = block ? contentIndexOf(block) : output.content.length - 1;
|
||||
let contentIndex: number;
|
||||
if (block) {
|
||||
// Persist the final input on the stored block and drop the transient
|
||||
// accumulation buffer, mirroring the function_call branch above.
|
||||
block.arguments = { input: rawInput };
|
||||
delete (block as { partialJson?: string }).partialJson;
|
||||
delete (block as { lastParseLen?: number }).lastParseLen;
|
||||
contentIndex = contentIndexOf(block);
|
||||
} else {
|
||||
output.content.push(toolCall);
|
||||
contentIndex = output.content.length - 1;
|
||||
}
|
||||
closeOpenItem(event.output_index, item.id, entry, item.call_id);
|
||||
stream.push({ type: "toolcall_end", contentIndex, toolCall, partial: output });
|
||||
}
|
||||
} else if (event.type === "response.completed") {
|
||||
} else if (event.type === "response.completed" || event.type === "response.incomplete") {
|
||||
const response = event.response;
|
||||
// Finalize any toolCall block whose output_item.done never arrived: the
|
||||
// throttled delta parser may have left block.arguments stale, and the
|
||||
// toolUse override below would hand the agent incomplete arguments.
|
||||
for (const open of openItemsInOrder) {
|
||||
if (open.block.type !== "toolCall") continue;
|
||||
const block = open.block;
|
||||
if (block.partialJson && !block.argumentsDone) {
|
||||
block.arguments =
|
||||
open.item.type === "custom_tool_call"
|
||||
? { input: block.partialJson }
|
||||
: parseStreamingJson(block.partialJson);
|
||||
}
|
||||
delete (block as { partialJson?: string }).partialJson;
|
||||
delete (block as { lastParseLen?: number }).lastParseLen;
|
||||
delete (block as { argumentsDone?: boolean }).argumentsDone;
|
||||
}
|
||||
if (response?.id) {
|
||||
output.responseId = response.id;
|
||||
}
|
||||
@@ -820,12 +896,19 @@ export async function processResponsesStream<TApi extends Api>(
|
||||
: "Unknown error (no error details in response)";
|
||||
throw new Error(message);
|
||||
}
|
||||
if (response?.status === "incomplete" && response.incomplete_details?.reason === "content_filter") {
|
||||
// A content-filtered turn is a failure, not a token-cap truncation —
|
||||
// mapping it to "length" would route the agent loop into "shorten your
|
||||
// output" recovery against a filtered prompt.
|
||||
throw new Error("incomplete: content_filter");
|
||||
}
|
||||
if (output.content.some(block => block.type === "toolCall") && output.stopReason === "stop") {
|
||||
output.stopReason = "toolUse";
|
||||
}
|
||||
} else if (event.type === "error") {
|
||||
throw new Error(`Error Code ${event.code}: ${event.message}` || "Unknown error");
|
||||
throw new Error(`Error Code ${event.code}: ${event.message}`);
|
||||
} else if (event.type === "response.failed") {
|
||||
populateResponsesUsageFromResponse(output, event.response?.usage);
|
||||
const error = event.response?.error ?? (event.response as any)?.status_details?.error;
|
||||
const details = event.response?.incomplete_details;
|
||||
const message = error
|
||||
@@ -852,8 +935,12 @@ export function mapOpenAIResponsesStopReason(status: OpenAI.Responses.ResponseSt
|
||||
case "queued":
|
||||
return "stop";
|
||||
default: {
|
||||
// Compile-time exhaustiveness; at runtime a brand-new status from the
|
||||
// server must degrade gracefully instead of failing a fully-streamed
|
||||
// response.
|
||||
const exhaustive: never = status;
|
||||
throw new Error(`Unhandled stop reason: ${exhaustive}`);
|
||||
logger.warn("Unhandled OpenAI Responses stop reason", { status: exhaustive });
|
||||
return "stop";
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -959,7 +1046,9 @@ export function applyResponsesReasoningParams<P extends OpenAI.Responses.Respons
|
||||
// multi-turn conversations when store is false (items aren't persisted server-side, so
|
||||
// we must include the full content). See: https://github.com/can1357/oh-my-pi/issues/41
|
||||
if (includeEncryptedReasoning) {
|
||||
params.include = ["reasoning.encrypted_content"];
|
||||
const include = params.include ?? [];
|
||||
if (!include.includes("reasoning.encrypted_content")) include.push("reasoning.encrypted_content");
|
||||
params.include = include;
|
||||
}
|
||||
|
||||
if (options?.reasoning || options?.reasoningSummary !== undefined) {
|
||||
@@ -1014,6 +1103,10 @@ export function populateResponsesUsageFromResponse(
|
||||
if (!usage) return;
|
||||
const cachedTokens = usage.input_tokens_details?.cached_tokens || 0;
|
||||
const reasoningTokens = usage.output_tokens_details?.reasoning_tokens || 0;
|
||||
// Wholesale replacement must not drop provider-annotated extras (Copilot
|
||||
// premium-request accounting): the failed/cancelled paths throw right after
|
||||
// this call with no later chance to re-apply.
|
||||
const premiumRequests = output.usage.premiumRequests;
|
||||
output.usage = {
|
||||
input: (usage.input_tokens || 0) - cachedTokens,
|
||||
output: usage.output_tokens || 0,
|
||||
@@ -1023,4 +1116,7 @@ export function populateResponsesUsageFromResponse(
|
||||
...(reasoningTokens > 0 ? { reasoningTokens } : {}),
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
};
|
||||
if (premiumRequests !== undefined) {
|
||||
output.usage.premiumRequests = premiumRequests;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { $env, extractHttpStatusFromError, structuredCloneJSON } from "@oh-my-pi/pi-utils";
|
||||
import { $env, extractHttpStatusFromError } from "@oh-my-pi/pi-utils";
|
||||
import OpenAI, { APIConnectionTimeoutError as OpenAIConnectionTimeoutError } from "openai";
|
||||
import type {
|
||||
Tool as OpenAITool,
|
||||
@@ -242,7 +242,7 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
);
|
||||
const premiumRequestsTotal = copilotPremiumRequests;
|
||||
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
||||
const { params } = buildParams(model, context, options, providerSessionState, baseUrl);
|
||||
const params = buildParams(model, context, options, providerSessionState, baseUrl);
|
||||
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
||||
const firstEventTimeoutMs =
|
||||
options?.streamFirstEventTimeoutMs ?? getOpenAIStreamFirstEventTimeoutMs(idleTimeoutMs);
|
||||
@@ -271,6 +271,12 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
const { data, response, request_id } = await client.responses
|
||||
.create(params, requestOptions)
|
||||
.withResponse();
|
||||
// Disarm the first-event watchdog as soon as headers arrive — a slow
|
||||
// onResponse callback must not abort an already-connected stream.
|
||||
if (requestTimeout !== undefined) {
|
||||
clearTimeout(requestTimeout);
|
||||
requestTimeout = undefined;
|
||||
}
|
||||
await notifyProviderResponse(options, response, model, request_id);
|
||||
return data;
|
||||
} catch (error) {
|
||||
@@ -306,10 +312,11 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
||||
if (!firstTokenTime) firstTokenTime = Date.now();
|
||||
},
|
||||
onOutputItemDone: item => {
|
||||
nativeOutputItems.push(structuredCloneJSON<unknown>(item) as unknown as Record<string, unknown>);
|
||||
// `processResponsesStream` hands over a private clone already; no
|
||||
// second deep copy needed (reasoning items carry multi-KB blobs).
|
||||
nativeOutputItems.push(item as unknown as Record<string, unknown>);
|
||||
},
|
||||
});
|
||||
if (premiumRequestsTotal !== undefined) output.usage.premiumRequests = premiumRequestsTotal;
|
||||
|
||||
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
||||
if (firstEventTimeoutError) {
|
||||
@@ -432,18 +439,11 @@ function buildParams(
|
||||
options: OpenAIResponsesOptions | undefined,
|
||||
providerSessionState: OpenAIResponsesProviderSessionState | undefined,
|
||||
resolvedBaseUrl?: string,
|
||||
): { conversationMessages: ResponseInput; params: OpenAIResponsesSamplingParams } {
|
||||
): OpenAIResponsesSamplingParams {
|
||||
const strictResponsesPairing =
|
||||
options?.strictResponsesPairing ??
|
||||
(isAzureOpenAIBaseUrl(model.baseUrl ?? "") || model.provider === "github-copilot");
|
||||
const conversationMessages = convertConversationMessages(
|
||||
model,
|
||||
context,
|
||||
strictResponsesPairing,
|
||||
providerSessionState,
|
||||
options,
|
||||
);
|
||||
const messages: ResponseInput = [...conversationMessages];
|
||||
const messages = convertConversationMessages(model, context, strictResponsesPairing, providerSessionState, options);
|
||||
|
||||
const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
|
||||
let systemInstructions: string | undefined;
|
||||
@@ -471,7 +471,9 @@ function buildParams(
|
||||
instructions: systemInstructions,
|
||||
stream: true,
|
||||
prompt_cache_key: promptCacheKey,
|
||||
prompt_cache_retention: promptCacheKey ? getPromptCacheRetention(model.baseUrl, cacheRetention) : undefined,
|
||||
prompt_cache_retention: promptCacheKey
|
||||
? getPromptCacheRetention(resolvedBaseUrl ?? model.baseUrl, cacheRetention)
|
||||
: undefined,
|
||||
store: false,
|
||||
stream_options: model.provider === "openai" ? { include_obfuscation: false } : undefined,
|
||||
};
|
||||
@@ -516,7 +518,7 @@ function buildParams(
|
||||
Object.assign(params, options.extraBody);
|
||||
}
|
||||
|
||||
return { conversationMessages, params };
|
||||
return params;
|
||||
}
|
||||
|
||||
function mapReasoningEffort(
|
||||
@@ -594,9 +596,13 @@ function convertConversationMessages(
|
||||
messages.push({ role: "user", content });
|
||||
} else if (msg.role === "assistant") {
|
||||
const assistantMsg = msg as AssistantMessage;
|
||||
const providerPayload = shouldReplayNativeHistory
|
||||
? getOpenAIResponsesHistoryPayload(assistantMsg.providerPayload, model.provider, assistantMsg.provider)
|
||||
: undefined;
|
||||
// Native items are model-bound (reasoning carries encrypted content minted
|
||||
// by the producing model); after a mid-session model switch fall back to
|
||||
// block re-encode, which strips foreign signatures.
|
||||
const providerPayload =
|
||||
shouldReplayNativeHistory && assistantMsg.api === model.api && assistantMsg.model === model.id
|
||||
? getOpenAIResponsesHistoryPayload(assistantMsg.providerPayload, model.provider, assistantMsg.provider)
|
||||
: undefined;
|
||||
const historyItems = providerPayload?.items;
|
||||
if (historyItems) {
|
||||
const sanitizedHistoryItems = sanitizeOpenAIResponsesHistoryItemsForReplay(filterReasoning(historyItems));
|
||||
|
||||
@@ -19,9 +19,24 @@ const enum ToolCallStatus {
|
||||
const MAX_TOOL_CALL_ID_LENGTH = 64;
|
||||
|
||||
function appendDuplicateSuffix(originalId: string, suffix: string, maxLength: number): string {
|
||||
if (originalId.length + suffix.length <= maxLength) return `${originalId}${suffix}`;
|
||||
// Responses-family ids are composites (`callId|itemId`): the wire call_id is
|
||||
// the FIRST segment (normalizeResponsesToolCallId splits on `|`), so the
|
||||
// suffix must land on every segment or the duplicate collapses back onto the
|
||||
// original call_id at encode time. The length budget applies per segment,
|
||||
// matching the per-segment caps of the provider normalizers.
|
||||
if (originalId.includes("|")) {
|
||||
return originalId
|
||||
.split("|")
|
||||
.map(segment => appendSegmentDuplicateSuffix(segment, suffix, maxLength))
|
||||
.join("|");
|
||||
}
|
||||
return appendSegmentDuplicateSuffix(originalId, suffix, maxLength);
|
||||
}
|
||||
|
||||
function appendSegmentDuplicateSuffix(segment: string, suffix: string, maxLength: number): string {
|
||||
if (segment.length + suffix.length <= maxLength) return `${segment}${suffix}`;
|
||||
const prefixBudget = Math.max(0, maxLength - suffix.length);
|
||||
return `${originalId.slice(0, prefixBudget)}${suffix}`;
|
||||
return `${segment.slice(0, prefixBudget)}${suffix}`;
|
||||
}
|
||||
|
||||
type PendingToolResultRewrite = { replacementId: string } | undefined;
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
* Anthropic OAuth flow (Claude Pro/Max)
|
||||
*/
|
||||
|
||||
import { claudeCodeVersion } from "../../providers/anthropic";
|
||||
import type { FetchImpl } from "../../types";
|
||||
import { OAuthCallbackFlow } from "./callback-server";
|
||||
import { generatePKCE } from "./pkce";
|
||||
@@ -13,7 +14,7 @@ const AUTHORIZE_URL = "https://claude.ai/oauth/authorize";
|
||||
const TOKEN_URL = "https://api.anthropic.com/v1/oauth/token";
|
||||
const BOOTSTRAP_URL = "https://api.anthropic.com/api/claude_cli/bootstrap";
|
||||
const CLAUDE_CODE_BOOTSTRAP_MODEL = "claude-opus-4-8";
|
||||
const CLAUDE_CODE_BOOTSTRAP_USER_AGENT = "claude-code/2.1.160";
|
||||
const CLAUDE_CODE_BOOTSTRAP_USER_AGENT = `claude-code/${claudeCodeVersion}`;
|
||||
const CALLBACK_PORT = 54545;
|
||||
const CALLBACK_PATH = "/callback";
|
||||
// Scopes required for direct OAuth-token inference (user:inference) plus account/session management.
|
||||
|
||||
@@ -432,8 +432,16 @@ export function streamSimple<TApi extends Api>(
|
||||
let lastKey: string | undefined;
|
||||
try {
|
||||
lastKey = (await apiKeyResolver({ lastChance: false, error: undefined, signal })) || undefined;
|
||||
} catch {
|
||||
lastKey = undefined;
|
||||
} catch (error) {
|
||||
// A thrown resolver is a broker/OAuth/network failure, not a missing
|
||||
// key — surface the cause instead of masking it as "No API key".
|
||||
outer.fail(
|
||||
new Error(
|
||||
`Failed to resolve API key for provider ${model.provider}: ${error instanceof Error ? error.message : String(error)}`,
|
||||
{ cause: error },
|
||||
),
|
||||
);
|
||||
return;
|
||||
}
|
||||
if (lastKey === undefined) {
|
||||
outer.fail(new Error(`No API key for provider: ${model.provider}`));
|
||||
@@ -446,6 +454,9 @@ export function streamSimple<TApi extends Api>(
|
||||
// resolver yields the same key it just tried or `undefined`; the
|
||||
// final step's attempt clears the capture flag so it emits directly.
|
||||
for (let step = 0; step < AUTH_RETRY_STEPS.length; step++) {
|
||||
// Caller aborted between attempts: don't mint a fresh token or fire
|
||||
// another doomed request — emit the captured failure instead.
|
||||
if (signal?.aborted) break;
|
||||
const nextKey = await resolveRetryKey(apiKeyResolver, AUTH_RETRY_STEPS[step]!, failure.error, signal);
|
||||
if (nextKey === undefined || nextKey === lastKey) continue;
|
||||
lastKey = nextKey;
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { scheduler } from "node:timers/promises";
|
||||
import { claudeCodeVersion } from "../providers/anthropic";
|
||||
import type {
|
||||
CredentialRankingStrategy,
|
||||
UsageAmount,
|
||||
@@ -24,7 +25,7 @@ const CLAUDE_HEADERS = {
|
||||
"anthropic-beta":
|
||||
"claude-code-20250219,oauth-2025-04-20,interleaved-thinking-2025-05-14,redact-thinking-2026-02-12,context-management-2025-06-27,prompt-caching-scope-2026-01-05,mid-conversation-system-2026-04-07,advanced-tool-use-2025-11-20,effort-2025-11-24,extended-cache-ttl-2025-04-11",
|
||||
"content-type": "application/json",
|
||||
"user-agent": "claude-cli/2.1.160 (external, cli)",
|
||||
"user-agent": `claude-cli/${claudeCodeVersion} (external, cli)`,
|
||||
connection: "keep-alive",
|
||||
} as const;
|
||||
|
||||
|
||||
@@ -8,7 +8,7 @@ export { isRecord } from "@oh-my-pi/pi-utils";
|
||||
export function normalizeSystemPrompts(systemPrompt: readonly string[] | string | undefined | null): string[] {
|
||||
if (systemPrompt === undefined || systemPrompt === null) return [];
|
||||
const prompts = Array.isArray(systemPrompt) ? systemPrompt : typeof systemPrompt === "string" ? [systemPrompt] : [];
|
||||
return prompts.map(prompt => prompt.toWellFormed()).filter(prompt => prompt.length > 0);
|
||||
return prompts.map(prompt => prompt.toWellFormed()).filter(prompt => prompt.trim().length > 0);
|
||||
}
|
||||
|
||||
export function toNumber(value: unknown): number | undefined {
|
||||
|
||||
@@ -49,3 +49,17 @@ export function createAbortSourceTracker(callerSignal?: AbortSignal): AbortSourc
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Race a shared promise against a caller's AbortSignal without coupling the
|
||||
* underlying work to that signal. The shared promise keeps running (and caches
|
||||
* its result) even when an individual caller bails out.
|
||||
*/
|
||||
export function raceWithSignal<T>(promise: Promise<T>, signal: AbortSignal | undefined): Promise<T> {
|
||||
if (!signal) return promise;
|
||||
if (signal.aborted) return Promise.reject(signal.reason ?? new Error("Request was aborted"));
|
||||
const { promise: aborted, reject } = Promise.withResolvers<never>();
|
||||
const onAbort = () => reject(signal.reason ?? new Error("Request was aborted"));
|
||||
signal.addEventListener("abort", onAbort, { once: true });
|
||||
return Promise.race([promise, aborted]).finally(() => signal.removeEventListener("abort", onAbort));
|
||||
}
|
||||
|
||||
@@ -1,69 +0,0 @@
|
||||
function abortReason(signal: AbortSignal): Error {
|
||||
const reason = signal.reason;
|
||||
if (reason instanceof Error) return reason;
|
||||
if (typeof reason === "string") return new Error(reason);
|
||||
return new Error("Request was aborted");
|
||||
}
|
||||
|
||||
/**
|
||||
* Iterates a provider stream until it yields, ends, errors, or the caller aborts.
|
||||
*/
|
||||
export async function* iterateUntilAbort<T>(iterable: AsyncIterable<T>, signal?: AbortSignal): AsyncGenerator<T> {
|
||||
const iterator = iterable[Symbol.asyncIterator]();
|
||||
const closeIterator = (): void => {
|
||||
const returnPromise = iterator.return?.();
|
||||
if (returnPromise) {
|
||||
void returnPromise.catch(() => {});
|
||||
}
|
||||
};
|
||||
|
||||
if (signal?.aborted) {
|
||||
closeIterator();
|
||||
throw abortReason(signal);
|
||||
}
|
||||
|
||||
const withResult = (promise: Promise<IteratorResult<T>>) =>
|
||||
promise.then(
|
||||
result => ({ kind: "next" as const, result }),
|
||||
error => ({ kind: "error" as const, error }),
|
||||
);
|
||||
|
||||
while (true) {
|
||||
if (signal?.aborted) {
|
||||
closeIterator();
|
||||
throw abortReason(signal);
|
||||
}
|
||||
const racers: Array<
|
||||
Promise<{ kind: "next"; result: IteratorResult<T> } | { kind: "error"; error: unknown } | { kind: "abort" }>
|
||||
> = [withResult(iterator.next())];
|
||||
let abortListener: (() => void) | undefined;
|
||||
let resolveAbort: ((value: { kind: "abort" }) => void) | undefined;
|
||||
if (signal) {
|
||||
const { promise, resolve } = Promise.withResolvers<{ kind: "abort" }>();
|
||||
resolveAbort = resolve;
|
||||
abortListener = () => resolve({ kind: "abort" });
|
||||
signal.addEventListener("abort", abortListener, { once: true });
|
||||
racers.push(promise);
|
||||
}
|
||||
|
||||
try {
|
||||
const outcome = await Promise.race(racers);
|
||||
if (outcome.kind === "abort") {
|
||||
closeIterator();
|
||||
throw abortReason(signal!);
|
||||
}
|
||||
if (outcome.kind === "error") {
|
||||
throw outcome.error;
|
||||
}
|
||||
if (outcome.result.done) {
|
||||
return;
|
||||
}
|
||||
yield outcome.result.value;
|
||||
} finally {
|
||||
if (abortListener && signal) {
|
||||
signal.removeEventListener("abort", abortListener);
|
||||
}
|
||||
resolveAbort?.({ kind: "abort" });
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,8 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
||||
queue: T[] = [];
|
||||
waiting: Array<{ resolve: (value: IteratorResult<T>) => void; reject: (err: unknown) => void }> = [];
|
||||
done = false;
|
||||
/** True once finalResultPromise has been resolved or rejected. */
|
||||
resultSettled = false;
|
||||
#failed = false;
|
||||
#error: unknown = undefined;
|
||||
finalResultPromise: Promise<R>;
|
||||
@@ -30,6 +32,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
||||
|
||||
if (this.isComplete(event)) {
|
||||
this.done = true;
|
||||
this.resultSettled = true;
|
||||
this.resolveFinalResult(this.extractResult(event));
|
||||
}
|
||||
|
||||
@@ -54,7 +57,13 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
||||
end(result?: R): void {
|
||||
this.done = true;
|
||||
if (result !== undefined) {
|
||||
this.resultSettled = true;
|
||||
this.resolveFinalResult(result);
|
||||
} else if (!this.resultSettled) {
|
||||
// end() without a terminal value must still settle result() —
|
||||
// otherwise complete()/result() awaits hang forever.
|
||||
this.resultSettled = true;
|
||||
this.rejectFinalResult(new Error("Stream ended without a final result"));
|
||||
}
|
||||
// Notify all waiting consumers that we're done
|
||||
while (this.waiting.length > 0) {
|
||||
@@ -75,6 +84,7 @@ export class EventStream<T, R = T> implements AsyncIterable<T> {
|
||||
this.done = true;
|
||||
this.#failed = true;
|
||||
this.#error = err;
|
||||
this.resultSettled = true;
|
||||
this.rejectFinalResult(err);
|
||||
while (this.waiting.length > 0) {
|
||||
const waiter = this.waiting.shift()!;
|
||||
@@ -126,6 +136,7 @@ export class AssistantMessageEventStream extends EventStream<AssistantMessageEve
|
||||
// Completion resolves the final result and still emits the terminal event.
|
||||
if (this.isComplete(event)) {
|
||||
this.done = true;
|
||||
this.resultSettled = true;
|
||||
this.resolveFinalResult(this.extractResult(event));
|
||||
}
|
||||
|
||||
@@ -135,7 +146,13 @@ export class AssistantMessageEventStream extends EventStream<AssistantMessageEve
|
||||
override end(result?: AssistantMessage): void {
|
||||
this.done = true;
|
||||
if (result !== undefined) {
|
||||
this.resultSettled = true;
|
||||
this.resolveFinalResult(result);
|
||||
} else if (!this.resultSettled) {
|
||||
// Mirror the base class: a result-less end() must not leave
|
||||
// result() pending forever.
|
||||
this.resultSettled = true;
|
||||
this.rejectFinalResult(new Error("Stream ended without a final result"));
|
||||
}
|
||||
this.endWaiting();
|
||||
}
|
||||
|
||||
@@ -2,6 +2,8 @@ import { $env } from "@oh-my-pi/pi-utils";
|
||||
|
||||
const DEFAULT_STREAM_IDLE_TIMEOUT_MS = 120_000;
|
||||
const DEFAULT_STREAM_FIRST_EVENT_TIMEOUT_MS = 100_000;
|
||||
/** Re-mint persistent race promises every N iterations (see hoisted-racer comment). */
|
||||
const RACER_REMINT_INTERVAL = 1024;
|
||||
|
||||
function normalizeIdleTimeoutMs(value: string | undefined, fallback: number): number | undefined {
|
||||
if (value === undefined) return fallback;
|
||||
@@ -164,109 +166,184 @@ export async function* iterateWithIdleTimeout<T>(
|
||||
(firstItemTimeoutMs === undefined || firstItemTimeoutMs <= 0) &&
|
||||
(options.idleTimeoutMs === undefined || options.idleTimeoutMs <= 0);
|
||||
|
||||
while (true) {
|
||||
let activeTimeoutMs: number | undefined;
|
||||
if (awaitingFirstItem) {
|
||||
if (firstItemDeadlineMs !== undefined) {
|
||||
activeTimeoutMs = firstItemDeadlineMs - Date.now();
|
||||
if (activeTimeoutMs <= 0) {
|
||||
options.onFirstItemTimeout?.();
|
||||
closeIterator();
|
||||
throw new Error(options.firstItemErrorMessage ?? options.errorMessage);
|
||||
}
|
||||
}
|
||||
} else if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) {
|
||||
activeTimeoutMs = options.idleTimeoutMs - (Date.now() - lastProgressAt);
|
||||
if (activeTimeoutMs <= 0) {
|
||||
options.onIdle?.();
|
||||
closeIterator();
|
||||
throw new Error(options.errorMessage);
|
||||
}
|
||||
// Persistent racers, hoisted out of the per-item loop. The abort promise can
|
||||
// only ever resolve once (abort latches), and a timeout resolution always
|
||||
// precedes a throw — so neither needs per-item re-creation. This keeps the
|
||||
// token hot path free of timer create/destroy and listener churn.
|
||||
//
|
||||
// Each Promise.race() call still attaches a reaction record to every pending
|
||||
// racer, and those records live until the racer settles — so a never-firing
|
||||
// abort/timeout promise would accumulate one record per streamed item for
|
||||
// the stream's whole life. The loop re-mints both promises every
|
||||
// RACER_REMINT_INTERVAL iterations to keep that retention bounded; the
|
||||
// listener and timer callbacks resolve through late-bound variables so a
|
||||
// re-mint never strands them.
|
||||
let abortPromise: Promise<{ kind: "abort" }> | undefined;
|
||||
let abortListener: (() => void) | undefined;
|
||||
let resolveAbort: ((value: { kind: "abort" }) => void) | undefined;
|
||||
if (abortSignal) {
|
||||
const { promise, resolve } = Promise.withResolvers<{ kind: "abort" }>();
|
||||
resolveAbort = resolve;
|
||||
abortListener = () => resolveAbort?.({ kind: "abort" });
|
||||
abortSignal.addEventListener("abort", abortListener, { once: true });
|
||||
abortPromise = promise;
|
||||
}
|
||||
|
||||
let timeoutPromise: Promise<{ kind: "timeout" }> | undefined;
|
||||
let resolveTimeout: ((value: { kind: "timeout" }) => void) | undefined;
|
||||
let timeoutFired = false;
|
||||
let timer: NodeJS.Timeout | undefined;
|
||||
let timerFireAtMs = Infinity;
|
||||
|
||||
const currentDeadlineMs = (): number | undefined => {
|
||||
if (awaitingFirstItem) return firstItemDeadlineMs;
|
||||
if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) {
|
||||
return lastProgressAt + options.idleTimeoutMs;
|
||||
}
|
||||
|
||||
const nextResultPromise = withRacy(iterator.next());
|
||||
|
||||
const racers: Array<
|
||||
Promise<
|
||||
| { kind: "next"; result: IteratorResult<T> }
|
||||
| { kind: "error"; error: unknown }
|
||||
| { kind: "timeout" }
|
||||
| { kind: "abort" }
|
||||
>
|
||||
> = [nextResultPromise];
|
||||
|
||||
let timer: NodeJS.Timeout | undefined;
|
||||
let resolveTimeout: ((value: { kind: "timeout" }) => void) | undefined;
|
||||
const enforceTimeout = !noTimeoutEnforced && activeTimeoutMs !== undefined && activeTimeoutMs > 0;
|
||||
if (enforceTimeout) {
|
||||
return undefined;
|
||||
};
|
||||
const onTimerFire = (): void => {
|
||||
timer = undefined;
|
||||
timerFireAtMs = Infinity;
|
||||
const deadlineMs = currentDeadlineMs();
|
||||
if (deadlineMs === undefined) return;
|
||||
const remainingMs = deadlineMs - Date.now();
|
||||
if (remainingMs > 0) {
|
||||
// Progress moved the deadline since this timer was armed — re-arm for
|
||||
// the remainder. One stale wake per idle period, not one per item.
|
||||
timerFireAtMs = deadlineMs;
|
||||
timer = setTimeout(onTimerFire, remainingMs);
|
||||
return;
|
||||
}
|
||||
timeoutFired = true;
|
||||
resolveTimeout?.({ kind: "timeout" });
|
||||
};
|
||||
const armTimer = (deadlineMs: number): void => {
|
||||
if (timeoutPromise === undefined || timeoutFired) {
|
||||
// A fired-but-unconsumed resolution (the item won the same race) is
|
||||
// stale — racing it again would fake a timeout, so mint a fresh one.
|
||||
const { promise, resolve } = Promise.withResolvers<{ kind: "timeout" }>();
|
||||
timeoutPromise = promise;
|
||||
resolveTimeout = resolve;
|
||||
timer = setTimeout(() => resolve({ kind: "timeout" }), activeTimeoutMs);
|
||||
racers.push(promise);
|
||||
timeoutFired = false;
|
||||
}
|
||||
|
||||
let abortListener: (() => void) | undefined;
|
||||
let resolveAbort: ((value: { kind: "abort" }) => void) | undefined;
|
||||
if (abortSignal) {
|
||||
const { promise, resolve } = Promise.withResolvers<{ kind: "abort" }>();
|
||||
resolveAbort = resolve;
|
||||
abortListener = () => resolve({ kind: "abort" });
|
||||
abortSignal.addEventListener("abort", abortListener, { once: true });
|
||||
racers.push(promise);
|
||||
if (timer !== undefined) {
|
||||
// An armed timer firing at or before the new deadline re-arms itself.
|
||||
if (timerFireAtMs <= deadlineMs) return;
|
||||
clearTimeout(timer);
|
||||
}
|
||||
timerFireAtMs = deadlineMs;
|
||||
timer = setTimeout(onTimerFire, Math.max(0, deadlineMs - Date.now()));
|
||||
};
|
||||
|
||||
// Tracks whether this iteration handed an item to the consumer and resumed
|
||||
// normally. Any other exit — internal throw, `done` return, or the consumer
|
||||
// abandoning us via `.return()`/`.throw()` at the `yield` below — must close
|
||||
// the upstream iterator so the underlying SSE body / SDK stream (and its
|
||||
// socket) is released instead of being left suspended.
|
||||
let continuing = false;
|
||||
try {
|
||||
const outcome = await Promise.race(racers);
|
||||
if (outcome.kind === "abort") {
|
||||
closeIterator();
|
||||
throw abortReason(abortSignal!);
|
||||
}
|
||||
if (outcome.kind === "timeout") {
|
||||
if (!awaitingFirstItem) {
|
||||
options.onIdle?.();
|
||||
} else {
|
||||
options.onFirstItemTimeout?.();
|
||||
try {
|
||||
let raceCount = 0;
|
||||
while (true) {
|
||||
if (++raceCount % RACER_REMINT_INTERVAL === 0) {
|
||||
if (abortPromise !== undefined && !abortSignal!.aborted) {
|
||||
const { promise, resolve } = Promise.withResolvers<{ kind: "abort" }>();
|
||||
resolveAbort = resolve;
|
||||
abortPromise = promise;
|
||||
}
|
||||
if (timeoutPromise !== undefined && !timeoutFired) {
|
||||
const { promise, resolve } = Promise.withResolvers<{ kind: "timeout" }>();
|
||||
resolveTimeout = resolve;
|
||||
timeoutPromise = promise;
|
||||
}
|
||||
closeIterator();
|
||||
throw new Error(
|
||||
!awaitingFirstItem ? options.errorMessage : (options.firstItemErrorMessage ?? options.errorMessage),
|
||||
);
|
||||
}
|
||||
if (outcome.kind === "error") {
|
||||
throw outcome.error;
|
||||
let activeTimeoutMs: number | undefined;
|
||||
if (awaitingFirstItem) {
|
||||
if (firstItemDeadlineMs !== undefined) {
|
||||
activeTimeoutMs = firstItemDeadlineMs - Date.now();
|
||||
if (activeTimeoutMs <= 0) {
|
||||
options.onFirstItemTimeout?.();
|
||||
closeIterator();
|
||||
throw new Error(options.firstItemErrorMessage ?? options.errorMessage);
|
||||
}
|
||||
}
|
||||
} else if (options.idleTimeoutMs !== undefined && options.idleTimeoutMs > 0) {
|
||||
activeTimeoutMs = options.idleTimeoutMs - (Date.now() - lastProgressAt);
|
||||
if (activeTimeoutMs <= 0) {
|
||||
options.onIdle?.();
|
||||
closeIterator();
|
||||
throw new Error(options.errorMessage);
|
||||
}
|
||||
}
|
||||
if (outcome.result.done) {
|
||||
markFirstItemReceived();
|
||||
return;
|
||||
|
||||
const nextResultPromise = withRacy(iterator.next());
|
||||
|
||||
const racers: Array<
|
||||
Promise<
|
||||
| { kind: "next"; result: IteratorResult<T> }
|
||||
| { kind: "error"; error: unknown }
|
||||
| { kind: "timeout" }
|
||||
| { kind: "abort" }
|
||||
>
|
||||
> = [nextResultPromise];
|
||||
|
||||
const enforceTimeout = !noTimeoutEnforced && activeTimeoutMs !== undefined && activeTimeoutMs > 0;
|
||||
if (enforceTimeout) {
|
||||
armTimer(Date.now() + activeTimeoutMs!);
|
||||
racers.push(timeoutPromise!);
|
||||
}
|
||||
const item = outcome.result.value;
|
||||
// Non-progress items (e.g. provider keepalives, synthetic `start` events that
|
||||
// arrive before the model has produced any tokens) MUST NOT flip us out of
|
||||
// `awaitingFirstItem`. Otherwise the next iteration switches from the (longer)
|
||||
// first-item watchdog to the (shorter) idle watchdog while we're still waiting
|
||||
// on the model's first real output.
|
||||
if (isProgressItem(item)) {
|
||||
markFirstItemReceived();
|
||||
lastProgressAt = Date.now();
|
||||
if (abortPromise) {
|
||||
racers.push(abortPromise);
|
||||
}
|
||||
yield item;
|
||||
continuing = true;
|
||||
} finally {
|
||||
if (!continuing) closeIterator();
|
||||
if (timer !== undefined) clearTimeout(timer);
|
||||
// Resolve dangling promises so the racers don't leak (Promise.race is one-shot).
|
||||
resolveTimeout?.({ kind: "timeout" });
|
||||
if (abortListener && abortSignal) {
|
||||
abortSignal.removeEventListener("abort", abortListener);
|
||||
|
||||
// Tracks whether this iteration handed an item to the consumer and resumed
|
||||
// normally. Any other exit — internal throw, `done` return, or the consumer
|
||||
// abandoning us via `.return()`/`.throw()` at the `yield` below — must close
|
||||
// the upstream iterator so the underlying SSE body / SDK stream (and its
|
||||
// socket) is released instead of being left suspended.
|
||||
let continuing = false;
|
||||
try {
|
||||
const outcome = await Promise.race(racers);
|
||||
if (outcome.kind === "abort") {
|
||||
closeIterator();
|
||||
throw abortReason(abortSignal!);
|
||||
}
|
||||
if (outcome.kind === "timeout") {
|
||||
if (!awaitingFirstItem) {
|
||||
options.onIdle?.();
|
||||
} else {
|
||||
options.onFirstItemTimeout?.();
|
||||
}
|
||||
closeIterator();
|
||||
throw new Error(
|
||||
!awaitingFirstItem ? options.errorMessage : (options.firstItemErrorMessage ?? options.errorMessage),
|
||||
);
|
||||
}
|
||||
if (outcome.kind === "error") {
|
||||
throw outcome.error;
|
||||
}
|
||||
if (outcome.result.done) {
|
||||
markFirstItemReceived();
|
||||
return;
|
||||
}
|
||||
const item = outcome.result.value;
|
||||
// Non-progress items (e.g. provider keepalives, synthetic `start` events that
|
||||
// arrive before the model has produced any tokens) MUST NOT flip us out of
|
||||
// `awaitingFirstItem`. Otherwise the next iteration switches from the (longer)
|
||||
// first-item watchdog to the (shorter) idle watchdog while we're still waiting
|
||||
// on the model's first real output.
|
||||
if (isProgressItem(item)) {
|
||||
markFirstItemReceived();
|
||||
lastProgressAt = Date.now();
|
||||
}
|
||||
yield item;
|
||||
continuing = true;
|
||||
} finally {
|
||||
if (!continuing) closeIterator();
|
||||
}
|
||||
resolveAbort?.({ kind: "abort" });
|
||||
}
|
||||
} finally {
|
||||
if (timer !== undefined) clearTimeout(timer);
|
||||
// Settle the persistent racers so the final Promise.race releases them.
|
||||
resolveTimeout?.({ kind: "timeout" });
|
||||
if (abortListener && abortSignal) {
|
||||
abortSignal.removeEventListener("abort", abortListener);
|
||||
}
|
||||
resolveAbort?.({ kind: "abort" });
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -28,7 +28,7 @@ export function getRetryAfterMsFromHeaders(headers: HeadersLike): number | undef
|
||||
return Math.max(...candidates);
|
||||
}
|
||||
|
||||
function getHeadersFromError(error: unknown): HeadersLike {
|
||||
export function getHeadersFromError(error: unknown): HeadersLike {
|
||||
if (!error || typeof error !== "object") return undefined;
|
||||
const record = error as { headers?: unknown; response?: { headers?: unknown }; cause?: unknown };
|
||||
const direct = extractHeaders(record.headers) ?? extractHeaders(record.response?.headers);
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
import { scheduler } from "node:timers/promises";
|
||||
import { extractHttpStatusFromError, isRetryableError } from "@oh-my-pi/pi-utils";
|
||||
import { getHeadersFromError, getRetryAfterMsFromHeaders } from "./retry-after";
|
||||
|
||||
/**
|
||||
* GitHub Copilot intermittently rejects preview models (gpt-5.3-codex,
|
||||
@@ -24,6 +25,8 @@ export function isCopilotTransientModelError(error: unknown): boolean {
|
||||
|
||||
const COPILOT_MODEL_RETRY_MAX_ATTEMPTS = 3;
|
||||
const COPILOT_MODEL_RETRY_BASE_DELAY_MS = 400;
|
||||
/** Longest server-requested backoff we are willing to sit out before giving up. */
|
||||
const COPILOT_RETRY_AFTER_MAX_WAIT_MS = 30_000;
|
||||
|
||||
/**
|
||||
* Wrap an initial Copilot request so transient `model_not_supported` 400s are
|
||||
@@ -45,9 +48,27 @@ export async function callWithCopilotModelRetry<T>(
|
||||
return await fn();
|
||||
} catch (error) {
|
||||
lastError = error;
|
||||
if (!isCopilotTransientModelError(error) && !isRetryableError(error)) throw error;
|
||||
// A latched abort (caller cancel or local watchdog) makes any retry a
|
||||
// guaranteed-dead attempt — surface the original error, not the
|
||||
// scheduler's AbortError.
|
||||
if (options.signal?.aborted) throw error;
|
||||
const transientModelError = isCopilotTransientModelError(error);
|
||||
if (!transientModelError && !isRetryableError(error)) throw error;
|
||||
if (attempt === COPILOT_MODEL_RETRY_MAX_ATTEMPTS - 1) break;
|
||||
await scheduler.wait(retryBaseDelayMs * (attempt + 1), { signal: options.signal });
|
||||
let delayMs = retryBaseDelayMs * (attempt + 1);
|
||||
if (!transientModelError) {
|
||||
const status = extractHttpStatusFromError(error);
|
||||
if (status !== undefined) {
|
||||
// Status-bearing retryable errors (429/5xx) are only re-sent when
|
||||
// the server told us when to come back — a blind fixed-delay retry
|
||||
// of a rate limit just burns the remaining attempts. Status-less
|
||||
// transport blips (socket close, h2 reset) keep the linear backoff.
|
||||
const retryAfterMs = getRetryAfterMsFromHeaders(getHeadersFromError(error));
|
||||
if (retryAfterMs === undefined || retryAfterMs > COPILOT_RETRY_AFTER_MAX_WAIT_MS) throw error;
|
||||
delayMs = Math.max(delayMs, retryAfterMs);
|
||||
}
|
||||
}
|
||||
await scheduler.wait(delayMs, { signal: options.signal });
|
||||
}
|
||||
}
|
||||
throw lastError;
|
||||
|
||||
@@ -52,7 +52,6 @@ export interface NormalizeSchemaOptions {
|
||||
|
||||
interface NormalizeSchemaWalkOptions extends NormalizeSchemaOptions {
|
||||
insideProperties: boolean;
|
||||
epoch: number;
|
||||
}
|
||||
|
||||
interface ResidualIncompatibilityChecks {
|
||||
@@ -219,13 +218,27 @@ function applyDescriptionSpill(
|
||||
|
||||
function normalizeSchemaNode(value: unknown, options: NormalizeSchemaWalkOptions): unknown {
|
||||
if (Array.isArray(value)) {
|
||||
if (!once(value, options.epoch)) return [];
|
||||
return value.map(entry => normalizeSchemaNode(entry, options));
|
||||
if (!enter(value)) return [];
|
||||
try {
|
||||
return value.map(entry => normalizeSchemaNode(entry, options));
|
||||
} finally {
|
||||
exit(value);
|
||||
}
|
||||
}
|
||||
if (!isJsonObject(value)) {
|
||||
return value;
|
||||
}
|
||||
if (!once(value, options.epoch)) return {};
|
||||
// `enter`/`exit` path-tracking (not a visited-set): DAG-shared subtrees are
|
||||
// normalized at every occurrence; only true cycles short-circuit to `{}`.
|
||||
if (!enter(value)) return {};
|
||||
try {
|
||||
return normalizeSchemaObjectNode(value, options);
|
||||
} finally {
|
||||
exit(value);
|
||||
}
|
||||
}
|
||||
|
||||
function normalizeSchemaObjectNode(value: JsonObject, options: NormalizeSchemaWalkOptions): unknown {
|
||||
let obj = options.normalizeFieldNames && !options.insideProperties ? applySnakeCaseRenames(value) : value;
|
||||
if (options.collapseNullFields && !options.insideProperties) {
|
||||
obj = preHandleNullFields(obj);
|
||||
@@ -795,7 +808,6 @@ export function normalizeSchema(value: unknown, options: NormalizeSchemaOptions)
|
||||
let normalized = normalizeSchemaNode(dereferenced, {
|
||||
...options,
|
||||
insideProperties: false,
|
||||
epoch: epochNext(),
|
||||
});
|
||||
if (options.stripResidualCombinersFixpoint) {
|
||||
normalized = stripResidualCombiners(normalized);
|
||||
|
||||
@@ -9,11 +9,13 @@
|
||||
*
|
||||
* Caveats: the stamp lives as long as the host object, even after callers
|
||||
* release their references to the cached value — only use this for caches
|
||||
* whose lifetime should match the host. Frozen hosts will throw on write in
|
||||
* strict mode; callers that may receive frozen input must handle that.
|
||||
* whose lifetime should match the host. Frozen hosts cannot be stamped;
|
||||
* `define` silently skips them, so memoization/visit-tracking degrades to
|
||||
* best-effort (recompute on every call, no cycle protection) instead of
|
||||
* throwing.
|
||||
*/
|
||||
|
||||
function define<T extends object>(target: T, key: symbol, value: unknown): void {
|
||||
if (Object.isFrozen(target)) return;
|
||||
Object.defineProperty(target, key, { value, writable: true, configurable: true });
|
||||
}
|
||||
|
||||
@@ -79,7 +81,13 @@ export function once<T extends object>(target: T, epoch: number): boolean {
|
||||
*/
|
||||
const kDepth = Symbol("pi.schema.depth");
|
||||
|
||||
/** Returns `true` on first entry, `false` if `target` is already on the current path. */
|
||||
/**
|
||||
* Returns `true` on first entry, `false` if `target` is already on the
|
||||
* current path. A `false` return does NOT deepen the counter — callers pair
|
||||
* `exit` only with successful enters (`if (!enter(n)) bail; try {…} finally
|
||||
* { exit(n); }`), so incrementing on the cycle branch would leak depth and
|
||||
* make every later top-level walk of the same object misreport a cycle.
|
||||
*/
|
||||
export function enter<T extends object>(target: T): boolean {
|
||||
const slot = target as Record<symbol, number | undefined>;
|
||||
const cur = slot[kDepth];
|
||||
@@ -87,11 +95,15 @@ export function enter<T extends object>(target: T): boolean {
|
||||
define(target, kDepth, 1);
|
||||
return true;
|
||||
}
|
||||
slot[kDepth] = cur + 1;
|
||||
return cur === 0;
|
||||
if (cur !== 0) return false;
|
||||
slot[kDepth] = 1;
|
||||
return true;
|
||||
}
|
||||
|
||||
export function exit<T extends object>(target: T): void {
|
||||
const slot = target as Record<symbol, number>;
|
||||
slot[kDepth]--;
|
||||
const slot = target as Record<symbol, number | undefined>;
|
||||
const cur = slot[kDepth];
|
||||
// Frozen targets never received the kDepth stamp in `enter` — nothing to unwind.
|
||||
if (cur === undefined) return;
|
||||
slot[kDepth] = cur - 1;
|
||||
}
|
||||
|
||||
@@ -36,6 +36,8 @@ const DSML_PARAMETER_OPEN_RE = new RegExp(
|
||||
"y",
|
||||
);
|
||||
const DSML_PARAMETER_CLOSE_RE = new RegExp(`</${DSML_PIPE}DSML${DSML_PIPE}parameter>`, "y");
|
||||
/** Canonical DSML section-open shape; `|` positions accept either pipe variant. */
|
||||
const DSML_SECTION_OPEN_TEMPLATE = "<|DSML|tool_calls>";
|
||||
|
||||
const THINK_OPEN = "<think>";
|
||||
const THINK_CLOSE = "</think>";
|
||||
@@ -81,6 +83,7 @@ type XmlToolState =
|
||||
readonly paramName: string;
|
||||
readonly isString: boolean;
|
||||
value: string;
|
||||
truncated?: boolean;
|
||||
};
|
||||
|
||||
type ThinkingTag = { readonly open: string; readonly close: string };
|
||||
@@ -429,12 +432,25 @@ export class StreamMarkupHealing {
|
||||
continue;
|
||||
}
|
||||
} else if (this.#tryMatch(config.parameterClose)) {
|
||||
state.args[state.paramName] = coerceXmlParamValue(state.value, state.isString);
|
||||
// A capped value executes with silently corrupted input unless the
|
||||
// truncation is made explicit — the marker fails JSON params loudly
|
||||
// and tells the model/tool what happened to string params.
|
||||
const paramValue = state.truncated
|
||||
? `${state.value}\n…[parameter truncated: exceeded ${MAX_XML_PARAM_VALUE_LENGTH} bytes]`
|
||||
: state.value;
|
||||
state.args[state.paramName] = coerceXmlParamValue(paramValue, state.isString);
|
||||
config.setState({ kind: "invoke", name: state.invokeName, args: state.args });
|
||||
continue;
|
||||
}
|
||||
|
||||
if (this.#startsWithPartialXmlTag()) break;
|
||||
if (state.kind === "idle") {
|
||||
// In idle, a bare `<` is legitimate output (`a < b`, generics, JSX).
|
||||
// Only hold back tails that could still grow into the DSML
|
||||
// section-open tag; everything else flows through immediately.
|
||||
if (this.#startsWithPartialDsmlSectionOpen()) break;
|
||||
} else if (this.#startsWithPartialXmlTag()) {
|
||||
break;
|
||||
}
|
||||
|
||||
const ch = this.#buffer[this.#offset]!;
|
||||
this.#offset += 1;
|
||||
@@ -443,11 +459,15 @@ export class StreamMarkupHealing {
|
||||
continue;
|
||||
}
|
||||
if (state.kind === "parameter") {
|
||||
if (state.value.length >= MAX_XML_PARAM_VALUE_LENGTH) {
|
||||
config.setState({ kind: "idle" });
|
||||
continue;
|
||||
if (state.value.length < MAX_XML_PARAM_VALUE_LENGTH) {
|
||||
state.value += ch;
|
||||
} else {
|
||||
// Beyond the cap the value stops growing, but we stay in
|
||||
// `parameter` state so the rest of the envelope — including its
|
||||
// close tags — is still swallowed instead of leaking into
|
||||
// visible text. The close handler appends an explicit marker.
|
||||
state.truncated = true;
|
||||
}
|
||||
state.value += ch;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -511,6 +531,21 @@ export class StreamMarkupHealing {
|
||||
return true;
|
||||
}
|
||||
|
||||
#startsWithPartialDsmlSectionOpen(): boolean {
|
||||
const tailLength = this.#buffer.length - this.#offset;
|
||||
if (tailLength === 0 || tailLength >= DSML_SECTION_OPEN_TEMPLATE.length) return false;
|
||||
for (let i = 0; i < tailLength; i++) {
|
||||
const ch = this.#buffer[this.#offset + i]!;
|
||||
const expected = DSML_SECTION_OPEN_TEMPLATE[i]!;
|
||||
if (expected === "|") {
|
||||
if (ch !== "|" && ch !== "|") return false;
|
||||
} else if (ch !== expected) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
#bufferIsPrefixOf(token: string, remainingLength: number): boolean {
|
||||
for (let i = 0; i < remainingLength; i++) {
|
||||
if (this.#buffer[this.#offset + i] !== token[i]) return false;
|
||||
|
||||
@@ -979,6 +979,23 @@ export function validateToolCall(tools: Tool[], toolCall: ToolCall): ToolCall["a
|
||||
return validateToolArguments(tool, toolCall);
|
||||
}
|
||||
|
||||
/** Cap per-field string lengths when embedding received args in an error message. */
|
||||
const MAX_ERROR_ARG_STRING_LENGTH = 256;
|
||||
|
||||
function truncateArgsForError(value: unknown): unknown {
|
||||
if (typeof value === "string") {
|
||||
if (value.length <= MAX_ERROR_ARG_STRING_LENGTH) return value;
|
||||
return `${value.slice(0, MAX_ERROR_ARG_STRING_LENGTH)}… [truncated ${value.length - MAX_ERROR_ARG_STRING_LENGTH} chars]`;
|
||||
}
|
||||
if (Array.isArray(value)) return value.map(truncateArgsForError);
|
||||
if (value !== null && typeof value === "object") {
|
||||
const out: Record<string, unknown> = {};
|
||||
for (const [key, entry] of Object.entries(value)) out[key] = truncateArgsForError(entry);
|
||||
return out;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
/**
|
||||
* Validates tool call arguments against the tool's schema (Zod or plain JSON
|
||||
* Schema). Applies LLM-quirk coercions (numeric strings, JSON-string
|
||||
@@ -1025,12 +1042,15 @@ export function validateToolArguments(tool: Tool, toolCall: ToolCall): ToolCall[
|
||||
// existing tests; the detailed body is informational.
|
||||
const errors = result.messages.join("\n") || "Unknown validation error";
|
||||
|
||||
// Truncate long per-field strings: the full payload (potentially hundreds
|
||||
// of KB for write/edit-class calls) would otherwise round-trip back to the
|
||||
// model inside the tool error.
|
||||
const receivedArgs = changed
|
||||
? {
|
||||
original: originalArgs,
|
||||
normalized: normalizedArgs,
|
||||
original: truncateArgsForError(originalArgs),
|
||||
normalized: truncateArgsForError(normalizedArgs),
|
||||
}
|
||||
: originalArgs;
|
||||
: truncateArgsForError(originalArgs);
|
||||
|
||||
const errorMessage = `Validation failed for tool "${
|
||||
toolCall.name
|
||||
|
||||
@@ -1,139 +0,0 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { iterateUntilAbort } from "@oh-my-pi/pi-ai/utils/abortable-iterator";
|
||||
|
||||
function makeSource<T>(handlers: { next: () => Promise<IteratorResult<T>>; onReturn?: () => void }): AsyncIterable<T> {
|
||||
return {
|
||||
[Symbol.asyncIterator](): AsyncIterator<T> {
|
||||
return {
|
||||
next: handlers.next,
|
||||
async return(): Promise<IteratorResult<T>> {
|
||||
handlers.onReturn?.();
|
||||
return { done: true, value: undefined as unknown as T };
|
||||
},
|
||||
};
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
describe("iterateUntilAbort", () => {
|
||||
it("observes aborts that happen between yielded items and calls iterator.return()", async () => {
|
||||
const controller = new AbortController();
|
||||
let nextCalls = 0;
|
||||
let returnCalled = false;
|
||||
const source = makeSource<number>({
|
||||
next: async () => {
|
||||
nextCalls += 1;
|
||||
if (nextCalls === 1) return { done: false, value: 1 };
|
||||
const { promise } = Promise.withResolvers<IteratorResult<number>>();
|
||||
return promise;
|
||||
},
|
||||
onReturn: () => {
|
||||
returnCalled = true;
|
||||
},
|
||||
});
|
||||
const iterator = iterateUntilAbort(source, controller.signal);
|
||||
|
||||
await expect(iterator.next()).resolves.toEqual({ done: false, value: 1 });
|
||||
controller.abort();
|
||||
await expect(iterator.next()).rejects.toThrow(/abort/i);
|
||||
expect(nextCalls).toBe(1);
|
||||
expect(returnCalled).toBe(true);
|
||||
});
|
||||
|
||||
it("observes aborts that fire DURING an in-flight iterator.next()", async () => {
|
||||
const controller = new AbortController();
|
||||
let returnCalled = false;
|
||||
const source = makeSource<number>({
|
||||
next: async () => {
|
||||
const { promise } = Promise.withResolvers<IteratorResult<number>>();
|
||||
return promise; // never resolves
|
||||
},
|
||||
onReturn: () => {
|
||||
returnCalled = true;
|
||||
},
|
||||
});
|
||||
const iterator = iterateUntilAbort(source, controller.signal);
|
||||
|
||||
const pending = iterator.next();
|
||||
setTimeout(() => controller.abort(new Error("torn down")), 5);
|
||||
|
||||
await expect(pending).rejects.toThrow(/torn down/);
|
||||
expect(returnCalled).toBe(true);
|
||||
});
|
||||
|
||||
it("rejects immediately when the signal is already aborted before the first next()", async () => {
|
||||
const controller = new AbortController();
|
||||
controller.abort(new Error("preflight"));
|
||||
let returnCalled = false;
|
||||
const source = makeSource<number>({
|
||||
next: async () => ({ done: false, value: 1 }),
|
||||
onReturn: () => {
|
||||
returnCalled = true;
|
||||
},
|
||||
});
|
||||
|
||||
const iterator = iterateUntilAbort(source, controller.signal);
|
||||
await expect(iterator.next()).rejects.toThrow(/preflight/);
|
||||
expect(returnCalled).toBe(true);
|
||||
});
|
||||
|
||||
it("yields every item and terminates cleanly when the source completes naturally", async () => {
|
||||
const items = [1, 2, 3];
|
||||
let i = 0;
|
||||
const source = makeSource<number>({
|
||||
next: async () =>
|
||||
i < items.length
|
||||
? { done: false, value: items[i++]! }
|
||||
: { done: true, value: undefined as unknown as number },
|
||||
});
|
||||
|
||||
const collected: number[] = [];
|
||||
for await (const item of iterateUntilAbort(source)) {
|
||||
collected.push(item);
|
||||
}
|
||||
expect(collected).toEqual(items);
|
||||
});
|
||||
|
||||
it("propagates errors from the underlying iterator.next()", async () => {
|
||||
const source = makeSource<number>({
|
||||
next: async () => {
|
||||
throw new Error("upstream blew up");
|
||||
},
|
||||
});
|
||||
|
||||
await expect(async () => {
|
||||
for await (const _ of iterateUntilAbort(source)) {
|
||||
// no body
|
||||
}
|
||||
}).toThrow("upstream blew up");
|
||||
});
|
||||
|
||||
it("does not leak abort listeners across iterations", async () => {
|
||||
const controller = new AbortController();
|
||||
const addSpy = vi.spyOn(controller.signal, "addEventListener");
|
||||
const removeSpy = vi.spyOn(controller.signal, "removeEventListener");
|
||||
|
||||
const items = [1, 2, 3, 4, 5];
|
||||
let i = 0;
|
||||
const source = makeSource<number>({
|
||||
next: async () =>
|
||||
i < items.length
|
||||
? { done: false, value: items[i++]! }
|
||||
: { done: true, value: undefined as unknown as number },
|
||||
});
|
||||
|
||||
for await (const _ of iterateUntilAbort(source, controller.signal)) {
|
||||
// no body
|
||||
}
|
||||
// Every addEventListener("abort", ...) must be paired with a removeEventListener
|
||||
// call (no leaks across iterations).
|
||||
const adds = addSpy.mock.calls.filter(([type]) => type === "abort").length;
|
||||
const removes = removeSpy.mock.calls.filter(([type]) => type === "abort").length;
|
||||
expect(adds).toBe(removes);
|
||||
expect(adds).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
@@ -22,7 +22,7 @@ import {
|
||||
stripClaudeToolPrefix,
|
||||
} from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import { getEnvApiKey } from "@oh-my-pi/pi-ai/stream";
|
||||
import type { Context, Model, TJsonSchema, TokenTaskBudget, Tool } from "@oh-my-pi/pi-ai/types";
|
||||
import type { AssistantMessage, Context, Model, TJsonSchema, TokenTaskBudget, Tool } from "@oh-my-pi/pi-ai/types";
|
||||
import * as z from "zod/v4";
|
||||
import { withEnv } from "./helpers";
|
||||
|
||||
@@ -301,6 +301,86 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
expect(payload.max_tokens).toBe(8_192);
|
||||
});
|
||||
|
||||
it("keeps the full model output ceiling for API-key requests", async () => {
|
||||
const payload = (await captureAnthropicPayload(
|
||||
{ ...ANTHROPIC_MODEL, id: "claude-opus-4-8", name: "Claude Opus 4.8", maxTokens: 128_000 },
|
||||
{
|
||||
systemPrompt: ["Stay concise."],
|
||||
messages: [{ role: "user", content: "Hi", timestamp: Date.now() }],
|
||||
},
|
||||
{ isOAuth: false },
|
||||
)) as { max_tokens?: number };
|
||||
expect(payload.max_tokens).toBe(128_000);
|
||||
});
|
||||
|
||||
it("does not place cache_control on thinking blocks in the trailing cache window", async () => {
|
||||
const thinkingOnlyAssistant: AssistantMessage = {
|
||||
role: "assistant",
|
||||
content: [{ type: "thinking", thinking: "long deliberation", thinkingSignature: "sig-1" }],
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
model: ANTHROPIC_MODEL.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const payload = (await captureAnthropicPayload(
|
||||
ANTHROPIC_MODEL,
|
||||
{
|
||||
systemPrompt: ["Stay concise."],
|
||||
messages: [{ role: "user", content: "Think about it", timestamp: Date.now() }, thinkingOnlyAssistant],
|
||||
},
|
||||
{ isOAuth: false },
|
||||
)) as { messages?: Array<{ role: string; content: string | Array<{ type: string; cache_control?: unknown }> }> };
|
||||
|
||||
// The thinking-only assistant turn sits inside the trailing two-message
|
||||
// cache window (the Continue. pad is appended after it) but must not get
|
||||
// a breakpoint — Anthropic rejects cache_control on thinking blocks.
|
||||
const assistant = payload.messages?.find(message => message.role === "assistant");
|
||||
expect(Array.isArray(assistant?.content)).toBe(true);
|
||||
for (const block of assistant?.content as Array<{ type: string; cache_control?: unknown }>) {
|
||||
expect(block.cache_control).toBeUndefined();
|
||||
}
|
||||
const last = payload.messages?.at(-1);
|
||||
expect((last?.content as Array<{ cache_control?: unknown }>)[0]?.cache_control).toBeDefined();
|
||||
});
|
||||
|
||||
it("adds effort and mid-conversation betas to API-key requests that use those features", async () => {
|
||||
let capturedBeta: string | undefined;
|
||||
const fetchMock = (async (_input: string | URL | Request, init?: RequestInit) => {
|
||||
capturedBeta = (init?.headers as Record<string, string> | undefined)?.["anthropic-beta"];
|
||||
return new Response(
|
||||
JSON.stringify({ type: "error", error: { type: "invalid_request_error", message: "captured" } }),
|
||||
{ status: 400, headers: { "Content-Type": "application/json" } },
|
||||
);
|
||||
}) as typeof fetch;
|
||||
const adaptiveModel: Model<"anthropic-messages"> = {
|
||||
...ANTHROPIC_MODEL,
|
||||
id: "claude-opus-4-8-20260528",
|
||||
name: "Claude Opus 4.8",
|
||||
thinking: { mode: "anthropic-adaptive", minLevel: Effort.Minimal, maxLevel: Effort.XHigh },
|
||||
};
|
||||
|
||||
await streamAnthropic(
|
||||
adaptiveModel,
|
||||
{ systemPrompt: ["Stay concise."], messages: [{ role: "user", content: "Hi", timestamp: Date.now() }] },
|
||||
{ apiKey: "sk-ant-api-test", thinkingEnabled: false, fetch: fetchMock },
|
||||
).result();
|
||||
|
||||
// thinking-off on an adaptive-only model still pins output_config.effort,
|
||||
// and the converter may emit mid-conversation system turns on Opus 4.8 —
|
||||
// both fields need their betas on API-key requests too.
|
||||
expect(capturedBeta).toContain("effort-2025-11-24");
|
||||
expect(capturedBeta).toContain("mid-conversation-system-2026-04-07");
|
||||
});
|
||||
|
||||
it("billing-header fingerprint uses first user message, not leading developer message", async () => {
|
||||
const userText = "Hello from user with enough chars padding here";
|
||||
|
||||
@@ -397,6 +477,45 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
);
|
||||
});
|
||||
|
||||
it("forwards model-supplied User-Agent on API-key requests", () => {
|
||||
// Direct Anthropic API (X-Api-Key branch).
|
||||
const directHeaders = buildAnthropicHeaders({
|
||||
apiKey: "sk-ant-api-test",
|
||||
isOAuth: false,
|
||||
stream: true,
|
||||
modelHeaders: { "User-Agent": "corp-gateway-client/2.0" },
|
||||
});
|
||||
expect(directHeaders["User-Agent"]).toBe("corp-gateway-client/2.0");
|
||||
|
||||
// Non-Anthropic gateway (Bearer branch).
|
||||
const gatewayHeaders = buildAnthropicHeaders({
|
||||
apiKey: "gateway-token",
|
||||
isOAuth: false,
|
||||
stream: true,
|
||||
baseUrl: "https://gateway.example.com/anthropic",
|
||||
modelHeaders: { "User-Agent": "corp-gateway-client/2.0" },
|
||||
});
|
||||
expect(gatewayHeaders["User-Agent"]).toBe("corp-gateway-client/2.0");
|
||||
});
|
||||
|
||||
it("omits Claude Code betas on API-key requests by default", () => {
|
||||
const headers = buildAnthropicHeaders({
|
||||
apiKey: "sk-ant-api-test",
|
||||
isOAuth: false,
|
||||
stream: true,
|
||||
extraBetas: ["web-search-2025-03-05"],
|
||||
});
|
||||
expect(headers["anthropic-beta"]).toBe("web-search-2025-03-05");
|
||||
|
||||
// And no empty anthropic-beta header when there are no betas at all.
|
||||
const bare = buildAnthropicHeaders({
|
||||
apiKey: "sk-ant-api-test",
|
||||
isOAuth: false,
|
||||
stream: true,
|
||||
});
|
||||
expect(bare["anthropic-beta"]).toBeUndefined();
|
||||
});
|
||||
|
||||
it("skips Claude Code instruction injection for claude-3-5-haiku models", async () => {
|
||||
const payload = (await captureAnthropicPayload(
|
||||
{ ...ANTHROPIC_MODEL, id: "claude-3-5-haiku", name: "Claude 3.5 Haiku" },
|
||||
@@ -1095,6 +1214,76 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
expect(strictNames).toEqual(["python"]);
|
||||
});
|
||||
|
||||
it("demotes allowlisted tools with strict-incompatible schema keywords to non-strict", async () => {
|
||||
const tools: Tool[] = [
|
||||
{
|
||||
name: "edit",
|
||||
description: "Edit a value",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { q: { oneOf: [{ type: "string" }, { type: "integer" }] } },
|
||||
required: ["q"],
|
||||
} as TJsonSchema,
|
||||
},
|
||||
{
|
||||
name: "python",
|
||||
description: "python tool",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { tagged: { type: "object", patternProperties: { "^x-": { type: "string" } } } },
|
||||
required: ["tagged"],
|
||||
} as TJsonSchema,
|
||||
},
|
||||
{
|
||||
name: "find",
|
||||
description: "find tool",
|
||||
parameters: {
|
||||
type: "object",
|
||||
properties: { pattern: { type: "string" } },
|
||||
required: ["pattern"],
|
||||
} as TJsonSchema,
|
||||
},
|
||||
];
|
||||
const payload = (await captureAnthropicPayload(
|
||||
ANTHROPIC_MODEL,
|
||||
{
|
||||
systemPrompt: ["Stay concise."],
|
||||
messages: [{ role: "user", content: "Hi", timestamp: Date.now() }],
|
||||
tools,
|
||||
},
|
||||
{ isOAuth: false },
|
||||
)) as { tools?: Array<{ name?: string; strict?: boolean }> };
|
||||
|
||||
// oneOf/allOf/$ref compile unpredictably under the strict grammar and
|
||||
// patternProperties contradicts the injected additionalProperties:false;
|
||||
// such tools must stay non-strict while clean allowlisted tools keep it.
|
||||
const strictNames = (payload.tools ?? []).filter(tool => tool.strict === true).map(tool => tool.name);
|
||||
expect(strictNames).toEqual(["find"]);
|
||||
});
|
||||
|
||||
it("keeps the interleaved-thinking beta for dated Opus 4.0 ids", () => {
|
||||
const legacy = buildAnthropicClientOptions({
|
||||
model: { ...ANTHROPIC_MODEL, id: "claude-opus-4-20250514", name: "Claude Opus 4" },
|
||||
apiKey: "sk-ant-api-test",
|
||||
extraBetas: [],
|
||||
stream: true,
|
||||
interleavedThinking: true,
|
||||
hasTools: false,
|
||||
});
|
||||
// The date suffix must not parse as minor=20250514 (>= 4.7 display support).
|
||||
expect(legacy.defaultHeaders["anthropic-beta"]).toContain("interleaved-thinking-2025-05-14");
|
||||
|
||||
const modern = buildAnthropicClientOptions({
|
||||
model: { ...ANTHROPIC_MODEL, id: "claude-opus-4-7", name: "Claude Opus 4.7" },
|
||||
apiKey: "sk-ant-api-test",
|
||||
extraBetas: [],
|
||||
stream: true,
|
||||
interleavedThinking: true,
|
||||
hasTools: false,
|
||||
});
|
||||
expect(modern.defaultHeaders["anthropic-beta"] ?? "").not.toContain("interleaved-thinking-2025-05-14");
|
||||
});
|
||||
|
||||
it("adds legacy fine-grained tool-streaming beta only for tool requests on incompatible models", () => {
|
||||
const incompatibleModel: Model<"anthropic-messages"> = {
|
||||
...ANTHROPIC_MODEL,
|
||||
@@ -1126,10 +1315,9 @@ describe("Anthropic request fingerprint alignment", () => {
|
||||
hasTools: true,
|
||||
});
|
||||
|
||||
expect(withoutTools.defaultHeaders["anthropic-beta"]).not.toContain("fine-grained-tool-streaming-2025-05-14");
|
||||
expect(withCompatibleTools.defaultHeaders["anthropic-beta"]).not.toContain(
|
||||
"fine-grained-tool-streaming-2025-05-14",
|
||||
);
|
||||
// No betas at all → the header is omitted entirely on API-key requests.
|
||||
expect(withoutTools.defaultHeaders["anthropic-beta"]).toBeUndefined();
|
||||
expect(withCompatibleTools.defaultHeaders["anthropic-beta"]).toBeUndefined();
|
||||
expect(withIncompatibleTools.defaultHeaders["anthropic-beta"]).toContain(
|
||||
"fine-grained-tool-streaming-2025-05-14",
|
||||
);
|
||||
|
||||
@@ -72,6 +72,25 @@ describe("AnthropicMessagesClient error mapping", () => {
|
||||
expect(error).toBeInstanceOf(AnthropicApiError);
|
||||
expect((error as AnthropicApiError).message).toBe("500 status code (no body)");
|
||||
});
|
||||
|
||||
it("does not let fetchOptions override core request fields", async () => {
|
||||
const { calls, fetch } = createFetchMock([new Response(null, { status: 200 })]);
|
||||
const preAborted = AbortSignal.abort();
|
||||
const client = new AnthropicMessagesClient({
|
||||
apiKey: "sk-test",
|
||||
maxRetries: 0,
|
||||
fetch,
|
||||
fetchOptions: { method: "GET", signal: preAborted },
|
||||
});
|
||||
|
||||
const response = await client.messages.create(params).asResponse();
|
||||
|
||||
// fetchOptions exists for transport extras (tls); a caller-supplied signal
|
||||
// or method must not disconnect the timeout controller or break the POST.
|
||||
expect(response.status).toBe(200);
|
||||
expect(calls[0]?.init.method).toBe("POST");
|
||||
expect(calls[0]?.init.signal?.aborted).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("AnthropicMessagesClient retries", () => {
|
||||
|
||||
@@ -73,6 +73,22 @@ describe("Anthropic mid-conversation system messages", () => {
|
||||
expect(params.at(-1)?.role).toBe("system");
|
||||
});
|
||||
|
||||
it("keeps developer turns carrying image content as user messages", () => {
|
||||
const model = makeModel({ input: ["text", "image"] });
|
||||
const visualDeveloper: DeveloperMessage = {
|
||||
role: "developer",
|
||||
content: [
|
||||
{ type: "text", text: "Match this reference design." },
|
||||
{ type: "image", data: "aGVsbG8=", mimeType: "image/png" },
|
||||
],
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const params = convertAnthropicMessages([user("hi"), visualDeveloper], model, false);
|
||||
// Placement qualifies (follows user, last entry), but system content is
|
||||
// text-only on the wire — the image-bearing turn must stay role: user.
|
||||
expect(params.map(p => p.role)).toEqual(["user", "user"]);
|
||||
});
|
||||
|
||||
it("maps developer messages to system on Claude Mythos 5", () => {
|
||||
const model = makeModel({ id: "claude-mythos-5", name: "Claude Mythos 5" });
|
||||
const params = convertAnthropicMessages([user("hi"), developer("Use project rules.")], model, false);
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { claudeCodeVersion } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import { AnthropicOAuthFlow, refreshAnthropicToken } from "@oh-my-pi/pi-ai/registry/oauth/anthropic";
|
||||
import {
|
||||
buildAnthropicAuthConfig,
|
||||
@@ -201,7 +202,7 @@ describe("anthropic oauth alignment", () => {
|
||||
expect(init?.method).toBe("GET");
|
||||
const headers = init?.headers as Record<string, string> | undefined;
|
||||
expect(headers?.Authorization).toBe("Bearer access-token");
|
||||
expect(headers?.["User-Agent"]).toBe("claude-code/2.1.160");
|
||||
expect(headers?.["User-Agent"]).toBe(`claude-code/${claudeCodeVersion}`);
|
||||
expect(headers?.["anthropic-beta"]).toBe("oauth-2025-04-20");
|
||||
return new Response(
|
||||
JSON.stringify({
|
||||
|
||||
@@ -51,6 +51,44 @@ describe("Anthropic assistant-prefill fallback", () => {
|
||||
expect(params.at(-1)?.content).toBe("Continue.");
|
||||
});
|
||||
|
||||
it("repairs consecutive assistant turns left by dropped empty user messages", () => {
|
||||
const assistant = (text: string): AssistantMessage => ({
|
||||
role: "assistant",
|
||||
content: [{ type: "text", text }],
|
||||
api: "anthropic-messages",
|
||||
provider: "anthropic",
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
// An empty nudge submission is dropped by the converter, which would leave
|
||||
// the two assistant turns adjacent — Anthropic 400s on that shape.
|
||||
const emptyNudge: UserMessage = { role: "user", content: [{ type: "text", text: "" }], timestamp: Date.now() };
|
||||
|
||||
const params = convertAnthropicMessages(
|
||||
[
|
||||
{ role: "user", content: "answer me", timestamp: Date.now() },
|
||||
assistant("partial answer"),
|
||||
emptyNudge,
|
||||
assistant("full answer"),
|
||||
{ role: "user", content: "thanks", timestamp: Date.now() },
|
||||
],
|
||||
model,
|
||||
false,
|
||||
);
|
||||
|
||||
expect(params.map(p => p.role)).toEqual(["user", "assistant", "user", "assistant", "user"]);
|
||||
expect(params[2]?.content).toBe("Continue.");
|
||||
});
|
||||
|
||||
it("does not append Continue. when the last turn is already user", () => {
|
||||
const params = convertAnthropicMessages(
|
||||
[
|
||||
|
||||
@@ -136,7 +136,7 @@ function getStrictFlags(params: unknown): boolean[] {
|
||||
|
||||
function createTextSuccessEvents(
|
||||
text: string,
|
||||
options: { duplicateMessageStart?: boolean } = {},
|
||||
options: { duplicateMessageStart?: boolean; stopReason?: string } = {},
|
||||
): MockAnthropicEvent[] {
|
||||
const events: MockAnthropicEvent[] = [
|
||||
{
|
||||
@@ -156,7 +156,7 @@ function createTextSuccessEvents(
|
||||
{ type: "content_block_stop", index: 0 },
|
||||
{
|
||||
type: "message_delta",
|
||||
delta: { stop_reason: "end_turn" },
|
||||
delta: { stop_reason: options.stopReason ?? "end_turn" },
|
||||
usage: {
|
||||
input_tokens: 12,
|
||||
output_tokens: 4,
|
||||
@@ -278,6 +278,45 @@ describe("anthropic stream envelope handling", () => {
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("drops replayed closed blocks after a duplicate message_start instead of duplicating content", async () => {
|
||||
const events: MockAnthropicEvent[] = [
|
||||
{
|
||||
type: "message_start",
|
||||
message: { id: "msg_first", usage: { input_tokens: 12, output_tokens: 0 } },
|
||||
},
|
||||
{ type: "content_block_start", index: 0, content_block: { type: "text", text: "" } },
|
||||
{ type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "hello" } },
|
||||
{ type: "content_block_stop", index: 0 },
|
||||
// A replaying proxy splices the same envelope again before the
|
||||
// terminal message_delta arrives.
|
||||
{ type: "message_start", message: { id: "msg_replay", usage: { input_tokens: 12, output_tokens: 0 } } },
|
||||
{ type: "content_block_start", index: 0, content_block: { type: "text", text: "" } },
|
||||
{ type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "hello" } },
|
||||
{ type: "content_block_stop", index: 0 },
|
||||
{
|
||||
type: "message_delta",
|
||||
delta: { stop_reason: "end_turn" },
|
||||
usage: { input_tokens: 12, output_tokens: 4 },
|
||||
},
|
||||
{ type: "message_stop" },
|
||||
];
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => createMockRequest(events) as never);
|
||||
|
||||
const stream = streamAnthropic(model, context, { apiKey: "sk-ant-test" });
|
||||
const collected: AssistantMessageEvent[] = [];
|
||||
for await (const event of stream) {
|
||||
collected.push(event);
|
||||
}
|
||||
const result = await stream.result();
|
||||
|
||||
expect(countEvents(collected, "text_start")).toBe(1);
|
||||
expect(countEvents(collected, "text_end")).toBe(1);
|
||||
expect(countEvents(collected, "error")).toBe(0);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.responseId).toBe("msg_first");
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("ignores ping before message_start and streams the response once", async () => {
|
||||
let attempt = 0;
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => {
|
||||
@@ -303,6 +342,107 @@ describe("anthropic stream envelope handling", () => {
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("maps model_context_window_exceeded to a length stop", async () => {
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(
|
||||
() =>
|
||||
createMockRequest(
|
||||
createTextSuccessEvents("hello", { stopReason: "model_context_window_exceeded" }),
|
||||
) as never,
|
||||
);
|
||||
|
||||
const stream = streamAnthropic(model, context, { apiKey: "sk-ant-test" });
|
||||
const events: AssistantMessageEvent[] = [];
|
||||
for await (const event of stream) {
|
||||
events.push(event);
|
||||
}
|
||||
const result = await stream.result();
|
||||
|
||||
expect(countEvents(events, "error")).toBe(0);
|
||||
expect(countEvents(events, "done")).toBe(1);
|
||||
expect(result.stopReason).toBe("length");
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("completes the turn instead of failing when the API sends an unknown stop reason", async () => {
|
||||
let attempt = 0;
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => {
|
||||
attempt += 1;
|
||||
return createMockRequest(createTextSuccessEvents("hello", { stopReason: "weird_new_reason" })) as never;
|
||||
});
|
||||
|
||||
const stream = streamAnthropic(model, context, { apiKey: "sk-ant-test" });
|
||||
const events: AssistantMessageEvent[] = [];
|
||||
for await (const event of stream) {
|
||||
events.push(event);
|
||||
}
|
||||
const result = await stream.result();
|
||||
|
||||
// The unknown reason arrives after all content streamed; it must not burn
|
||||
// a retry or surface as an error.
|
||||
expect(attempt).toBe(1);
|
||||
expect(countEvents(events, "error")).toBe(0);
|
||||
expect(countEvents(events, "done")).toBe(1);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("ignores a spliced second envelope's message_delta after the terminal stop", async () => {
|
||||
const events: MockAnthropicEvent[] = [
|
||||
...createTextSuccessEvents("hello"),
|
||||
// Transparent reconnect splices a fresh envelope onto the same stream.
|
||||
{ type: "message_start", message: { id: "msg_second", usage: { input_tokens: 99, output_tokens: 99 } } },
|
||||
{ type: "message_delta", delta: { stop_reason: "tool_use" }, usage: { input_tokens: 99, output_tokens: 99 } },
|
||||
{ type: "message_stop" },
|
||||
];
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => createMockRequest(events) as never);
|
||||
|
||||
const stream = streamAnthropic(model, context, { apiKey: "sk-ant-test" });
|
||||
const collected: AssistantMessageEvent[] = [];
|
||||
for await (const event of stream) {
|
||||
collected.push(event);
|
||||
}
|
||||
const result = await stream.result();
|
||||
|
||||
// The completed first envelope owns the stop reason and usage; the splice
|
||||
// must not relabel a finished turn or overwrite its counters.
|
||||
expect(countEvents(collected, "error")).toBe(0);
|
||||
expect(countEvents(collected, "done")).toBe(1);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.usage.output).toBe(4);
|
||||
expect(result.responseId).toBe("msg_text_success");
|
||||
expect(result.content).toEqual([{ type: "text", text: "hello" }]);
|
||||
});
|
||||
|
||||
it("tolerates envelopes missing usage and delta payloads", async () => {
|
||||
const events: MockAnthropicEvent[] = [
|
||||
{ type: "message_start", message: { id: "msg_lenient" } },
|
||||
{ type: "content_block_start", index: 0, content_block: { type: "text", text: "" } },
|
||||
{ type: "content_block_delta", index: 0 },
|
||||
{ type: "content_block_delta", index: 0, delta: { type: "text_delta", text: "hi" } },
|
||||
{ type: "content_block_stop", index: 0 },
|
||||
{ type: "message_delta" },
|
||||
{ type: "message_delta", delta: { stop_reason: "end_turn" } },
|
||||
{ type: "message_stop" },
|
||||
];
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => createMockRequest(events) as never);
|
||||
|
||||
const stream = streamAnthropic(model, context, { apiKey: "sk-ant-test" });
|
||||
const collected: AssistantMessageEvent[] = [];
|
||||
for await (const event of stream) {
|
||||
collected.push(event);
|
||||
}
|
||||
const result = await stream.result();
|
||||
|
||||
// Proxies that omit usage/delta objects must degrade to anomaly logs, not
|
||||
// TypeErrors that fail the turn.
|
||||
expect(countEvents(collected, "error")).toBe(0);
|
||||
expect(countEvents(collected, "done")).toBe(1);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.responseId).toBe("msg_lenient");
|
||||
expect(result.content).toEqual([{ type: "text", text: "hi" }]);
|
||||
});
|
||||
|
||||
it("ignores unknown preamble events before message_start and streams the response once", async () => {
|
||||
let attempt = 0;
|
||||
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => {
|
||||
@@ -444,7 +584,7 @@ describe("anthropic stream envelope handling", () => {
|
||||
const result = await stream.result();
|
||||
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toContain("compiled grammar is too large");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.content).toEqual([{ type: "text", text: "recovered" }]);
|
||||
expect(countEvents(events, "done")).toBe(1);
|
||||
expect(countEvents(events, "error")).toBe(0);
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { afterEach, describe, expect, it, vi } from "bun:test";
|
||||
import { streamAnthropic } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type { AnthropicMessagesClientLike } from "@oh-my-pi/pi-ai/providers/anthropic-client";
|
||||
import { AnthropicApiError, type AnthropicMessagesClientLike } from "@oh-my-pi/pi-ai/providers/anthropic-client";
|
||||
import type { Context, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import { waitForDelayOrAbort } from "./helpers";
|
||||
|
||||
@@ -136,6 +136,14 @@ function createAnthropicMockStream({
|
||||
};
|
||||
}
|
||||
|
||||
function createRejectedAnthropicRequest(error: Error): MockAnthropicRequest {
|
||||
return {
|
||||
async withResponse() {
|
||||
throw error;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
type PromiseOutcome<T> = { kind: "fulfilled"; value: T } | { kind: "rejected"; error: unknown };
|
||||
|
||||
async function drainMicrotasksUntil(predicate: () => boolean, errorMessage: string): Promise<void> {
|
||||
@@ -225,6 +233,54 @@ describe("anthropic first-event timeout retries", () => {
|
||||
expect(result.responseId).toBe("msg_retry_success");
|
||||
});
|
||||
|
||||
it("keeps the first-event watchdog armed when only pings arrive before message_start", async () => {
|
||||
vi.useFakeTimers();
|
||||
let attempt = 0;
|
||||
let firstAttemptIteratorStarted = false;
|
||||
const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => {
|
||||
attempt += 1;
|
||||
return createAnthropicMockStream({
|
||||
signal: requestOptions?.signal,
|
||||
events: attempt === 1 ? [{ type: "ping" }] : createSuccessfulAnthropicEvents("retry recovered"),
|
||||
hangAfterEvents: attempt === 1,
|
||||
onIteratorStart:
|
||||
attempt === 1
|
||||
? () => {
|
||||
firstAttemptIteratorStarted = true;
|
||||
}
|
||||
: undefined,
|
||||
}) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const client = { messages: { create } } as AnthropicMessagesClientLike;
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const resultPromise = streamAnthropic(model, context, {
|
||||
client,
|
||||
streamFirstEventTimeoutMs: 1,
|
||||
streamIdleTimeoutMs: 60_000,
|
||||
providerRetryWait,
|
||||
}).result();
|
||||
|
||||
await drainMicrotasksUntil(
|
||||
() => firstAttemptIteratorStarted,
|
||||
"Anthropic mock stream did not enter the ping-then-hang first attempt",
|
||||
);
|
||||
await drainMicrotasksUntil(() => vi.getTimerCount() > 0, "Anthropic watchdog timer was not armed");
|
||||
|
||||
// A keepalive must not consume the first-event watchdog: if it did, the
|
||||
// stall would be classified as a (non-retryable) 60s idle timeout and
|
||||
// advancing 1ms would never settle the stream.
|
||||
vi.advanceTimersByTime(1);
|
||||
const result = await resolveAfterMicrotasks(
|
||||
resultPromise,
|
||||
"Anthropic ping-then-stall did not retry via the first-event watchdog",
|
||||
);
|
||||
|
||||
expect(attempt).toBe(2);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.content).toEqual([{ type: "text", text: "retry recovered" }]);
|
||||
});
|
||||
|
||||
it("does not arm the Anthropic first-event watchdog before the stream connects", async () => {
|
||||
let seenRequestTimeout: number | undefined;
|
||||
let seenRequestMaxRetries: number | undefined;
|
||||
@@ -365,3 +421,35 @@ describe("anthropic first-event timeout retries", () => {
|
||||
]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("anthropic provider retry delays", () => {
|
||||
it("waits at least the server-suggested retry-after before retrying a retryable API error", async () => {
|
||||
let attempt = 0;
|
||||
const create = ((_body: unknown, requestOptions?: { signal?: AbortSignal }) => {
|
||||
attempt += 1;
|
||||
if (attempt === 1) {
|
||||
return createRejectedAnthropicRequest(
|
||||
new AnthropicApiError(
|
||||
529,
|
||||
'529 {"type":"error","error":{"type":"overloaded_error","message":"Overloaded"}}',
|
||||
new Headers({ "retry-after": "30" }),
|
||||
),
|
||||
) as never;
|
||||
}
|
||||
return createAnthropicMockStream({
|
||||
signal: requestOptions?.signal,
|
||||
events: createSuccessfulAnthropicEvents("after backoff"),
|
||||
}) as never;
|
||||
}) as unknown as AnthropicMessagesClientLike["messages"]["create"];
|
||||
const client = { messages: { create } } as AnthropicMessagesClientLike;
|
||||
const providerRetryWait = vi.fn(async () => {});
|
||||
|
||||
const result = await streamAnthropic(model, context, { client, providerRetryWait }).result();
|
||||
|
||||
// Header says 30s; the 2s exponential backoff must not undercut it.
|
||||
expect(attempt).toBe(2);
|
||||
expect(providerRetryWait).toHaveBeenCalledWith(30_000, undefined);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.content).toEqual([{ type: "text", text: "after backoff" }]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -93,6 +93,47 @@ describe("Anthropic-compatible unsigned thinking replay (#2005)", () => {
|
||||
expect(blocks[1]).toEqual({ type: "text", text: "Sure." });
|
||||
});
|
||||
|
||||
it("sanitizes lone surrogates in cross-API tool arguments only", () => {
|
||||
const loneSurrogate = "broken \ud83d end";
|
||||
const makeToolCallAssistant = (api: AssistantMessage["api"]): AssistantMessage => ({
|
||||
role: "assistant",
|
||||
content: [
|
||||
{
|
||||
type: "toolCall",
|
||||
id: "call_1",
|
||||
name: "write",
|
||||
arguments: { text: loneSurrogate, nested: { parts: [loneSurrogate] } },
|
||||
},
|
||||
],
|
||||
api,
|
||||
provider: api === "anthropic-messages" ? "custom-anthropic" : "openai",
|
||||
model: "reasoning-model",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp: 0,
|
||||
});
|
||||
|
||||
// Cross-API replay: Anthropic's strict UTF-8 validation rejects lone
|
||||
// surrogates, so string leaves are deep-sanitized.
|
||||
const crossBlocks = assistantWireBlocks([makeUser(), makeToolCallAssistant("openai-responses")], makeModel());
|
||||
const crossToolUse = crossBlocks.find(block => block.type === "tool_use") as WireToolUseBlock;
|
||||
expect(crossToolUse.input.text).toBe("broken \ufffd end");
|
||||
expect((crossToolUse.input.nested as { parts: string[] }).parts[0]).toBe("broken \ufffd end");
|
||||
|
||||
// Same-API replay stays byte-identical (the args came from Anthropic's own
|
||||
// JSON; rewriting them would destabilize prompt-cache prefixes).
|
||||
const sameBlocks = assistantWireBlocks([makeUser(), makeToolCallAssistant("anthropic-messages")], makeModel());
|
||||
const sameToolUse = sameBlocks.find(block => block.type === "tool_use") as WireToolUseBlock;
|
||||
expect(sameToolUse.input.text).toBe(loneSurrogate);
|
||||
});
|
||||
|
||||
it("covers the Xiaomi MiMo Anthropic-compatible reporter configuration without provider allowlists", () => {
|
||||
const model = makeModel({
|
||||
provider: "user-custom",
|
||||
|
||||
@@ -237,6 +237,37 @@ describe("anthropic-messages parseRequest", () => {
|
||||
expect(withMetadata.options.extra).toBeUndefined();
|
||||
expect(withMetadata.options.metadata).toEqual({ user_id: "u_1" });
|
||||
});
|
||||
|
||||
it("rejects malformed known-type blocks instead of passing them through the unknown-block catch-all", () => {
|
||||
// `{type:"text", text: 123}` fails the typed schema and must not fall
|
||||
// into the loose catch-all (would corrupt history and TypeError downstream).
|
||||
expect(() =>
|
||||
parseRequest({
|
||||
model: "m",
|
||||
max_tokens: 1,
|
||||
messages: [{ role: "user", content: [{ type: "text", text: 123 }] }],
|
||||
}),
|
||||
).toThrow();
|
||||
expect(() =>
|
||||
parseRequest({
|
||||
model: "m",
|
||||
max_tokens: 1,
|
||||
messages: [
|
||||
{ role: "user", content: "hi" },
|
||||
{ role: "assistant", content: [{ type: "tool_use", id: "", name: "lookup" }] },
|
||||
],
|
||||
}),
|
||||
).toThrow();
|
||||
// Genuinely unknown variants are still accepted and flattened.
|
||||
const unknown = parseRequest({
|
||||
model: "m",
|
||||
max_tokens: 1,
|
||||
messages: [
|
||||
{ role: "user", content: [{ type: "web_search_tool_result", tool_use_id: "srvtoolu_1", content: [] }] },
|
||||
],
|
||||
});
|
||||
expect(unknown.context.messages).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("anthropic-messages encodeResponse", () => {
|
||||
@@ -466,4 +497,11 @@ describe("anthropic-messages encodeStream", () => {
|
||||
expect(last.event).toBe("error");
|
||||
expect(last.data).toEqual({ type: "error", error: { type: "api_error", message: "boom" } });
|
||||
});
|
||||
|
||||
it("emits a complete envelope when the stream ends without an explicit done", async () => {
|
||||
const sse = await collectSse(encodeStream(makeStream([]), "m"));
|
||||
expect(sse.map(e => e.event)).toEqual(["message_start", "message_delta", "message_stop"]);
|
||||
const delta = sse[1]!.data as { delta: { stop_reason: string } };
|
||||
expect(delta.delta.stop_reason).toBe("end_turn");
|
||||
});
|
||||
});
|
||||
|
||||
@@ -469,6 +469,37 @@ describe("AuthStorage openai-codex email dedupe", () => {
|
||||
}
|
||||
});
|
||||
|
||||
it("reopens a current-schema db without issuing write transactions", async () => {
|
||||
if (!tempDir) throw new Error("test setup failed");
|
||||
|
||||
const reopenDbPath = path.join(tempDir, "reopen-noop-agent.db");
|
||||
const first = await SqliteAuthCredentialStore.open(reopenDbPath);
|
||||
// api_key rows never derive an identity_key, so this leaves a NULL row
|
||||
// the boot-time backfill scan must skip without a no-op UPDATE.
|
||||
first.saveApiKey("openai", "sk-reopen-noop");
|
||||
first.close();
|
||||
|
||||
// PRAGMA data_version, read from a second connection, increments whenever
|
||||
// another connection commits a write; reopening a current-schema store
|
||||
// (already-WAL pragmas, IF NOT EXISTS DDL, current version row, and an
|
||||
// underivable NULL identity_key row) must not move it.
|
||||
const observer = new Database(reopenDbPath, { readonly: true });
|
||||
try {
|
||||
const before = (observer.prepare("PRAGMA data_version").get() as { data_version: number }).data_version;
|
||||
const reopened = await SqliteAuthCredentialStore.open(reopenDbPath);
|
||||
try {
|
||||
expect(reopened.listAuthCredentials("openai")).toHaveLength(1);
|
||||
expect(readAuthSchemaVersion(reopenDbPath)).toBe(4);
|
||||
} finally {
|
||||
reopened.close();
|
||||
}
|
||||
const after = (observer.prepare("PRAGMA data_version").get() as { data_version: number }).data_version;
|
||||
expect(after).toBe(before);
|
||||
} finally {
|
||||
observer.close();
|
||||
}
|
||||
});
|
||||
|
||||
it("migrates v3 auth schema away from unixepoch defaults", async () => {
|
||||
if (!tempDir) throw new Error("test setup failed");
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { claudeCodeVersion } from "@oh-my-pi/pi-ai/providers/anthropic";
|
||||
import type { UsageFetchContext } from "@oh-my-pi/pi-ai/usage";
|
||||
import { claudeUsageProvider } from "@oh-my-pi/pi-ai/usage/claude";
|
||||
|
||||
@@ -75,7 +76,7 @@ describe("claude usage request headers", () => {
|
||||
|
||||
const headers = calls[0]?.init?.headers;
|
||||
expect(getHeaderCaseInsensitive(headers, "authorization")).toBe(`Bearer ${token}`);
|
||||
expect(getHeaderCaseInsensitive(headers, "user-agent")).toBe("claude-cli/2.1.160 (external, cli)");
|
||||
expect(getHeaderCaseInsensitive(headers, "user-agent")).toBe(`claude-cli/${claudeCodeVersion} (external, cli)`);
|
||||
|
||||
const beta = getHeaderCaseInsensitive(headers, "anthropic-beta");
|
||||
expect(beta).toBeDefined();
|
||||
|
||||
@@ -120,6 +120,57 @@ describe("callWithCopilotModelRetry", () => {
|
||||
expect(calls).toBe(1);
|
||||
});
|
||||
|
||||
it("does not blind-retry a 429 that carries no Retry-After guidance", async () => {
|
||||
let calls = 0;
|
||||
const err = copilotError({ status: 429, message: "rate limited" });
|
||||
await expect(
|
||||
callWithCopilotModelRetry(
|
||||
async () => {
|
||||
calls += 1;
|
||||
throw err;
|
||||
},
|
||||
{ provider: "github-copilot", retryBaseDelayMs: 0 },
|
||||
),
|
||||
).rejects.toBe(err);
|
||||
expect(calls).toBe(1);
|
||||
});
|
||||
|
||||
it("honors Retry-After on a 429 and retries", async () => {
|
||||
let calls = 0;
|
||||
const result = await callWithCopilotModelRetry(
|
||||
async () => {
|
||||
calls += 1;
|
||||
if (calls === 1) {
|
||||
const err = copilotError({ status: 429, message: "rate limited" });
|
||||
(err as unknown as { headers: Record<string, string> }).headers = { "retry-after": "0.01" };
|
||||
throw err;
|
||||
}
|
||||
return "ok" as const;
|
||||
},
|
||||
{ provider: "github-copilot", retryBaseDelayMs: 0 },
|
||||
);
|
||||
expect(result).toBe("ok");
|
||||
expect(calls).toBe(2);
|
||||
});
|
||||
|
||||
it("still retries status-less transport blips with the linear backoff", async () => {
|
||||
let calls = 0;
|
||||
const result = await callWithCopilotModelRetry(
|
||||
async () => {
|
||||
calls += 1;
|
||||
if (calls === 1) {
|
||||
throw new Error(
|
||||
'HTTP2StreamReset fetching "https://api.example.com/x". For more information, pass `verbose: true` in the second argument to fetch()',
|
||||
);
|
||||
}
|
||||
return "ok" as const;
|
||||
},
|
||||
{ provider: "github-copilot", retryBaseDelayMs: 0 },
|
||||
);
|
||||
expect(result).toBe("ok");
|
||||
expect(calls).toBe(2);
|
||||
});
|
||||
|
||||
it("stops retrying when the caller aborts during backoff", async () => {
|
||||
const controller = new AbortController();
|
||||
controller.abort();
|
||||
|
||||
@@ -33,4 +33,18 @@ describe("AssistantMessageEventStream", () => {
|
||||
expect(stream.queue[0]).toMatchObject({ type: "text_delta", delta: "a" });
|
||||
expect(stream.queue[1]).toMatchObject({ type: "text_delta", delta: "b" });
|
||||
});
|
||||
|
||||
it("rejects result() when ended without a terminal value", async () => {
|
||||
const stream = new AssistantMessageEventStream();
|
||||
stream.end();
|
||||
await expect(stream.result()).rejects.toThrow(/ended without a final result/);
|
||||
});
|
||||
|
||||
it("keeps the pushed terminal result when end() follows a done event", async () => {
|
||||
const stream = new AssistantMessageEventStream();
|
||||
const message = createPartial("final");
|
||||
stream.push({ type: "done", reason: "stop", message });
|
||||
stream.end();
|
||||
await expect(stream.result()).resolves.toBe(message);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -224,6 +224,38 @@ describe("Anthropic Copilot auth config", () => {
|
||||
expect(result.baseURL).toBe("http://127.0.0.1:8317");
|
||||
});
|
||||
|
||||
it("sends Content-Type and anthropic-version on Copilot anthropic requests", () => {
|
||||
const result = buildAnthropicClientOptions({
|
||||
model: makeCopilotClaudeModel(),
|
||||
apiKey: "ghu_test",
|
||||
extraBetas: [],
|
||||
stream: true,
|
||||
dynamicHeaders: {},
|
||||
});
|
||||
|
||||
// The client posts JSON.stringify(params); without these the request goes
|
||||
// out with no Content-Type at all (Bun does not default it for string
|
||||
// bodies when a plain headers object is supplied).
|
||||
expect(result.defaultHeaders["Content-Type"]).toBe("application/json");
|
||||
expect(result.defaultHeaders["anthropic-version"]).toBe("2023-06-01");
|
||||
});
|
||||
|
||||
it("merges Copilot headers case-insensitively so auth headers cannot duplicate", () => {
|
||||
const result = buildAnthropicClientOptions({
|
||||
model: { ...makeCopilotClaudeModel(), headers: { ...OPENCODE_HEADERS, authorization: "Bearer override" } },
|
||||
apiKey: "ghu_test",
|
||||
extraBetas: [],
|
||||
stream: true,
|
||||
dynamicHeaders: {},
|
||||
});
|
||||
|
||||
// A miscased duplicate would survive Object.assign and the Headers
|
||||
// constructor then joins both values comma-separated on the wire.
|
||||
const authKeys = Object.keys(result.defaultHeaders).filter(key => key.toLowerCase() === "authorization");
|
||||
expect(authKeys).toHaveLength(1);
|
||||
expect(result.defaultHeaders[authKeys[0]]).toBe("Bearer override");
|
||||
});
|
||||
|
||||
it("builds anthropic auth URLs from the normalized service root", () => {
|
||||
const url = buildAnthropicUrl({
|
||||
apiKey: "test-key",
|
||||
|
||||
@@ -474,6 +474,29 @@ describe("model thinking runtime helpers", () => {
|
||||
);
|
||||
});
|
||||
|
||||
it("exposes xhigh for OpenRouter-hosted Anthropic adaptive models", () => {
|
||||
const fable = createModel({
|
||||
id: "anthropic/claude-fable-5",
|
||||
api: "openai-completions",
|
||||
provider: "openrouter",
|
||||
});
|
||||
const opus46 = createModel({
|
||||
id: "anthropic/claude-opus-4.6",
|
||||
api: "openai-completions",
|
||||
provider: "openrouter",
|
||||
});
|
||||
const sonnet46 = createModel({
|
||||
id: "anthropic/claude-sonnet-4.6",
|
||||
api: "openai-completions",
|
||||
provider: "openrouter",
|
||||
});
|
||||
|
||||
expect(fable.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(opus46.thinking?.maxLevel).toBe(Effort.XHigh);
|
||||
expect(sonnet46.thinking?.maxLevel).toBe(Effort.High);
|
||||
expect(requireSupportedEffort(fable, Effort.XHigh)).toBe(Effort.XHigh);
|
||||
});
|
||||
|
||||
it("enables xhigh for openai-responses and openai-codex-responses APIs", () => {
|
||||
const responsesModel = createModel({
|
||||
id: "custom-responses",
|
||||
|
||||
@@ -2201,6 +2201,16 @@ describe("openai-codex streaming", () => {
|
||||
send(): void {
|
||||
if (this.#index === 0) {
|
||||
// First attempt: a function call whose arguments are only whitespace.
|
||||
// A completed reasoning item lands in nativeOutputItems before the
|
||||
// degenerate tool call begins; it must not survive the retry.
|
||||
this.sendJson({
|
||||
type: "response.output_item.added",
|
||||
item: { type: "reasoning", id: "rs_stale", summary: [] },
|
||||
});
|
||||
this.sendJson({
|
||||
type: "response.output_item.done",
|
||||
item: { type: "reasoning", id: "rs_stale", summary: [{ type: "summary_text", text: "stale" }] },
|
||||
});
|
||||
this.sendJson({
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", id: "fc_ws", call_id: "call_ws", name: "todo", arguments: "" },
|
||||
@@ -2269,6 +2279,350 @@ describe("openai-codex streaming", () => {
|
||||
expect(toolCall.name).toBe("todo");
|
||||
expect(toolCall.id).toBe("call_ws|fc_ws");
|
||||
expect(toolCall.arguments).toEqual({ ops: [{ op: "start", task: "x" }] });
|
||||
// Native items from the abandoned first attempt must not leak into the
|
||||
// replayed turn's history payload (stale reasoning would be re-sent as
|
||||
// input on the next request).
|
||||
const payload = result.providerPayload as { items?: Array<{ id?: string }> } | undefined;
|
||||
const payloadIds = (payload?.items ?? []).map(item => item.id);
|
||||
expect(payloadIds).toContain("fc_ws");
|
||||
expect(payloadIds).not.toContain("rs_stale");
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("interrupts whitespace-only custom tool input deltas", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
const fetchMock = vi.fn(async () => {
|
||||
throw new Error("SSE fallback should not run for degenerate custom tool input");
|
||||
});
|
||||
|
||||
let sendCount = 0;
|
||||
class WhitespaceCustomInputWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(): void {
|
||||
sendCount += 1;
|
||||
this.sendJson({
|
||||
type: "response.output_item.added",
|
||||
item: { type: "custom_tool_call", id: "ctc_ws", call_id: "call_ctc_ws", name: "apply_patch", input: "" },
|
||||
});
|
||||
for (let sequence = 1; sequence <= 300; sequence += 1) {
|
||||
this.sendJson({
|
||||
type: "response.custom_tool_call_input.delta",
|
||||
delta: sequence % 2 === 0 ? " ".repeat(64) : "\t",
|
||||
item_id: "ctc_ws",
|
||||
output_index: 0,
|
||||
sequence_number: sequence,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
global.WebSocket = WhitespaceCustomInputWebSocket as unknown as typeof WebSocket;
|
||||
|
||||
const model = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-whitespace-custom-input-session",
|
||||
providerSessionState,
|
||||
}).result();
|
||||
|
||||
// One initial attempt + CODEX_WHITESPACE_LOOP_RETRY_LIMIT (2) bounded retries.
|
||||
expect(sendCount).toBe(3);
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toContain("whitespace-only tool-call argument delta");
|
||||
expect(result.errorMessage).toContain("ctc_ws");
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("delivers a queued terminal event when the server closes immediately after it", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
const fetchMock = vi.fn(async () => {
|
||||
throw new Error("SSE fallback should not run when the response completed");
|
||||
});
|
||||
|
||||
let constructorCount = 0;
|
||||
class EagerCloseWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
constructorCount += 1;
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(): void {
|
||||
// Every frame lands in the connection queue synchronously, before the
|
||||
// consumer microtask drains any of them; the close event used to wipe
|
||||
// the queued terminal event and turn success into a transport error.
|
||||
this.emitCodexResponse({ messageId: "msg_eager", responseId: "resp_eager", text: "Hello eager" });
|
||||
this.readyState = MockWebSocket.CLOSED;
|
||||
this.emit("close", { code: 1000 } as unknown as Event);
|
||||
}
|
||||
}
|
||||
global.WebSocket = EagerCloseWebSocket as unknown as typeof WebSocket;
|
||||
|
||||
const model = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-eager-close-session",
|
||||
providerSessionState,
|
||||
}).result();
|
||||
|
||||
expect(constructorCount).toBe(1);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.content).toEqual([expect.objectContaining({ type: "text", text: "Hello eager" })]);
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("surfaces a connection-limit error instead of replaying a delivered tool call over SSE", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
const fetchMock = vi.fn(async () => {
|
||||
throw new Error("SSE replay must not run after a toolcall_end was delivered");
|
||||
});
|
||||
|
||||
let constructorCount = 0;
|
||||
class ConnectionLimitWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
constructorCount += 1;
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(): void {
|
||||
this.sendJson({
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", id: "fc_limit", call_id: "call_limit", name: "todo", arguments: "" },
|
||||
});
|
||||
this.sendJson({
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", id: "fc_limit", call_id: "call_limit", name: "todo", arguments: "{}" },
|
||||
});
|
||||
this.sendJson({
|
||||
type: "error",
|
||||
code: "websocket_connection_limit_reached",
|
||||
message: "connection limit reached",
|
||||
});
|
||||
}
|
||||
}
|
||||
global.WebSocket = ConnectionLimitWebSocket as unknown as typeof WebSocket;
|
||||
|
||||
const model = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-connection-limit-toolcall-session",
|
||||
providerSessionState,
|
||||
}).result();
|
||||
|
||||
expect(constructorCount).toBe(1);
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toContain("connection limit reached");
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("joins an in-flight websocket handshake instead of tearing it down", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
const fetchMock = vi.fn(async () => {
|
||||
throw new Error("SSE fallback should not run when the handshake is joined");
|
||||
});
|
||||
|
||||
let constructorCount = 0;
|
||||
const sockets: DeferredOpenWebSocket[] = [];
|
||||
class DeferredOpenWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
constructorCount += 1;
|
||||
sockets.push(this);
|
||||
}
|
||||
|
||||
open(): void {
|
||||
this.readyState = MockWebSocket.OPEN;
|
||||
this.emit("open", new Event("open"));
|
||||
}
|
||||
|
||||
close(): void {
|
||||
const wasPending = this.readyState === MockWebSocket.CONNECTING;
|
||||
super.close();
|
||||
if (wasPending) this.emit("close", { code: 1000 } as unknown as Event);
|
||||
}
|
||||
|
||||
send(): void {
|
||||
this.emitCodexResponse({ messageId: "msg_join", responseId: "resp_join", text: "Joined" });
|
||||
}
|
||||
}
|
||||
global.WebSocket = DeferredOpenWebSocket as unknown as typeof WebSocket;
|
||||
|
||||
const model = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const providerSessionState = new Map<string, ProviderSessionState>();
|
||||
// Prewarm starts the handshake; the stream call races it before the socket
|
||||
// opens. Tearing down the CONNECTING socket would reject the prewarm with a
|
||||
// fatal "websocket closed before open" and disable websockets for the session.
|
||||
const prewarmPromise = prewarmOpenAICodexResponses(model, {
|
||||
apiKey: token,
|
||||
sessionId: "ws-join-session",
|
||||
providerSessionState,
|
||||
});
|
||||
const streamResult = streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-join-session",
|
||||
providerSessionState,
|
||||
}).result();
|
||||
|
||||
// Let both callers reach the handshake before the socket opens.
|
||||
await Bun.sleep(5);
|
||||
for (const socket of sockets) socket.open();
|
||||
|
||||
await prewarmPromise;
|
||||
const result = await streamResult;
|
||||
|
||||
expect(constructorCount).toBe(1);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.content).toEqual([expect.objectContaining({ type: "text", text: "Joined" })]);
|
||||
const details = getOpenAICodexTransportDetails(model, {
|
||||
sessionId: "ws-join-session",
|
||||
providerSessionState,
|
||||
});
|
||||
expect(details.websocketDisabled).toBe(false);
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("bounds connection-limit reconnects and replays over SSE when the budget is exhausted", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
Bun.env.PI_CODEX_WEBSOCKET_RETRY_BUDGET = "2";
|
||||
Bun.env.PI_CODEX_WEBSOCKET_RETRY_DELAY_MS = "1";
|
||||
const token = createCodexTestToken();
|
||||
|
||||
const sse = `${[
|
||||
`data: ${JSON.stringify({
|
||||
type: "response.output_item.added",
|
||||
item: { type: "message", id: "msg_sse", role: "assistant", status: "in_progress", content: [] },
|
||||
})}`,
|
||||
`data: ${JSON.stringify({ type: "response.content_part.added", part: { type: "output_text", text: "" } })}`,
|
||||
`data: ${JSON.stringify({ type: "response.output_text.delta", delta: "Recovered" })}`,
|
||||
`data: ${JSON.stringify({
|
||||
type: "response.output_item.done",
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_sse",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "Recovered" }],
|
||||
},
|
||||
})}`,
|
||||
`data: ${JSON.stringify({ type: "response.completed", response: { id: "resp_sse", status: "completed" } })}`,
|
||||
].join("\n\n")}\n\n`;
|
||||
const fetchMock = vi.fn(
|
||||
async () => new Response(sse, { status: 200, headers: { "content-type": "text/event-stream" } }),
|
||||
);
|
||||
|
||||
let constructorCount = 0;
|
||||
class AlwaysLimitedWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
constructorCount += 1;
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(): void {
|
||||
this.sendJson({
|
||||
type: "error",
|
||||
code: "websocket_connection_limit_reached",
|
||||
message: "connection limit reached",
|
||||
});
|
||||
}
|
||||
}
|
||||
global.WebSocket = AlwaysLimitedWebSocket as unknown as typeof WebSocket;
|
||||
|
||||
const model = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-connection-limit-bounded-session",
|
||||
providerSessionState: new Map<string, ProviderSessionState>(),
|
||||
}).result();
|
||||
|
||||
// 1 initial connection + PI_CODEX_WEBSOCKET_RETRY_BUDGET bounded reconnects,
|
||||
// then a single SSE replay — never an unbounded reconnect loop.
|
||||
expect(constructorCount).toBe(3);
|
||||
expect(fetchMock).toHaveBeenCalledTimes(1);
|
||||
expect(result.stopReason).toBe("stop");
|
||||
expect(result.errorMessage).toBeUndefined();
|
||||
expect(result.content).toEqual([expect.objectContaining({ type: "text", text: "Recovered" })]);
|
||||
});
|
||||
|
||||
it("surfaces a whitespace flood arriving after a delivered tool call instead of replaying", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-codex-stream-");
|
||||
setAgentDir(tempDir.path());
|
||||
const token = createCodexTestToken();
|
||||
const fetchMock = vi.fn(async () => {
|
||||
throw new Error("SSE replay must not run after a toolcall_end was delivered");
|
||||
});
|
||||
|
||||
let sendCount = 0;
|
||||
class PostDoneWhitespaceWebSocket extends MockWebSocket {
|
||||
constructor(url: string, options?: { headers?: WsHeaders }) {
|
||||
super(url, options);
|
||||
this.scheduleOpen();
|
||||
}
|
||||
|
||||
send(): void {
|
||||
sendCount += 1;
|
||||
this.sendJson({
|
||||
type: "response.output_item.added",
|
||||
item: { type: "function_call", id: "fc_flood", call_id: "call_flood", name: "todo", arguments: "" },
|
||||
});
|
||||
this.sendJson({
|
||||
type: "response.output_item.done",
|
||||
item: { type: "function_call", id: "fc_flood", call_id: "call_flood", name: "todo", arguments: "{}" },
|
||||
});
|
||||
// Degenerate frames keep arriving after the item closed. They count as
|
||||
// progress events, so without the breaker observing them the idle
|
||||
// watchdog never fires and the turn hangs forever.
|
||||
for (let sequence = 1; sequence <= 300; sequence += 1) {
|
||||
this.sendJson({
|
||||
type: "response.function_call_arguments.delta",
|
||||
delta: " ".repeat(64),
|
||||
item_id: "fc_flood",
|
||||
output_index: 0,
|
||||
sequence_number: sequence,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
global.WebSocket = PostDoneWhitespaceWebSocket as unknown as typeof WebSocket;
|
||||
|
||||
const model = createCodexTestModel("https://chatgpt.com/backend-api");
|
||||
const result = await streamOpenAICodexResponses(model, createCodexTestContext(), {
|
||||
fetch: fetchMock as FetchImpl,
|
||||
apiKey: token,
|
||||
sessionId: "ws-post-done-whitespace-session",
|
||||
providerSessionState: new Map<string, ProviderSessionState>(),
|
||||
}).result();
|
||||
|
||||
// A toolcall_end already reached the consumer: replay is refused and the
|
||||
// breaker error surfaces on the first attempt.
|
||||
expect(sendCount).toBe(1);
|
||||
expect(result.stopReason).toBe("error");
|
||||
expect(result.errorMessage).toContain("whitespace-only tool-call argument delta");
|
||||
// The completed tool call is preserved on the error message.
|
||||
expect(result.content).toEqual([expect.objectContaining({ type: "toolCall", name: "todo" })]);
|
||||
expect(fetchMock).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
|
||||
@@ -125,6 +125,24 @@ describe("openai-codex orphan tool-call repair", () => {
|
||||
expect(output).toBeDefined();
|
||||
expect(output?.output as string).toMatch(/interrupted/i);
|
||||
});
|
||||
|
||||
it("folds an orphan custom_tool_call_output into an assistant message", async () => {
|
||||
const body: RequestBody = {
|
||||
model: "gpt-5.1-codex",
|
||||
input: [
|
||||
{ type: "message", role: "user", content: [{ type: "input_text", text: "hi" }] },
|
||||
{ type: "custom_tool_call_output", call_id: "call_custom_orphan", name: "apply_patch", output: "Done!" },
|
||||
],
|
||||
};
|
||||
|
||||
const transformed = await transformRequestBody(body, createCodexModel(body.model), {});
|
||||
const input = transformed.input || [];
|
||||
|
||||
expect(input.some(item => item.type === "custom_tool_call_output")).toBe(false);
|
||||
const note = input.find(item => item.type === "message" && item.role === "assistant");
|
||||
expect(note?.content).toMatch(/call_custom_orphan/);
|
||||
expect(note?.content).toMatch(/Done!/);
|
||||
});
|
||||
});
|
||||
|
||||
describe("openai-codex reasoning effort validation", () => {
|
||||
|
||||
@@ -70,6 +70,7 @@ function baseContext(): Context {
|
||||
async function captureOpenAICompletionsPayload(
|
||||
model: Model<"openai-completions">,
|
||||
context: Context = baseContext(),
|
||||
options?: { reasoning?: "minimal" | "low" | "medium" | "high" | "xhigh" },
|
||||
): Promise<unknown> {
|
||||
const { promise, resolve } = Promise.withResolvers<unknown>();
|
||||
const fetchMock = createMockFetch(["[DONE]"]);
|
||||
@@ -78,6 +79,7 @@ async function captureOpenAICompletionsPayload(
|
||||
fetch: fetchMock,
|
||||
signal: createAbortedSignal(),
|
||||
onPayload: payload => resolve(payload),
|
||||
...options,
|
||||
});
|
||||
return promise;
|
||||
}
|
||||
@@ -170,6 +172,84 @@ describe("openai-completions compatibility", () => {
|
||||
expect(assistant.content).toBe("hello world");
|
||||
});
|
||||
|
||||
it("prepends thinking text to string assistant content when requiresThinkingAsText is set", () => {
|
||||
const model: Model<"openai-completions"> = {
|
||||
...getBundledModel("openai", "gpt-4o-mini"),
|
||||
api: "openai-completions",
|
||||
};
|
||||
const assistantMessage: AssistantMessage = {
|
||||
role: "assistant",
|
||||
content: [
|
||||
{ type: "thinking", thinking: "chain of thought" },
|
||||
{ type: "text", text: "final answer" },
|
||||
],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(
|
||||
model,
|
||||
{ messages: [assistantMessage] },
|
||||
{
|
||||
...detectCompat(model),
|
||||
requiresThinkingAsText: true,
|
||||
},
|
||||
);
|
||||
const assistant = messages.find(message => message.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
if (assistant?.role !== "assistant") throw new Error("assistant message missing");
|
||||
// Regression: thinking+text replay used to call `.unshift` on the string
|
||||
// content set above (TypeError). Both blocks must survive as one string.
|
||||
expect(typeof assistant.content).toBe("string");
|
||||
expect(assistant.content).toBe("chain of thought\n\nfinal answer");
|
||||
});
|
||||
|
||||
it("emits thinking-only assistant content as a plain string when requiresThinkingAsText is set", () => {
|
||||
const model: Model<"openai-completions"> = {
|
||||
...getBundledModel("openai", "gpt-4o-mini"),
|
||||
api: "openai-completions",
|
||||
};
|
||||
const assistantMessage: AssistantMessage = {
|
||||
role: "assistant",
|
||||
content: [{ type: "thinking", thinking: "only thoughts" }],
|
||||
api: model.api,
|
||||
provider: model.provider,
|
||||
model: model.id,
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
const messages = convertMessages(
|
||||
model,
|
||||
{ messages: [assistantMessage] },
|
||||
{
|
||||
...detectCompat(model),
|
||||
requiresThinkingAsText: true,
|
||||
},
|
||||
);
|
||||
const assistant = messages.find(message => message.role === "assistant");
|
||||
expect(assistant).toBeDefined();
|
||||
if (assistant?.role !== "assistant") throw new Error("assistant message missing");
|
||||
expect(assistant.content).toBe("only thoughts");
|
||||
});
|
||||
|
||||
it("preserves multiple system prompts as leading system messages for chat completions", () => {
|
||||
const model: Model<"openai-completions"> = {
|
||||
...getBundledModel("openai", "gpt-4o-mini"),
|
||||
@@ -647,6 +727,23 @@ describe("kimi model detection via detectCompat", () => {
|
||||
expect(detectCompat(openRouterKimi).thinkingFormat).toBe("openrouter");
|
||||
});
|
||||
|
||||
it("maps OpenRouter Anthropic adaptive reasoning efforts to the Anthropic scale", async () => {
|
||||
const model: Model<"openai-completions"> = {
|
||||
...getBundledModel("openai", "gpt-4o-mini"),
|
||||
api: "openai-completions",
|
||||
provider: "openrouter",
|
||||
baseUrl: "https://openrouter.ai/api/v1",
|
||||
id: "anthropic/claude-fable-5",
|
||||
reasoning: true,
|
||||
};
|
||||
|
||||
const highPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "high" });
|
||||
const xhighPayload = await captureOpenAICompletionsPayload(model, baseContext(), { reasoning: "xhigh" });
|
||||
|
||||
expect(getNestedObject(highPayload, "reasoning")).toEqual({ effort: "xhigh" });
|
||||
expect(getNestedObject(xhighPayload, "reasoning")).toEqual({ effort: "max" });
|
||||
});
|
||||
|
||||
// Regression for #1071: OpenCode-Go/Zen handle reasoning content server-side
|
||||
// and reject client-supplied `reasoning_content` ("Extra inputs are not
|
||||
// permitted"). Kimi on opencode-* MUST NOT have reasoning_content injected,
|
||||
|
||||
@@ -0,0 +1,321 @@
|
||||
// Terminal-event contracts for `processResponsesStream`:
|
||||
//
|
||||
// 1. `response.incomplete` is a terminal frame (max_output_tokens / content
|
||||
// filter truncation). It must populate usage and map to stopReason
|
||||
// "length" — previously it was ignored entirely, so truncated responses
|
||||
// reported stopReason "stop" with zero usage and no cost.
|
||||
// 2. `response.output_item.done` for a custom_tool_call must persist the final
|
||||
// input on the stored content block and drop the transient `partialJson`
|
||||
// accumulation buffer, mirroring the function_call branch.
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { processResponsesStream } from "@oh-my-pi/pi-ai/providers/openai-responses-shared";
|
||||
import type { AssistantMessage, Model } from "@oh-my-pi/pi-ai/types";
|
||||
import type { ResponseStreamEvent } from "openai/resources/responses/responses";
|
||||
|
||||
function makeModel(): Model<"openai-responses"> {
|
||||
return {
|
||||
api: "openai-responses",
|
||||
name: "GPT Test",
|
||||
id: "gpt-test",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
contextWindow: 8192,
|
||||
maxTokens: 2048,
|
||||
input: ["text"],
|
||||
reasoning: false,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
};
|
||||
}
|
||||
|
||||
function makeOutput(): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [],
|
||||
timestamp: Date.now(),
|
||||
provider: "openai",
|
||||
model: "gpt-test",
|
||||
api: "openai-responses",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "stop",
|
||||
};
|
||||
}
|
||||
|
||||
async function* makeStream(events: unknown[]): AsyncIterable<ResponseStreamEvent> {
|
||||
for (const e of events) yield e as ResponseStreamEvent;
|
||||
}
|
||||
|
||||
type EmittedEvent = { type?: string } & Record<string, unknown>;
|
||||
|
||||
describe("processResponsesStream: terminal events", () => {
|
||||
test("maps response.incomplete to a length stop with usage populated", async () => {
|
||||
const output = makeOutput();
|
||||
const emitted: EmittedEvent[] = [];
|
||||
const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never;
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "message", id: "msg_1", role: "assistant", status: "in_progress", content: [] },
|
||||
},
|
||||
{
|
||||
type: "response.content_part.added",
|
||||
output_index: 0,
|
||||
item_id: "msg_1",
|
||||
part: { type: "output_text", text: "", annotations: [] },
|
||||
},
|
||||
{
|
||||
type: "response.output_text.delta",
|
||||
output_index: 0,
|
||||
item_id: "msg_1",
|
||||
delta: "Hello, trunc",
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_1",
|
||||
role: "assistant",
|
||||
status: "incomplete",
|
||||
content: [{ type: "output_text", text: "Hello, trunc", annotations: [] }],
|
||||
},
|
||||
},
|
||||
{
|
||||
type: "response.incomplete",
|
||||
sequence_number: 5,
|
||||
response: {
|
||||
id: "resp_incomplete",
|
||||
status: "incomplete",
|
||||
incomplete_details: { reason: "max_output_tokens" },
|
||||
usage: {
|
||||
input_tokens: 7,
|
||||
output_tokens: 9,
|
||||
total_tokens: 16,
|
||||
input_tokens_details: { cached_tokens: 2 },
|
||||
},
|
||||
},
|
||||
},
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.stopReason).toBe("length");
|
||||
expect(output.responseId).toBe("resp_incomplete");
|
||||
expect(output.usage.input).toBe(5);
|
||||
expect(output.usage.cacheRead).toBe(2);
|
||||
expect(output.usage.output).toBe(9);
|
||||
expect(output.usage.totalTokens).toBe(16);
|
||||
expect(output.content).toEqual([expect.objectContaining({ type: "text", text: "Hello, trunc" })]);
|
||||
});
|
||||
|
||||
test("persists final custom tool input on the block and drops the accumulation buffer", async () => {
|
||||
const output = makeOutput();
|
||||
const emitted: EmittedEvent[] = [];
|
||||
const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never;
|
||||
|
||||
const patch = "*** Begin Patch";
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.output_item.added",
|
||||
output_index: 0,
|
||||
item: { type: "custom_tool_call", id: "ctc_1", call_id: "call_c", name: "apply_patch", input: "" },
|
||||
},
|
||||
{
|
||||
type: "response.custom_tool_call_input.delta",
|
||||
output_index: 0,
|
||||
item_id: "ctc_1",
|
||||
delta: patch,
|
||||
},
|
||||
{
|
||||
type: "response.custom_tool_call_input.done",
|
||||
output_index: 0,
|
||||
item_id: "ctc_1",
|
||||
input: patch,
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "custom_tool_call", id: "ctc_1", call_id: "call_c", name: "apply_patch", input: patch },
|
||||
},
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.content).toHaveLength(1);
|
||||
const block = output.content[0];
|
||||
if (block?.type !== "toolCall") throw new Error("expected a toolCall block");
|
||||
expect(block.customWireName).toBe("apply_patch");
|
||||
expect(block.arguments).toEqual({ input: patch });
|
||||
expect("partialJson" in block).toBe(false);
|
||||
|
||||
const end = emitted.find(e => e.type === "toolcall_end") as
|
||||
| { toolCall: { arguments: Record<string, unknown> } }
|
||||
| undefined;
|
||||
expect(end?.toolCall.arguments).toEqual({ input: patch });
|
||||
});
|
||||
});
|
||||
|
||||
describe("processResponsesStream: lost output_item.added recovery", () => {
|
||||
test("synthesizes the tool-call block when output_item.added was lost", async () => {
|
||||
const output = makeOutput();
|
||||
const emitted: EmittedEvent[] = [];
|
||||
const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never;
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "function_call",
|
||||
id: "fc_lost",
|
||||
call_id: "call_lost",
|
||||
name: "read",
|
||||
arguments: '{"path":"a.txt"}',
|
||||
},
|
||||
},
|
||||
{ type: "response.completed", response: { id: "resp_lost", status: "completed" } },
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.content).toHaveLength(1);
|
||||
const block = output.content[0];
|
||||
if (block?.type !== "toolCall") throw new Error("expected a toolCall block");
|
||||
expect(block.arguments).toEqual({ path: "a.txt" });
|
||||
// The toolUse override fires because the call now exists in content; the
|
||||
// agent loop executes tools from message.content.
|
||||
expect(output.stopReason).toBe("toolUse");
|
||||
const end = emitted.find(e => e.type === "toolcall_end") as { contentIndex: number } | undefined;
|
||||
expect(end?.contentIndex).toBe(0);
|
||||
});
|
||||
|
||||
test("synthesizes the text block when message output_item.added was lost", async () => {
|
||||
const output = makeOutput();
|
||||
const emitted: EmittedEvent[] = [];
|
||||
const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never;
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: {
|
||||
type: "message",
|
||||
id: "msg_lost",
|
||||
role: "assistant",
|
||||
status: "completed",
|
||||
content: [{ type: "output_text", text: "Recovered text", annotations: [] }],
|
||||
},
|
||||
},
|
||||
{ type: "response.completed", response: { id: "resp_lost_msg", status: "completed" } },
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.content).toEqual([expect.objectContaining({ type: "text", text: "Recovered text" })]);
|
||||
const end = emitted.find(e => e.type === "text_end") as { content: string } | undefined;
|
||||
expect(end?.content).toBe("Recovered text");
|
||||
});
|
||||
|
||||
test("routes reasoning finalization by output_index when item ids are absent", async () => {
|
||||
const output = makeOutput();
|
||||
const stream = { push: () => {}, end: () => {} } as never;
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{ type: "response.output_item.added", output_index: 0, item: { type: "reasoning", summary: [] } },
|
||||
{ type: "response.output_item.added", output_index: 1, item: { type: "reasoning", summary: [] } },
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 0,
|
||||
item: { type: "reasoning", summary: [{ type: "summary_text", text: "first" }] },
|
||||
},
|
||||
{
|
||||
type: "response.output_item.done",
|
||||
output_index: 1,
|
||||
item: { type: "reasoning", summary: [{ type: "summary_text", text: "second" }] },
|
||||
},
|
||||
{ type: "response.completed", response: { id: "resp_reasoning", status: "completed" } },
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.content).toHaveLength(2);
|
||||
const [first, second] = output.content;
|
||||
if (first?.type !== "thinking" || second?.type !== "thinking") throw new Error("expected thinking blocks");
|
||||
expect(first.thinking).toBe("first");
|
||||
expect(second.thinking).toBe("second");
|
||||
expect(first.thinkingSignature).toBeDefined();
|
||||
expect(second.thinkingSignature).toBeDefined();
|
||||
});
|
||||
|
||||
test("treats content_filter incomplete responses as errors, not length", async () => {
|
||||
const output = makeOutput();
|
||||
const stream = { push: () => {}, end: () => {} } as never;
|
||||
|
||||
await expect(
|
||||
processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.incomplete",
|
||||
response: {
|
||||
id: "resp_cf",
|
||||
status: "incomplete",
|
||||
incomplete_details: { reason: "content_filter" },
|
||||
},
|
||||
},
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
),
|
||||
).rejects.toThrow("incomplete: content_filter");
|
||||
});
|
||||
|
||||
test("preserves premiumRequests across usage population", async () => {
|
||||
const output = makeOutput();
|
||||
output.usage.premiumRequests = 3;
|
||||
const stream = { push: () => {}, end: () => {} } as never;
|
||||
|
||||
await processResponsesStream(
|
||||
makeStream([
|
||||
{
|
||||
type: "response.completed",
|
||||
response: {
|
||||
id: "resp_premium",
|
||||
status: "completed",
|
||||
usage: { input_tokens: 4, output_tokens: 2, total_tokens: 6 },
|
||||
},
|
||||
},
|
||||
]),
|
||||
output,
|
||||
stream,
|
||||
makeModel(),
|
||||
);
|
||||
|
||||
expect(output.usage.premiumRequests).toBe(3);
|
||||
expect(output.usage.input).toBe(4);
|
||||
expect(output.usage.output).toBe(2);
|
||||
});
|
||||
});
|
||||
@@ -1012,3 +1012,37 @@ describe("circular schema safety", () => {
|
||||
expect(() => sanitizeSchemaForStrictMode(circular)).not.toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// DAG-shared subtrees and frozen inputs (normalizeSchemaNode enter/exit)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe("DAG-shared subtree normalization", () => {
|
||||
it("normalizes a subschema object reused across two properties instead of blanking the second occurrence", () => {
|
||||
const shared = { type: "string", description: "shared leaf" };
|
||||
const schema = {
|
||||
type: "object",
|
||||
properties: { a: shared, b: shared },
|
||||
};
|
||||
|
||||
const result = normalizeSchemaForGoogle(schema) as {
|
||||
properties: { a: Record<string, unknown>; b: Record<string, unknown> };
|
||||
};
|
||||
expect(result.properties.a).toEqual({ type: "string", description: "shared leaf" });
|
||||
expect(result.properties.b).toEqual({ type: "string", description: "shared leaf" });
|
||||
});
|
||||
|
||||
it("does not throw on a frozen input schema", () => {
|
||||
const shared = Object.freeze({ type: "number" });
|
||||
const schema = Object.freeze({
|
||||
type: "object",
|
||||
properties: Object.freeze({ x: shared, y: shared }),
|
||||
});
|
||||
|
||||
const result = normalizeSchemaForGoogle(schema) as {
|
||||
properties: { x: Record<string, unknown>; y: Record<string, unknown> };
|
||||
};
|
||||
expect(result.properties.x).toEqual({ type: "number" });
|
||||
expect(result.properties.y).toEqual({ type: "number" });
|
||||
});
|
||||
});
|
||||
|
||||
@@ -218,6 +218,25 @@ describe("StreamMarkupHealing DSML envelope pattern", () => {
|
||||
expect(calls[0].name).toBe("bash");
|
||||
expect(JSON.parse(calls[0].arguments)).toEqual({ cmd: "ls -la" });
|
||||
});
|
||||
|
||||
it("passes a bare '<' in idle prose through without holding it back", () => {
|
||||
const healing = new StreamMarkupHealing({ pattern: "dsml" });
|
||||
// No '>' anywhere in the tail — the old any-'<' hold-back froze display here.
|
||||
expect(healing.feed("if a < b:\n return a")).toBe("if a < b:\n return a");
|
||||
});
|
||||
|
||||
it("still holds back a tail that is a partial DSML section-open tag", () => {
|
||||
const healing = new StreamMarkupHealing({ pattern: "dsml" });
|
||||
expect(healing.feed("run ")).toBe("run ");
|
||||
expect(healing.feed("<|DSML|tool")).toBe("");
|
||||
expect(healing.feed("_calls>")).toBe("");
|
||||
expect(
|
||||
healing.feed(
|
||||
'<|DSML|invoke name="bash"><|DSML|parameter name="cmd">ls</|DSML|parameter></|DSML|invoke></|DSML|tool_calls>',
|
||||
),
|
||||
).toBe("");
|
||||
expect(healing.drainCompleted()).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe("StreamMarkupHealing thinking pattern", () => {
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
// Duplicate Responses-family tool-call ids are composites (`callId|itemId`).
|
||||
// The dedup suffix must reach the wire call_id — the FIRST segment, which
|
||||
// normalizeResponsesToolCallId extracts at encode time — otherwise the request
|
||||
// carries two function_call/function_call_output pairs sharing one call_id.
|
||||
// Regression for the suffix previously landing on the composite as a whole
|
||||
// (`call_x|fc_y` → `call_x|fc_y_dup1`, wire call_id `call_x` for both copies).
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { transformMessages } from "@oh-my-pi/pi-ai/providers/transform-messages";
|
||||
import type { AssistantMessage, Message, Model, ToolResultMessage } from "@oh-my-pi/pi-ai/types";
|
||||
import { normalizeResponsesToolCallId } from "@oh-my-pi/pi-ai/utils";
|
||||
|
||||
function makeModel(): Model<"openai-responses"> {
|
||||
return {
|
||||
api: "openai-responses",
|
||||
name: "GPT Test",
|
||||
id: "gpt-test",
|
||||
provider: "openai",
|
||||
baseUrl: "https://api.openai.com/v1",
|
||||
contextWindow: 8192,
|
||||
maxTokens: 2048,
|
||||
input: ["text"],
|
||||
reasoning: false,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
};
|
||||
}
|
||||
|
||||
function assistantWithCall(id: string): AssistantMessage {
|
||||
return {
|
||||
role: "assistant",
|
||||
content: [{ type: "toolCall", id, name: "read", arguments: { path: "a" } }],
|
||||
api: "openai-responses",
|
||||
provider: "openai",
|
||||
model: "gpt-test",
|
||||
usage: {
|
||||
input: 0,
|
||||
output: 0,
|
||||
cacheRead: 0,
|
||||
cacheWrite: 0,
|
||||
totalTokens: 0,
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
|
||||
},
|
||||
stopReason: "toolUse",
|
||||
timestamp: Date.now(),
|
||||
};
|
||||
}
|
||||
|
||||
function toolResult(id: string, text: string): ToolResultMessage {
|
||||
return {
|
||||
role: "toolResult",
|
||||
toolCallId: id,
|
||||
toolName: "read",
|
||||
content: [{ type: "text", text }],
|
||||
isError: false,
|
||||
timestamp: Date.now(),
|
||||
} as ToolResultMessage;
|
||||
}
|
||||
|
||||
describe("deduplicateToolCallIds with composite Responses ids", () => {
|
||||
it("suffixes the call_id segment so wire ids stay distinct", () => {
|
||||
const dupId = "call_x|fc_y";
|
||||
const messages: Message[] = [
|
||||
assistantWithCall(dupId),
|
||||
toolResult(dupId, "first"),
|
||||
assistantWithCall(dupId),
|
||||
toolResult(dupId, "second"),
|
||||
];
|
||||
|
||||
const transformed = transformMessages(messages, makeModel());
|
||||
|
||||
const callIds = transformed
|
||||
.filter((m): m is AssistantMessage => m.role === "assistant")
|
||||
.flatMap(m => m.content)
|
||||
.filter(b => b.type === "toolCall")
|
||||
.map(b => normalizeResponsesToolCallId((b as { id: string }).id).callId);
|
||||
expect(callIds).toHaveLength(2);
|
||||
expect(new Set(callIds).size).toBe(2);
|
||||
|
||||
// Each rewritten toolResult resolves to the same wire call_id as its call.
|
||||
const resultIds = transformed
|
||||
.filter((m): m is ToolResultMessage => m.role === "toolResult")
|
||||
.map(m => normalizeResponsesToolCallId(m.toolCallId).callId);
|
||||
expect(resultIds).toEqual(callIds);
|
||||
});
|
||||
});
|
||||
@@ -4,6 +4,98 @@
|
||||
### Added
|
||||
|
||||
- Added isolated profile support via `--profile <name>` / `OMP_PROFILE` and shell alias bootstrap via `--alias <command>`, including launch/ACP bootstrap handling, extension-flag-safe parsing, profile-scoped user config discovery, and symlinked extension-directory discovery.
|
||||
- New `omp usage` command: a detailed per-account breakdown of provider usage limits (bars, windows, reset times, plan metadata) covering every stored credential — accounts with no usage endpoint are listed as "no usage data" rows. Each provider section ends with per-window capacity stats ("need: 5h → 3 of 5 accounts"). Flags: `--provider` to filter, `--json` for the broker-shaped report payload, and `--redact` to mask account emails/ids down to a two-char anchor plus a minimal middle-out differentiator (`ca*9*`) for screenshot-safe sharing.
|
||||
- Startup hangs are now self-diagnosing (speculative fix for the "zero output, hangs even on `omp -h`" report class): a watchdog prints a stderr line every 10s naming the deepest in-flight startup phase (via `logger.openSpanPath()`) until a mode runner takes over, pausing around legitimate interactive waits (fork/move prompts, the `--resume` session picker); `PI_DEBUG_STARTUP` is restored as streaming synchronous `[startup]` phase markers covering command-module imports and the native addon load, which the post-startup `PI_TIMING` tree structurally cannot show for a hang; and waiting on piped-stdin EOF announces itself after 1s instead of blocking silently.
|
||||
|
||||
### Changed
|
||||
|
||||
- Cached custom model alias maps and built them lazily on first custom model reference lookup, avoiding unnecessary startup model-registry initialization
|
||||
- Cached resolved auth broker configuration and snapshot reads for the process lifetime so repeated startup paths reuse the same `OMP_AUTH_BROKER_*` resolution instead of re-running config/token discovery
|
||||
- Reused task-agent discovery results for repeated `TaskTool.create` calls in the same working directory to avoid repeated plugin scans during subagent startup
|
||||
- Tightened the system prompt and tool prompts: deduped restated warnings (bash "catch yourself" list, search/find shell-fallback recaps, read instruction/critical overlap, the AST metavariable primer duplicated across both ast tool descriptions), factored the repeated repo-default clause in the `gh` search ops, dropped a dead `rsed` reference and an internal `tool-timeouts.ts` pointer, and pruned internal mechanism the agent can't act on (screenshot temp-file/downscaling pipeline, browser spawn lifecycle, `gh` "replaces former op" history and run-watch grace period, output-minimizer heuristics, BM25 ranking name, `task.maxConcurrency` pointer)
|
||||
- Extended the prompt-efficiency pass to the full prompt surface (subagent/plan-mode/notice/title/commit system prompts, agent definitions, goals, memories, review and autoresearch prompts): RFC-keyed prescriptive prose, fixed garbled grammar and a stale `<PLAN_TITLE>` placeholder in the plan-approval reminder, deduped intra-file restatements, and corrected the `todo` op table's claim that `rm` requires a `task`/`phase` (bare `rm` clears the whole list)
|
||||
- Replace tool prompt no longer recommends `sed -i`/`cat`-heredoc commands that the bash interceptor blocks; its bash-alternatives table now only lists non-intercepted commands
|
||||
- Capped concurrent IRC cards in the transcript's live region at 4: cards landing below a still-running tool cannot commit to native scrollback, so an unbounded burst pushed the live block's uncommitted rows above the window top (content read as cut off until the cards expired). The oldest live-region card now retires as soon as a new one would exceed the cap.
|
||||
- Interactive PTY mode (`pty: true`) no longer injects the non-interactive environment (`TERM=dumb`, `GIT_EDITOR=true`, `PAGER=cat`, `NO_COLOR=1`) that defeated its purpose — the PTY child now gets a real `TERM=xterm-256color`; and when a PTY is requested but unavailable (headless/RPC), the result now carries an explicit downgrade notice instead of silently running through a dumb pipe.
|
||||
- Raw sqlite `?q=` queries are now capped at 1000 rows with an "add a LIMIT clause" notice — `statement.all()` on a multi-million-row table previously materialized every row, blocking the process for minutes.
|
||||
- Plain-file range reads no longer scan to EOF on files over 4MB just to count total lines (the count is reported as approximate), and multi-range reads slice from a single pass instead of re-streaming the file once per range.
|
||||
- `gh run_watch` now polls adaptively (3s for the first minute, then 15s), survives rate-limit errors with backoff instead of dying and discarding accumulated context, reuses job data for completed runs, and gives up with a clear message after ~90s when a commit has no workflow runs at all (previously an infinite 3-second poll loop).
|
||||
- The legacy patch-mode fuzzy matcher pre-normalizes file and pattern lines once per seek with a Levenshtein lower-bound bail, replacing the per-position re-normalization that made a single mismatched hunk against a 10k-line file cost multi-second synchronous stalls; the streaming hashline preview also caches file text and tree-sitter block resolution across ticks instead of re-reading and re-parsing every target file per streamed chunk.
|
||||
- The DAP client reader now uses chunk-list buffering and the output buffer is a chunk deque with a running byte count — debugging a chatty program previously cost O(n²) `Buffer.concat` per chunk plus whole-buffer byte-scans per 1KB trim, freezing the session.
|
||||
- GitHub caching: the per-lookup auth key is memoized against `hosts.yml` mtime (was a blocking `readFileSync` on every `issue://`/`pr://` read including cache hits), background refreshes are deduped by row identity, and PR diffs are stored once per row instead of twice (unified + rendered copies).
|
||||
- Task progress snapshots shallow-copy per-agent progress instead of `structuredClone`-ing nested tool payloads (up to 500KB) on every progress event; streaming assistant-message reveal caches per-block grapheme counts and skips the markdown render LRU for in-flight partials, eliminating 2-3 full Intl.Segmenter walks per 33ms tick and tens of MB of retained stale partial snapshots on long replies.
|
||||
- Python eval cells: the availability probe is cached per cwd (was two interpreter spawns per cell even with a hot kernel), and stdout frames coalesce per write instead of one locked+flushed JSON frame each.
|
||||
- Multi-entry edits now stop at the first failing entry and report exactly which entries were applied and which were not — continuing after a failure applied later entries authored against line numbers that assumed the failed entry succeeded, and a retry of the whole batch then double-applied the survivors.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed the bundled `explore` agent's `thinking-level: med` frontmatter — not a valid effort (`minimal`/`low`/`medium`/`high`/`xhigh`), so it silently parsed to undefined and the agent ran without its intended thinking level
|
||||
- Discovery context-file reads (`~/.claude`, `~/.cursor`, project trees, `@`-imports) now stat-gate to regular files before reading: a FIFO/socket/char device dropped where a context file is expected previously blocked startup forever on a read that can never see EOF.
|
||||
- Kept IRC cards from being removed after their TTL once everything above them finalized: their rows may already be committed to native scrollback, and removing them was an interior deletion of the committed prefix that the engine could only repair by recommitting everything below the gap (duplicated blocks). Such cards now stay in the transcript as durable history.
|
||||
- Fixed the recommit storm that sprayed stale snapshots of a running task's progress tree into native scrollback. The stable-prefix ratchet promoted any row quiet for one 30-frame window, so slowly ticking rows (per-agent tool/cost counters updating every few seconds) were repeatedly promoted, committed, rewritten, and recommitted by the engine audit for the whole run. The ratchet now floors itself permanently at the first row that mutates after being promoted — settled heads (a task's prompt/context) still reach scrollback, genuine tickers never re-promote.
|
||||
- **Fixed the artifact spill dropping the first ~20KB of output**: head-retained bytes were never written to the artifact file, so for every bash/eval/ssh command exceeding the 50KB spill threshold, the `artifact://` advertised as the "full capture" was permanently missing its head — the agent re-reading it got truncated data presented as lossless.
|
||||
- Fixed streaming-output chunk throttling dropping chunks instead of coalescing them: streaming previews and the auto-background "output so far" text the model reasons over contained output with arbitrary middles silently spliced out.
|
||||
- **Fixed `vault://` writes bypassing both the approval ladder and plan mode**: internal-URL writes were uniformly rated tier `read` (auto-allowed even in always-ask) and the internal-router branch returned before the plan-mode guard, so the model could silently overwrite real Obsidian notes; writes through schemes with a mutating handler are now tier `write` and plan-mode-enforced.
|
||||
- Fixed writes into `.tar.gz`/`.tgz` archives silently stripping gzip compression (the rewritten archive was a bare tar under the `.gz` name — masked on re-read because Bun auto-detects, broken for `tar xzf`/CI consumers), and made archive rewrites atomic via temp-file + rename so a crash mid-write can no longer destroy every other member; symlinked archive paths resolve to their target before the swap so the rename writes through instead of replacing the link with a regular file.
|
||||
- Fixed merge-conflict detection being completely inert on CRLF files: the scanner split on `\n` and compared `=== "======="`, so `=======\r` never matched and the agent edited around live conflict markers without warning; CRLF files now detect, splice, and round-trip their line endings correctly.
|
||||
- Fixed cross-line search (`\n` in the pattern) silently returning zero matches: the native searcher was never switched to multi-line mode (only the regex flag was set), so the advertised feature matched nothing on real files while reporting a confident "No matches found".
|
||||
- Fixed search results lying about completeness: one hot file could consume the entire 2000-match global budget in path order making later files unreachable by any `skip` (now capped per file with the footer hedging `of N+` when truncated), paginating past the last page returned "No matches found" instead of "No more results", directory scans now report how many >4MB files were skipped instead of silently excluding them, adjacent matches in virtual resources no longer emit duplicated backwards-numbered context lines, and patterns are no longer `trim()`ed (only all-whitespace is rejected — leading/trailing whitespace is meaningful regex).
|
||||
- Fixed the search tool's native grep being uncancellable: neither the abort signal nor any timeout was threaded through, so Esc on a huge-tree search left the native walk burning CPU to completion; both now propagate (30s default timeout).
|
||||
- Fixed archive and sqlite reads that could OOM or hang the process: tar/tgz archives are stat-gated at 256MB before being loaded, zip members reject attacker-declared uncompressed sizes over 64MB before allocation, and binary plain files now return a NUL-sniff notice instead of filling the line budget with mojibake.
|
||||
- Fixed malformed internal-URL selectors (`artifact://3:-100`) silently dumping the whole resource instead of erroring, selectors directly on an archive root (`a.zip:500`, `a.zip:raw`) being misparsed as member names, archive members minting editable hashline tags keyed to the archive path (they are immutable resources), URL selector tokens being case-sensitive (`:RAW` 404ed), `artifact://N` resolving into another session's artifacts in multi-session hosts, and not-found paths with archive/sqlite extensions stacking multiple 5s workspace-wide suffix globs (now shared per read, with glob metachars escaped so `foo[1].ts` can match itself).
|
||||
- Fixed leading `cd X &&` extraction breaking shell-expanded paths — `cd "$(git rev-parse --show-toplevel)" && make` failed with "Working directory does not exist" because the captured path was resolved literally; extraction now defers to the shell when the path contains `$`, backticks, or `(`.
|
||||
- Fixed the echo/printf write-redirect interceptor rule blocking legitimate commands containing `>` inside quotes (`echo "a -> b"`, `printf 'use 2>&1'`); the rule is now quote-aware, and also catches `>|` clobber redirects and `$VAR` targets it previously missed.
|
||||
- Fixed every completed auto-backgrounded bash invocation leaking its persistent native `Shell` in the process-global session map, and the running-job cap failing all bash commands outright — at capacity, commands now degrade to direct foreground execution (explicit `async: true` still errors).
|
||||
- Fixed a duplicate-delivery race where a bash job completing just inside the auto-background threshold could be returned as the tool result and re-injected as a completion notification, and fixed auto-background silently preempting the ACP client-terminal route when an editor advertises terminal capability.
|
||||
- Fixed timed-out/cancelled PTY and client-bridge commands surfacing raw output with no annotation (the model couldn't distinguish timeout from failure and retried identically); the timeout/abort notice is now always appended.
|
||||
- Fixed `ask` reporting timeout auto-selection as "User selected: X" — fabricated consent for consequential questions; the result now says "(auto-selected after timeout)" with a `timedOut` detail flag, the transcript card marks the auto-selection distinctly, and a deliberate Esc seconds past the deadline is treated as a cancel instead of being reclassified as a timeout.
|
||||
- Fixed `todo` accepting duplicate task content/phase names in `init` (duplicates were permanently unaddressable — every targeting op hit the first match while auto-promotion kept resurrecting the twin) and persisting half-applied batches on error; failed batches no longer mutate state.
|
||||
- Fixed the auto-generated-file guard caching markers by path alone with no invalidation — a file regenerated after first check stayed editable (and vice versa); entries are now validated against mtime+size.
|
||||
- Fixed editor-bridged (ACP) writes skipping the post-write bookkeeping the direct path performs (`bumpFileMutationVersion`, shebang chmod), so mutation-version consumers saw stale state depending on whether an editor was attached.
|
||||
- Fixed `conflict://*` resolution failing spuriously when an out-of-band edit shifted a conflict block (stale duplicate registrations are now tolerated as already-resolved — but a DISTINCT conflict block that is merely byte-identical and still present in the file stays addressable), and partial conflict-resolution failures now set `isError` instead of burying failed files mid-text in a success result.
|
||||
- Fixed patch-mode prefix/substring matches silently truncating line content the model never saw: every non-exact match strategy now emits a warning with strategy + similarity, and prefix/substring matches are rejected unless the discarded fragment survives in the replacement lines.
|
||||
- Fixed ast-edit and file-mention snapshots being recorded under non-canonical paths (invisible to stale-tag recovery under symlinked cwds), and the ast-edit apply step leaving every just-issued preview tag stale — post-apply snapshots are re-recorded and fresh tags surfaced in the result.
|
||||
- Fixed notebook cells containing literal `# %% [markdown]` marker text being silently split into extra cells on any edit; marker-shaped source lines are now escaped on render and restored on parse.
|
||||
- Fixed the LSP client being published before `initialize` completed (concurrent callers hit "server not initialized" flakes on first use), reader-loop death leaving a permanent zombie client where every request times out at 30s forever (bad messages are now isolated per-message and a dead reader tears the client down for respawn), framing stalls on header blocks without `Content-Length` (now resynced past the junk in both LSP and DAP), `lsp status` hardcoding `ready` for every client including wedged ones, and shutdown skipping clients still mid-initialize (their server processes outlived exit).
|
||||
- Fixed numeric LSP code-action selectors being shadowed by substring title matches — `query: "2"` could apply a *different* quickfix whose title contained "2"; numeric queries now select strictly by index.
|
||||
- Fixed `file://` URIs built without percent-encoding: a `%` in a path threw `URIError` on round-trip and a `#` truncated the server-side path, desynchronizing diagnostics and workspace edits; URIs from lax servers carrying a raw `#`/`?` now route to the lenient parser instead of parsing "successfully" as fragment/query and misrouting edits.
|
||||
- Fixed multiple LSP inserts at the same position applying in reverse of spec order (transposed import/reference insertions), and `applyWorkspaceEdit` now overlap-validates every file before writing any, so a conflicting rename no longer leaves the workspace half-renamed.
|
||||
- Fixed `lsp reload` hanging for the whole tool timeout (`didChangeConfiguration` was sent as a request; it is a notification), biome failures being silently reported as "no diagnostics", a hung language server adding up to 30s to every edit (writethrough init is now deadline-bounded at 5s with deterministic spawn failures negative-cached for 3 minutes), and DAP `pause()` burning its full timeout when the stopped event raced the subscription; concurrent DAP breakpoint mutations are also serialized per session (last-writer no longer silently drops the other's breakpoints), queued mutations honor the caller's abort at dequeue, and the DAP output buffer retains a full 128KB tail instead of dropping whole chunks below the cap.
|
||||
- **Fixed concurrent isolated background tasks interleaving `git stash push/pop` + cherry-pick on the shared repository** — the merge sequence now runs under the repo lock, eliminating a lost-uncommitted-changes race; a stash-pop failure after successful cherry-picks also no longer reports merged branches as "unmerged" (the duplicate-commit trap) and instead tells the user to pop the stash manually.
|
||||
- Fixed async task batches getting stuck "running" forever (unscheduled/failed-to-register tasks never counted toward completion), error-result jobs being marked `completed`, semaphore-queued tasks counting against the 15-job global cap (batches >15 dropped the remainder and starved other async work), duplicate task ids skipping validation on the async path, and an abort racing subagent session startup leaking the late-created session's LSP/MCP processes.
|
||||
- Fixed task fail-fast abandoning in-flight siblings uncancelled (the worker signal now propagates), patch-mode merges blocking ALL successful siblings' patches when one task failed, and `$@` command expansion interpreting `$`-replacement patterns in user input.
|
||||
- Fixed eval cells double-writing artifacts (the tool and the per-cell executor each opened a sink on the same artifact path, corrupting >50KB outputs), JS `parallel()` early-rejecting in violation of its documented barrier (orphaning in-flight `agent()` thunks with worker-side promises hung forever), Python child subprocesses inheriting the NDJSON frame pipe (their stdout was dropped and could corrupt protocol frames — it is now captured and forwarded), JS cell timeouts silently wiping persistent VM state without annotation, and the JS console bridge throwing on `console.dir`/`time`/`group`/`assert`/`trace`.
|
||||
- Fixed `pr_push` never invalidating the PR/diff cache (the canonical push-then-verify flow read a pre-push diff for up to 5 minutes), current-branch `gh pr merge`/`close` with no positional never invalidating at all (exactly the staleness the cache layer claims to eliminate; numeric flag values like `--milestone 3` also no longer steal the positional), multi-PR checkouts discarding successful checkouts and racing in-flight git mutations on first failure (`allSettled` with per-PR reporting), run-watch ending with a failure result and zero logs when an auto-retry raced the grace-period refetch, the per-watch completed-run job cache serving a rerun's FIRST-attempt jobs after the rerun completed (entries are evicted whenever a run is observed non-completed), pagination terminating on post-filter page length, millisecond precision leaking into GitHub search date qualifiers, leading-dash PR identifiers reaching `gh` as flags, and `issue://?state=` typos silently coercing to the open list.
|
||||
- **Fixed the Exa API key being written to the log file** on every failed MCP request (the key rode the query string of logged URLs; key/token/secret/auth params are now redacted), and **removed the web-search query rewrite that replaced every `202x` substring with the current year** — it corrupted CVE identifiers and made historical-year searches silently impossible.
|
||||
- Fixed reopening the sole browser tab with a different `dialogs` policy disposing Chromium and then using the dead handle, a stale tab release evicting a live replacement browser from the registry (spawning duplicate Chromium processes), and concurrent same-name `open` calls leaking a worker + refcount via a check-then-set race (acquisitions are now single-flight per name); queued opens honor an abort at dequeue, and an init-payload failure releases the temporary browser hold instead of pinning the refcount forever.
|
||||
- Fixed fetch decoding every response as UTF-8 regardless of declared charset (Shift_JIS/EUC-KR/GBK pages rendered as mojibake through the whole reader pipeline; `Content-Type` and `<meta charset>` are now honored via `TextDecoder`), binary URLs being downloaded twice (body skipped on the first pass for convertible types), >50MB truncation being silent (now flagged in notes), all transport error detail being swallowed into a bare "Failed to fetch URL" (the cause is surfaced and 429s get one `Retry-After`-honoring, abort-aware retry), MCP SSE keep-alive lines escaping as raw `SyntaxError`s, MCP calls having no default timeout (now 60s), and a YouTube fetch budget expiry being misreported as a user abort that also skipped temp-file cleanup.
|
||||
- Fixed archive directory listings silently ignoring the selector offset — `a.zip:dir:50` now starts the listing at the 50th entry instead of relisting from the top.
|
||||
|
||||
## [15.10.10] - 2026-06-09
|
||||
|
||||
### Added
|
||||
|
||||
- Added a read-only `view` op to the `todo` tool that echoes the current list without mutating state, so the agent can recover exact task text instead of guessing it from memory.
|
||||
|
||||
### Changed
|
||||
|
||||
- Rewrote the bash tool's coreutils guidance (tool prompt and system prompt) around an explicit litmus: pipelines that compute a new fact (`wc -l`, `sort | uniq -c`, `comm`, `diff`) are legitimate bash, while commands that merely move, page, or trim bytes a dedicated tool can fetch remain banned — output trimming destroys data the `artifact://` capture would have saved.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed the model selector dropping an immediate Enter when cached models were available but the selector's offline refresh was still pending.
|
||||
- Fixed dynamic `import(...)` inside functions passed to the browser tool's `tab.evaluate`/`page.evaluate` failing with `__omp_import__ is not defined`. The eval/browser JS runtime rewrites dynamic-import callees to the worker-injected `__omp_import__` helper, but puppeteer serializes evaluate callbacks with `Function.prototype.toString()` and re-runs them inside the page, where the helper does not exist. The rewriter now substitutes a guarded shim that falls back to native dynamic import when the helper is absent, so serialized code works in the page realm while in-worker imports keep resolving against the session cwd.
|
||||
- Transcript block freezing is now unconditional instead of gated on ED3-risk terminal detection: every finalized block replays its frozen snapshot once it crosses out of the live region, on all terminals including Windows, because the rewritten renderer's committed scrollback is immutable everywhere. Still-mutating blocks (pending tools, streaming messages, async thinking renderers) anchor the live region and keep repainting until they finalize, which structurally fixes stale/duplicated output from late async expansions ([#1823](https://github.com/can1357/oh-my-pi/issues/1823)).
|
||||
- Fixed the edit tool's post-edit diff preview occasionally echoing a context line twice with out-of-order numbering. Block-boundary context injection classified space-prefixed diff rows as old-file-only, so an unchanged line sitting in a net-offset region (old N / new N+k) was missing from the new file's visibility window; `findBlockContextLines` then re-surfaced it under its post-edit number and the row was spliced in after the adjacent change run. New-file boundary lines are now translated back to pre-edit numbers (the compact-preview renumbering contract) and merged into a single old-numbered insertion pass — also fixing closers below a net-offset edit being dropped or renumbered incorrectly.
|
||||
- Fixed the Anthropic web-search provider claiming the Claude Code identity on API-key requests: the CC billing header + system instruction were injected whenever the model wasn't Haiku 3.5, regardless of auth mode. Injection is now OAuth-gated like the streaming path, and OAuth search requests patch the billing header's `cch` attestation (via `wrapFetchForCch`) instead of shipping the `cch=00000` placeholder.
|
||||
- Fixed long streamed content appearing cut off mid-run: scrolled-off rows were erased from the viewport without ever being appended to terminal history. The transcript's commit boundary (`deriveLiveCommitState`) was all-or-nothing per block — one perpetually rewriting row (a task tool's ticking progress tree, per-agent cost/tool counters, spinner stats) suspended scrollback commits for the entire block, so once the block outgrew the viewport its static head (e.g. a task's prompt/context markdown) was neither committed nor on screen until the tool sealed, and was lost outright if the session ended mid-run. A stable-prefix ratchet now promotes leading rows that stayed visibly identical for a full 30-frame window as commit-safe, so the settled head reaches native scrollback while only the genuinely volatile tail stays deferred; a rewrite above the promoted run retreats the boundary and the engine audit recommits (duplication, never loss).
|
||||
- Fixed local tiny-title worker stdout/stderr leaking raw native model output such as `</title>` and cache/status lines into the interactive TUI scrollback ([#2206](https://github.com/can1357/oh-my-pi/issues/2206)).
|
||||
- Fixed task-agent discovery advertising Claude Code custom agents from `.claude/agents/*.md` as OMP subagents; direct task-agent discovery now only loads OMP-native `.omp` agent roots, while Claude marketplace plugin agents keep their existing provider path ([#2209](https://github.com/can1357/oh-my-pi/issues/2209)).
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the `clearOnShrink` setting and its `PI_CLEAR_ON_SHRINK` environment variable: the rewritten renderer always clears shrunken rows exactly, so the flicker/perf tradeoff the setting controlled no longer exists. Existing config entries are ignored.
|
||||
- Removed the prompt-submit native-scrollback reconciliation checkpoint and the eager streaming render mode from the interactive controllers — the renderer's append-only contract made both obsolete.
|
||||
|
||||
## [15.10.9] - 2026-06-09
|
||||
|
||||
@@ -9813,4 +9905,4 @@ Initial public release.
|
||||
- Git branch display in footer
|
||||
- Message queueing during streaming responses
|
||||
- OAuth integration for Gmail and Google Calendar access
|
||||
- HTML export with syntax highlighting and collapsible sections
|
||||
- HTML export with syntax highlighting and collapsible sections
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-coding-agent",
|
||||
"version": "15.10.9",
|
||||
"version": "15.10.10",
|
||||
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -23,6 +23,12 @@ export interface AsyncJob {
|
||||
* supply an id (e.g. legacy tests, SDK consumers without an agent context).
|
||||
*/
|
||||
ownerId?: string;
|
||||
/**
|
||||
* Job is registered but parked behind a caller-managed gate (e.g. a task
|
||||
* batch semaphore). Queued jobs do not count toward the running-job limit
|
||||
* until the caller invokes `markRunning()` from the run context.
|
||||
*/
|
||||
queued?: boolean;
|
||||
}
|
||||
|
||||
export interface AsyncJobManagerOptions {
|
||||
@@ -53,6 +59,8 @@ export interface AsyncJobRegisterOptions {
|
||||
/** Registry id of the agent that owns this job; used to scope cancelAll. */
|
||||
ownerId?: string;
|
||||
onProgress?: (text: string, details?: Record<string, unknown>) => void | Promise<void>;
|
||||
/** Register the job in queued state; see {@link AsyncJob.queued}. */
|
||||
queued?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -110,6 +118,17 @@ export class AsyncJobManager {
|
||||
this.#retentionMs = Math.max(0, Math.floor(options.retentionMs ?? DEFAULT_RETENTION_MS));
|
||||
}
|
||||
|
||||
/** True when the running-job count has reached the configured cap. */
|
||||
get atCapacity(): boolean {
|
||||
if (this.#disposed) return true;
|
||||
// Mirror register(): queued jobs hold no execution slot.
|
||||
let activeCount = 0;
|
||||
for (const job of this.#jobs.values()) {
|
||||
if (job.status === "running" && !job.queued) activeCount++;
|
||||
}
|
||||
return activeCount >= this.#maxRunningJobs;
|
||||
}
|
||||
|
||||
register(
|
||||
type: "bash" | "task",
|
||||
label: string,
|
||||
@@ -117,14 +136,21 @@ export class AsyncJobManager {
|
||||
jobId: string;
|
||||
signal: AbortSignal;
|
||||
reportProgress: (text: string, details?: Record<string, unknown>) => Promise<void>;
|
||||
/** Clear the queued flag once the job actually starts executing. */
|
||||
markRunning: () => void;
|
||||
}) => Promise<string>,
|
||||
options?: AsyncJobRegisterOptions,
|
||||
): string {
|
||||
if (this.#disposed) {
|
||||
throw new Error("Async job manager is disposed");
|
||||
}
|
||||
const runningCount = this.getRunningJobs().length;
|
||||
if (runningCount >= this.#maxRunningJobs) {
|
||||
// Queued jobs hold no execution slot yet — only count jobs that are
|
||||
// actually running so a large parked batch cannot starve registration.
|
||||
let activeCount = 0;
|
||||
for (const existing of this.#jobs.values()) {
|
||||
if (existing.status === "running" && !existing.queued) activeCount++;
|
||||
}
|
||||
if (activeCount >= this.#maxRunningJobs) {
|
||||
throw new Error(
|
||||
`Background job limit reached (${this.#maxRunningJobs}). Wait for running jobs to finish or cancel one.`,
|
||||
);
|
||||
@@ -144,6 +170,7 @@ export class AsyncJobManager {
|
||||
abortController,
|
||||
promise: Promise.resolve(),
|
||||
ownerId: options?.ownerId,
|
||||
queued: options?.queued === true,
|
||||
};
|
||||
|
||||
const reportProgress = async (text: string, details?: Record<string, unknown>): Promise<void> => {
|
||||
@@ -159,7 +186,14 @@ export class AsyncJobManager {
|
||||
};
|
||||
job.promise = (async () => {
|
||||
try {
|
||||
const text = await run({ jobId: id, signal: abortController.signal, reportProgress });
|
||||
const text = await run({
|
||||
jobId: id,
|
||||
signal: abortController.signal,
|
||||
reportProgress,
|
||||
markRunning: () => {
|
||||
job.queued = false;
|
||||
},
|
||||
});
|
||||
if (job.status === "cancelled") {
|
||||
job.resultText = text;
|
||||
this.#scheduleEviction(id);
|
||||
@@ -278,6 +312,26 @@ export class AsyncJobManager {
|
||||
return before - this.#deliveries.length;
|
||||
}
|
||||
|
||||
/**
|
||||
* Lift a foreground-wait suppression set via `acknowledgeDeliveries`. If the
|
||||
* job already finished while suppressed (its delivery enqueue was skipped),
|
||||
* re-enqueue the completion so the result is still delivered exactly once.
|
||||
*/
|
||||
resumeDeliveries(jobIds: string[]): void {
|
||||
for (const rawId of jobIds) {
|
||||
const jobId = rawId.trim();
|
||||
if (!jobId) continue;
|
||||
if (!this.#suppressedDeliveries.delete(jobId)) continue;
|
||||
const job = this.#jobs.get(jobId);
|
||||
if (!job || (job.status !== "completed" && job.status !== "failed")) continue;
|
||||
const queued =
|
||||
this.#deliveries.some(delivery => delivery.jobId === jobId) ||
|
||||
this.#inFlightDeliveries.some(delivery => delivery.jobId === jobId);
|
||||
if (queued) continue;
|
||||
this.#enqueueDelivery(jobId, job.status === "completed" ? (job.resultText ?? "") : (job.errorText ?? ""));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel running jobs. With `filter.ownerId` set, cancels only jobs the
|
||||
* matching agent registered; with no filter, cancels every running job
|
||||
|
||||
@@ -18,16 +18,16 @@ Working directory: `{{working_dir}}`
|
||||
{{baseline_warning}}
|
||||
{{/if}}
|
||||
|
||||
### What you must produce
|
||||
### What you MUST produce
|
||||
|
||||
Write `./autoresearch.sh` at the working directory. It is the canonical benchmark entrypoint and must:
|
||||
Write `./autoresearch.sh` at the working directory. It is the canonical benchmark entrypoint and MUST:
|
||||
|
||||
- exit 0 on success and non-zero on failure;
|
||||
- print the primary metric as a single line `METRIC <name>=<value>`;
|
||||
- print any secondary metrics as additional `METRIC <name>=<value>` lines;
|
||||
- run the same workload deterministically every time (no live network, no time-of-day dependencies, fixed seeds where applicable).
|
||||
|
||||
You **may** edit anything else needed to make `autoresearch.sh` work — benchmark binaries, `Cargo.toml`, `package.json`, helper scripts, fixtures. All those edits are part of the harness baseline and will be committed for you when you call `init_experiment` on an autoresearch branch.
|
||||
You MAY edit anything else needed to make `autoresearch.sh` work — benchmark binaries, `Cargo.toml`, `package.json`, helper scripts, fixtures. All those edits are part of the harness baseline and will be committed for you when you call `init_experiment` on an autoresearch branch.
|
||||
|
||||
### Steps
|
||||
|
||||
@@ -38,6 +38,6 @@ You **may** edit anything else needed to make `autoresearch.sh` work — benchma
|
||||
|
||||
### Rules
|
||||
|
||||
- Do **not** call `run_experiment`, `log_experiment`, or `update_notes` yet. They will error with "no active autoresearch session" until `init_experiment` runs.
|
||||
- Do **not** treat a compile-only check as a benchmark. The harness must actually execute the workload and emit `METRIC`.
|
||||
- Do **not** create `autoresearch.md`, `autoresearch.checks.sh`, `autoresearch.program.md`, `autoresearch.ideas.md`, `autoresearch.jsonl`, `.autoresearch/`, or `autoresearch.config.json`. Session state is tracked for you.
|
||||
- NEVER call `run_experiment`, `log_experiment`, or `update_notes` yet. They will error with "no active autoresearch session" until `init_experiment` runs.
|
||||
- NEVER treat a compile-only check as a benchmark. The harness MUST actually execute the workload and emit `METRIC`.
|
||||
- NEVER create `autoresearch.md`, `autoresearch.checks.sh`, `autoresearch.program.md`, `autoresearch.ideas.md`, `autoresearch.jsonl`, `.autoresearch/`, or `autoresearch.config.json`. Session state is tracked for you.
|
||||
|
||||
@@ -11,17 +11,17 @@ Primary goal:
|
||||
There is no goal recorded for this session yet. Infer what to optimize from the latest user message and the conversation; capture the goal in your notes (`update_notes`) once it is clear.
|
||||
{{/if}}
|
||||
|
||||
Session state and run artifacts are managed for you. The benchmark entrypoint is `bash autoresearch.sh` (committed during Phase 1). Do not edit `autoresearch.sh` mid-segment unless you intentionally bump segment via `init_experiment new_segment: true`. Do not create `autoresearch.md` or `.autoresearch/` in this repo.
|
||||
Session state and run artifacts are managed for you. The benchmark entrypoint is `bash autoresearch.sh` (committed during Phase 1). NEVER edit `autoresearch.sh` mid-segment unless you intentionally bump segment via `init_experiment new_segment: true`. NEVER create `autoresearch.md` or `.autoresearch/` in this repo.
|
||||
|
||||
Working directory: `{{working_dir}}`
|
||||
{{#if has_branch}}Active branch: `{{branch}}`{{/if}}
|
||||
{{#if has_baseline_commit}}Baseline commit: `{{baseline_commit}}`{{/if}}
|
||||
|
||||
You are running an autonomous experiment loop. Keep iterating until the user interrupts you or the configured maximum iteration count is reached.
|
||||
You are running an autonomous experiment loop. You MUST keep iterating until the user interrupts you or the configured maximum iteration count is reached.
|
||||
|
||||
### Available tools
|
||||
- `init_experiment` — open or reconfigure the session. Pass `new_segment: true` to start a fresh baseline within the current session.
|
||||
- `run_experiment` — run the benchmark (`bash autoresearch.sh`). Output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the harness are parsed back to you. The command is fixed; if you need a different workload, edit `autoresearch.sh` and bump segment via `init_experiment new_segment: true`.
|
||||
- `run_experiment` — run the benchmark (`bash autoresearch.sh`). Output is captured automatically and `METRIC name=value` / `ASI key=value` lines printed by the harness are parsed back to you. The command is fixed.
|
||||
- `log_experiment` — record the result. On `keep`, modified files are committed for you; on `discard`/`crash`/`checks_failed`, the worktree is reverted. Pass `flag_runs` to mark earlier runs as suspect; flagged runs are excluded from baseline and best-metric math.
|
||||
- `update_notes` — replace the durable session playbook (`body`) or append to the ideas backlog (`append_idea`). The notes are injected into your system prompt every iteration.
|
||||
|
||||
@@ -97,7 +97,7 @@ Finish the `log_experiment` step before starting another benchmark.
|
||||
{{/if}}
|
||||
|
||||
### Guardrails
|
||||
- Do not game the benchmark.
|
||||
- Do not overfit to synthetic inputs if the real workload is broader.
|
||||
- Preserve correctness.
|
||||
- NEVER game the benchmark.
|
||||
- NEVER overfit to synthetic inputs if the real workload is broader.
|
||||
- MUST preserve correctness.
|
||||
- If the user sends another message while a run is in progress, finish the current run and logging cycle first, then address the new input in the next iteration.
|
||||
|
||||
@@ -15,6 +15,16 @@ export async function readFile(filePath: string): Promise<string | null> {
|
||||
}
|
||||
|
||||
try {
|
||||
// Gate on the file type first: discovery scans foreign config dirs
|
||||
// (~/.claude, ~/.cursor, project trees), and reading a FIFO/socket/char
|
||||
// device with `.text()` blocks until EOF — i.e. forever — hanging
|
||||
// startup with zero output. `stat` follows symlinks, so symlinked
|
||||
// context files (CLAUDE.md -> AGENTS.md) still resolve.
|
||||
const stats = await fs.promises.stat(abs);
|
||||
if (!stats.isFile()) {
|
||||
contentCache.set(abs, null);
|
||||
return null;
|
||||
}
|
||||
const content = await Bun.file(abs).text();
|
||||
contentCache.set(abs, content);
|
||||
return content;
|
||||
|
||||
@@ -32,6 +32,7 @@ export const commands: CommandEntry[] = [
|
||||
{ name: "ssh", load: () => import("./commands/ssh").then(m => m.default) },
|
||||
{ name: "stats", load: () => import("./commands/stats").then(m => m.default) },
|
||||
{ name: "update", load: () => import("./commands/update").then(m => m.default) },
|
||||
{ name: "usage", load: () => import("./commands/usage").then(m => m.default) },
|
||||
{ name: "tiny-models", load: () => import("./commands/tiny-models").then(m => m.default) },
|
||||
{ name: "worktree", load: () => import("./commands/worktree").then(m => m.default), aliases: ["wt"] },
|
||||
{ name: "search", load: () => import("./commands/web-search").then(m => m.default), aliases: ["q"] },
|
||||
|
||||
@@ -64,22 +64,16 @@ export async function listModels(modelRegistry: ModelRegistry, searchPattern?: s
|
||||
}
|
||||
|
||||
const filteredCanonical = modelRegistry
|
||||
.getCanonicalModels({ availableOnly: true, candidates: filteredModels })
|
||||
.map(record => {
|
||||
const selected = modelRegistry.resolveCanonicalModel(record.id, {
|
||||
availableOnly: true,
|
||||
candidates: filteredModels,
|
||||
});
|
||||
if (!selected) return undefined;
|
||||
return {
|
||||
.getCanonicalModelSelections({ availableOnly: true, candidates: filteredModels })
|
||||
.map(
|
||||
({ record, model: selected }): CanonicalRow => ({
|
||||
canonical: record.id,
|
||||
selected: `${selected.provider}/${selected.id}`,
|
||||
variants: String(record.variants.length),
|
||||
context: formatNumber(selected.contextWindow),
|
||||
maxOut: formatNumber(selected.maxTokens),
|
||||
} satisfies CanonicalRow;
|
||||
})
|
||||
.filter((row): row is CanonicalRow => row !== undefined)
|
||||
}),
|
||||
)
|
||||
.sort((left, right) => left.canonical.localeCompare(right.canonical));
|
||||
|
||||
if (filteredModels.length === 0 && filteredCanonical.length === 0) {
|
||||
|
||||
@@ -0,0 +1,603 @@
|
||||
/**
|
||||
* Usage CLI command handler.
|
||||
*
|
||||
* Handles `omp usage` — fetches provider usage reports for every
|
||||
* authenticated account and prints a detailed per-account breakdown
|
||||
* (limits, windows, reset times, plan metadata). Accounts whose
|
||||
* credentials produced no usage report are listed too, so the output
|
||||
* always covers the full credential pool.
|
||||
*/
|
||||
import type { AuthStorage, UsageLimit, UsageReport, UsageUnit } from "@oh-my-pi/pi-ai";
|
||||
import { formatDuration, formatNumber } from "@oh-my-pi/pi-utils";
|
||||
import chalk from "chalk";
|
||||
import { ModelRegistry } from "../config/model-registry";
|
||||
import { discoverAuthStorage } from "../sdk";
|
||||
|
||||
const BAR_WIDTH = 28;
|
||||
|
||||
export interface UsageCommandArgs {
|
||||
json?: boolean;
|
||||
provider?: string;
|
||||
redact?: boolean;
|
||||
}
|
||||
|
||||
/** Identity slice of a stored credential, for "every account" coverage. */
|
||||
export interface UsageAccountIdentity {
|
||||
provider: string;
|
||||
type: "api_key" | "oauth";
|
||||
email?: string;
|
||||
accountId?: string;
|
||||
projectId?: string;
|
||||
enterpriseUrl?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal-reveal masks for identity strings (`--redact`).
|
||||
*
|
||||
* Every mask shows a two-character anchor. When two identities share the
|
||||
* anchor, the mask additionally reveals the shortest "middle-out"
|
||||
* differentiator — the shortest substring (closest to the string's middle on
|
||||
* ties) that no colliding identity contains — as `an*`, `ca*9*`, `ca*nb*`.
|
||||
* Prefix growth is deliberately avoided: it leaks the start of the local
|
||||
* part (`can.boluk@*`) when a couple of mid-string characters suffice.
|
||||
* Duplicate strings (same account on two providers) share a mask.
|
||||
*/
|
||||
export function buildRedactionMap(values: Iterable<string>): Map<string, string> {
|
||||
const unique = [...new Set(values)];
|
||||
const map = new Map<string, string>();
|
||||
const byAnchor = new Map<string, string[]>();
|
||||
for (const value of unique) {
|
||||
const anchor = value.slice(0, 2);
|
||||
const list = byAnchor.get(anchor) ?? [];
|
||||
list.push(value);
|
||||
byAnchor.set(anchor, list);
|
||||
}
|
||||
for (const value of unique) {
|
||||
const anchor = value.slice(0, 2);
|
||||
const peers = (byAnchor.get(anchor) ?? []).filter(other => other !== value);
|
||||
if (peers.length === 0) {
|
||||
map.set(value, `${anchor}*`);
|
||||
continue;
|
||||
}
|
||||
const infix = findDistinguishingInfix(value, peers);
|
||||
map.set(value, infix === undefined ? `${anchor}*` : `${anchor}*${infix}*`);
|
||||
}
|
||||
// Residual collisions (a value whose every substring also occurs in a
|
||||
// peer gets the bare anchor mask) fall back to prefix extension.
|
||||
const byMask = new Map<string, string[]>();
|
||||
for (const value of unique) {
|
||||
const mask = map.get(value)!;
|
||||
const list = byMask.get(mask) ?? [];
|
||||
list.push(value);
|
||||
byMask.set(mask, list);
|
||||
}
|
||||
for (const collided of byMask.values()) {
|
||||
if (collided.length < 2) continue;
|
||||
for (const value of collided) {
|
||||
let length = Math.min(2, value.length);
|
||||
while (
|
||||
length < value.length &&
|
||||
collided.some(other => other !== value && other.startsWith(value.slice(0, length)))
|
||||
) {
|
||||
length++;
|
||||
}
|
||||
map.set(value, `${value.slice(0, length)}*`);
|
||||
}
|
||||
}
|
||||
return map;
|
||||
}
|
||||
|
||||
/**
|
||||
* Shortest substring of `value` (past the revealed two-char anchor) that no
|
||||
* peer contains. Among equal-length candidates, picks the one centered
|
||||
* closest to the middle of the string. Returns undefined when every
|
||||
* substring also occurs in a peer (e.g. `value` is contained in a peer —
|
||||
* that peer's own differentiator keeps the masks distinct).
|
||||
*/
|
||||
function findDistinguishingInfix(value: string, peers: string[]): string | undefined {
|
||||
const start = Math.min(2, value.length);
|
||||
const center = value.length / 2;
|
||||
for (let length = 1; length <= value.length - start; length++) {
|
||||
let best: { infix: string; distance: number } | undefined;
|
||||
for (let pos = start; pos + length <= value.length; pos++) {
|
||||
const candidate = value.slice(pos, pos + length);
|
||||
if (peers.some(peer => peer.includes(candidate))) continue;
|
||||
const distance = Math.abs(pos + length / 2 - center);
|
||||
if (!best || distance < best.distance) best = { infix: candidate, distance };
|
||||
}
|
||||
if (best) return best.infix;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** Every identity string the output could surface — input for {@link buildRedactionMap}. */
|
||||
function collectIdentityStrings(reports: UsageReport[], accounts: UsageAccountIdentity[]): string[] {
|
||||
const values: string[] = [];
|
||||
const add = (value: unknown): void => {
|
||||
if (typeof value === "string" && value) values.push(value);
|
||||
};
|
||||
for (const report of reports) {
|
||||
const meta = report.metadata ?? {};
|
||||
add(meta.email);
|
||||
add(meta.accountId);
|
||||
add(meta.projectId);
|
||||
add(meta.orgId);
|
||||
for (const limit of report.limits) {
|
||||
add(limit.scope.accountId);
|
||||
add(limit.scope.projectId);
|
||||
add(limit.scope.orgId);
|
||||
}
|
||||
}
|
||||
for (const account of accounts) {
|
||||
add(account.email);
|
||||
add(account.accountId);
|
||||
add(account.projectId);
|
||||
add(account.enterpriseUrl);
|
||||
}
|
||||
return values;
|
||||
}
|
||||
|
||||
type LimitStatus = NonNullable<UsageLimit["status"]>;
|
||||
|
||||
function resolveFraction(limit: UsageLimit): number | undefined {
|
||||
const amount = limit.amount;
|
||||
if (amount.usedFraction !== undefined) return amount.usedFraction;
|
||||
if (amount.used !== undefined && amount.limit !== undefined && amount.limit > 0) {
|
||||
return amount.used / amount.limit;
|
||||
}
|
||||
if (amount.unit === "percent" && amount.used !== undefined) return amount.used / 100;
|
||||
if (amount.remainingFraction !== undefined) return Math.max(0, 1 - amount.remainingFraction);
|
||||
return undefined;
|
||||
}
|
||||
|
||||
function resolveStatus(limit: UsageLimit): LimitStatus {
|
||||
if (limit.status && limit.status !== "unknown") return limit.status;
|
||||
const fraction = resolveFraction(limit);
|
||||
if (fraction === undefined) return "unknown";
|
||||
if (fraction >= 1) return "exhausted";
|
||||
if (fraction >= 0.8) return "warning";
|
||||
return "ok";
|
||||
}
|
||||
|
||||
const STATUS_COLOR: Record<LimitStatus, (text: string) => string> = {
|
||||
exhausted: chalk.red,
|
||||
warning: chalk.yellow,
|
||||
ok: chalk.green,
|
||||
unknown: chalk.dim,
|
||||
};
|
||||
|
||||
/** Worst-of aggregation: exhausted > warning > ok > unknown. */
|
||||
function aggregateStatus(limits: UsageLimit[]): LimitStatus {
|
||||
const statuses = limits.map(resolveStatus);
|
||||
if (statuses.includes("exhausted")) return "exhausted";
|
||||
if (statuses.includes("warning")) return "warning";
|
||||
if (statuses.includes("ok")) return "ok";
|
||||
return "unknown";
|
||||
}
|
||||
|
||||
function formatProviderName(provider: string): string {
|
||||
return provider
|
||||
.split(/[-_]/g)
|
||||
.map(part => (part ? part[0].toUpperCase() + part.slice(1) : ""))
|
||||
.join(" ");
|
||||
}
|
||||
|
||||
function formatUnitValue(value: number, unit: UsageUnit): string {
|
||||
if (unit === "usd") return `$${value.toFixed(2)}`;
|
||||
return formatNumber(value);
|
||||
}
|
||||
|
||||
const UNIT_SUFFIX: Record<UsageUnit, string> = {
|
||||
tokens: " tokens",
|
||||
requests: " requests",
|
||||
minutes: " min",
|
||||
bytes: " bytes",
|
||||
percent: "",
|
||||
usd: "",
|
||||
unknown: "",
|
||||
};
|
||||
|
||||
function describeAmount(limit: UsageLimit): string {
|
||||
const amount = limit.amount;
|
||||
const parts: string[] = [];
|
||||
const absoluteUnit = amount.unit !== "percent" && amount.unit !== "unknown";
|
||||
if (absoluteUnit && amount.used !== undefined && amount.limit !== undefined) {
|
||||
parts.push(
|
||||
`${formatUnitValue(amount.used, amount.unit)} / ${formatUnitValue(amount.limit, amount.unit)}${UNIT_SUFFIX[amount.unit]}`,
|
||||
);
|
||||
} else if (absoluteUnit && amount.remaining !== undefined) {
|
||||
parts.push(`${formatUnitValue(amount.remaining, amount.unit)}${UNIT_SUFFIX[amount.unit]} left`);
|
||||
}
|
||||
const fraction = resolveFraction(limit);
|
||||
if (fraction !== undefined) {
|
||||
parts.push(`${(fraction * 100).toFixed(1)}% used`);
|
||||
} else if (amount.remainingFraction !== undefined) {
|
||||
parts.push(`${(amount.remainingFraction * 100).toFixed(1)}% left`);
|
||||
}
|
||||
if (parts.length === 0) parts.push("no data");
|
||||
return parts.join(" · ");
|
||||
}
|
||||
|
||||
function renderBar(limit: UsageLimit): string {
|
||||
const fraction = resolveFraction(limit);
|
||||
if (fraction === undefined) return chalk.dim("·".repeat(BAR_WIDTH));
|
||||
const clamped = Math.min(Math.max(fraction, 0), 1);
|
||||
const filled = Math.round(clamped * BAR_WIDTH);
|
||||
const color = STATUS_COLOR[resolveStatus(limit)];
|
||||
return color("█".repeat(filled)) + chalk.dim("░".repeat(BAR_WIDTH - filled));
|
||||
}
|
||||
|
||||
/** Append the window label when the limit label doesn't already carry it. */
|
||||
function limitTitle(limit: UsageLimit): string {
|
||||
let label = limit.label;
|
||||
const tier = limit.scope.tier;
|
||||
if (tier && !label.toLowerCase().includes(tier.toLowerCase())) label = `${label} (${tier})`;
|
||||
const windowLabel = limit.window?.label ?? limit.scope.windowId;
|
||||
if (!windowLabel) return label;
|
||||
if (windowLabel.toLowerCase() === "quota window") return label;
|
||||
if (label.toLowerCase().includes(windowLabel.toLowerCase())) return label;
|
||||
return `${label} (${windowLabel})`;
|
||||
}
|
||||
|
||||
function reportAccountLabel(report: UsageReport, index: number): string {
|
||||
const meta = report.metadata ?? {};
|
||||
for (const key of ["email", "accountId", "projectId"] as const) {
|
||||
const value = meta[key];
|
||||
if (typeof value === "string" && value) return value;
|
||||
}
|
||||
for (const limit of report.limits) {
|
||||
const scoped = limit.scope.accountId ?? limit.scope.projectId;
|
||||
if (scoped) return scoped;
|
||||
}
|
||||
return `account ${index + 1}`;
|
||||
}
|
||||
|
||||
/** Lowercased identity strings a report can be attributed to. */
|
||||
function reportIdentifiers(report: UsageReport): Set<string> {
|
||||
const ids = new Set<string>();
|
||||
const add = (value: unknown): void => {
|
||||
if (typeof value === "string" && value) ids.add(value.toLowerCase());
|
||||
};
|
||||
const meta = report.metadata ?? {};
|
||||
add(meta.email);
|
||||
add(meta.accountId);
|
||||
add(meta.projectId);
|
||||
add(meta.orgId);
|
||||
for (const limit of report.limits) {
|
||||
add(limit.scope.accountId);
|
||||
add(limit.scope.projectId);
|
||||
add(limit.scope.orgId);
|
||||
}
|
||||
return ids;
|
||||
}
|
||||
|
||||
/**
|
||||
* Stored credentials that no usage report could be attributed to.
|
||||
*
|
||||
* Conservative on purpose: when a provider's reports carry no identity at
|
||||
* all (or the credential is an API key alongside existing reports), we
|
||||
* can't attribute, so we don't claim the account is missing.
|
||||
*/
|
||||
export function collectUnreportedAccounts(
|
||||
reports: UsageReport[],
|
||||
accounts: UsageAccountIdentity[],
|
||||
): UsageAccountIdentity[] {
|
||||
const byProvider = new Map<string, UsageReport[]>();
|
||||
for (const report of reports) {
|
||||
const list = byProvider.get(report.provider) ?? [];
|
||||
list.push(report);
|
||||
byProvider.set(report.provider, list);
|
||||
}
|
||||
return accounts.filter(account => {
|
||||
const providerReports = byProvider.get(account.provider) ?? [];
|
||||
if (providerReports.length === 0) return true;
|
||||
if (account.type === "api_key") return false;
|
||||
const ids = [account.email, account.accountId, account.projectId]
|
||||
.filter((value): value is string => typeof value === "string" && value.length > 0)
|
||||
.map(value => value.toLowerCase());
|
||||
if (ids.length === 0) return false;
|
||||
const reported = new Set<string>();
|
||||
let anyIdentified = false;
|
||||
for (const report of providerReports) {
|
||||
const identifiers = reportIdentifiers(report);
|
||||
if (identifiers.size > 0) anyIdentified = true;
|
||||
for (const id of identifiers) reported.add(id);
|
||||
}
|
||||
if (!anyIdentified) return false;
|
||||
return !ids.some(id => reported.has(id));
|
||||
});
|
||||
}
|
||||
|
||||
function accountIdentityLabel(account: UsageAccountIdentity): string {
|
||||
if (account.type === "api_key") return "API key";
|
||||
return account.email ?? account.accountId ?? account.projectId ?? account.enterpriseUrl ?? "OAuth account";
|
||||
}
|
||||
|
||||
function formatAccountHeader(
|
||||
report: UsageReport,
|
||||
index: number,
|
||||
nowMs: number,
|
||||
redaction?: Map<string, string>,
|
||||
): string {
|
||||
const status = aggregateStatus(report.limits);
|
||||
const icon = STATUS_COLOR[status]("●");
|
||||
const label = reportAccountLabel(report, index);
|
||||
let header = `${icon} ${chalk.bold(redaction?.get(label) ?? label)}`;
|
||||
const planType = report.metadata?.planType;
|
||||
if (typeof planType === "string" && planType) header += chalk.dim(` · plan: ${planType}`);
|
||||
if (report.fetchedAt && nowMs - report.fetchedAt > 90_000) {
|
||||
header += chalk.dim(` · fetched ${formatDuration(nowMs - report.fetchedAt)} ago`);
|
||||
}
|
||||
return header;
|
||||
}
|
||||
|
||||
function formatLimitLine(limit: UsageLimit, labelWidth: number, nowMs: number): string[] {
|
||||
const status = resolveStatus(limit);
|
||||
const title = limitTitle(limit);
|
||||
const padded = title.padEnd(labelWidth);
|
||||
const details: string[] = [describeAmount(limit)];
|
||||
const resetsAt = limit.window?.resetsAt;
|
||||
if (resetsAt !== undefined && resetsAt > nowMs) {
|
||||
details.push(`resets in ${formatDuration(resetsAt - nowMs)}`);
|
||||
}
|
||||
const lines = [
|
||||
` ${STATUS_COLOR[status]("●")} ${padded} ${renderBar(limit)} ${chalk.dim(details.join(" · "))}`,
|
||||
];
|
||||
if (limit.notes && limit.notes.length > 0) {
|
||||
lines.push(` ${chalk.dim(limit.notes.join(" · "))}`);
|
||||
}
|
||||
return lines;
|
||||
}
|
||||
|
||||
/** Per-window capacity stat: how many accounts the current burn requires. */
|
||||
export interface ProviderWindowStat {
|
||||
/** Compact window label, e.g. "5h", "7d". */
|
||||
window: string;
|
||||
durationMs?: number;
|
||||
/** Accounts reporting a limit in this window. */
|
||||
accounts: number;
|
||||
/** Sum of each account's binding used fraction — accounts' worth of quota burned. */
|
||||
usedAccounts: number;
|
||||
/** Accounts the current burn requires: max(1, ceil(usedAccounts)). */
|
||||
needed: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Aggregate one provider's reports into per-window "accounts needed" stats.
|
||||
*
|
||||
* Limits are bucketed by window duration (5h, 7d, ...). Within a bucket each
|
||||
* account contributes its single highest used fraction — when an account has
|
||||
* several meters on the same window (tiered/metered limits), the most-burned
|
||||
* one is what binds.
|
||||
*/
|
||||
export function computeProviderWindowStats(reports: UsageReport[]): ProviderWindowStat[] {
|
||||
const buckets = new Map<string, { window: string; durationMs?: number; fractions: number[] }>();
|
||||
for (const report of reports) {
|
||||
const accountMax = new Map<string, number>();
|
||||
for (const limit of report.limits) {
|
||||
const fraction = resolveFraction(limit);
|
||||
if (fraction === undefined) continue;
|
||||
const durationMs = limit.window?.durationMs;
|
||||
const key =
|
||||
durationMs !== undefined ? `d:${durationMs}` : (limit.scope.windowId ?? limit.window?.label ?? limit.label);
|
||||
const previous = accountMax.get(key);
|
||||
if (previous === undefined || fraction > previous) accountMax.set(key, fraction);
|
||||
if (!buckets.has(key)) {
|
||||
const window =
|
||||
durationMs !== undefined
|
||||
? formatDuration(durationMs)
|
||||
: (limit.window?.label ?? limit.scope.windowId ?? limit.label);
|
||||
buckets.set(key, { window, durationMs, fractions: [] });
|
||||
}
|
||||
}
|
||||
for (const [key, fraction] of accountMax) buckets.get(key)!.fractions.push(fraction);
|
||||
}
|
||||
return [...buckets.values()]
|
||||
.sort((a, b) => (a.durationMs ?? Number.POSITIVE_INFINITY) - (b.durationMs ?? Number.POSITIVE_INFINITY))
|
||||
.map(bucket => {
|
||||
const usedAccounts = bucket.fractions.reduce((sum, fraction) => sum + fraction, 0);
|
||||
return {
|
||||
window: bucket.window,
|
||||
durationMs: bucket.durationMs,
|
||||
accounts: bucket.fractions.length,
|
||||
usedAccounts,
|
||||
needed: Math.max(1, Math.ceil(usedAccounts - 1e-9)),
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Render the full text breakdown: per provider, per account, every limit
|
||||
* with a bar, amounts, and reset times; unattributed credentials trail
|
||||
* each provider section as "no usage data" rows.
|
||||
*/
|
||||
export function formatUsageBreakdown(
|
||||
reports: UsageReport[],
|
||||
accounts: UsageAccountIdentity[],
|
||||
nowMs: number,
|
||||
redaction?: Map<string, string>,
|
||||
): string {
|
||||
const reportsByProvider = new Map<string, UsageReport[]>();
|
||||
for (const report of reports) {
|
||||
const list = reportsByProvider.get(report.provider) ?? [];
|
||||
list.push(report);
|
||||
reportsByProvider.set(report.provider, list);
|
||||
}
|
||||
const unreported = collectUnreportedAccounts(reports, accounts);
|
||||
const unreportedByProvider = new Map<string, UsageAccountIdentity[]>();
|
||||
for (const account of unreported) {
|
||||
const list = unreportedByProvider.get(account.provider) ?? [];
|
||||
list.push(account);
|
||||
unreportedByProvider.set(account.provider, list);
|
||||
}
|
||||
|
||||
const providers = [...new Set([...reportsByProvider.keys(), ...unreportedByProvider.keys()])].sort((a, b) =>
|
||||
a.localeCompare(b),
|
||||
);
|
||||
|
||||
const lines: string[] = [];
|
||||
const latestFetchedAt = Math.max(0, ...reports.map(report => report.fetchedAt ?? 0));
|
||||
const headerSuffix = latestFetchedAt ? chalk.dim(` · fetched ${formatDuration(nowMs - latestFetchedAt)} ago`) : "";
|
||||
lines.push(`${chalk.bold("Usage")}${headerSuffix}`);
|
||||
|
||||
for (const provider of providers) {
|
||||
const providerReports = reportsByProvider.get(provider) ?? [];
|
||||
const providerUnreported = unreportedByProvider.get(provider) ?? [];
|
||||
const accountCount = providerReports.length + providerUnreported.length;
|
||||
lines.push("");
|
||||
lines.push(
|
||||
`${chalk.bold.cyan(formatProviderName(provider))} ${chalk.dim(`— ${accountCount} ${accountCount === 1 ? "account" : "accounts"}`)}`,
|
||||
);
|
||||
|
||||
const labelWidth = providerReports
|
||||
.flatMap(report => report.limits)
|
||||
.reduce((max, limit) => Math.max(max, limitTitle(limit).length), 0);
|
||||
|
||||
providerReports.forEach((report, index) => {
|
||||
lines.push(` ${formatAccountHeader(report, index, nowMs, redaction)}`);
|
||||
if (report.limits.length === 0) {
|
||||
lines.push(` ${chalk.dim("no limits reported")}`);
|
||||
return;
|
||||
}
|
||||
for (const limit of report.limits) {
|
||||
lines.push(...formatLimitLine(limit, labelWidth, nowMs));
|
||||
}
|
||||
});
|
||||
|
||||
for (const account of providerUnreported) {
|
||||
const label = accountIdentityLabel(account);
|
||||
lines.push(` ${chalk.dim("○")} ${chalk.dim(`${redaction?.get(label) ?? label} — no usage data`)}`);
|
||||
}
|
||||
|
||||
const stats = computeProviderWindowStats(providerReports);
|
||||
if (stats.length > 0) {
|
||||
const parts = stats.map(
|
||||
stat =>
|
||||
`${stat.window} → ${stat.needed} of ${stat.accounts} ${stat.accounts === 1 ? "account" : "accounts"} (${stat.usedAccounts.toFixed(2)}× quota burned)`,
|
||||
);
|
||||
lines.push(` ${chalk.dim(`need: ${parts.join(" · ")}`)}`);
|
||||
}
|
||||
}
|
||||
|
||||
return lines.join("\n");
|
||||
}
|
||||
|
||||
function collectStoredAccounts(authStorage: AuthStorage): UsageAccountIdentity[] {
|
||||
const accounts: UsageAccountIdentity[] = [];
|
||||
const all = authStorage.getAll();
|
||||
for (const provider in all) {
|
||||
const entry = all[provider];
|
||||
const credentials = Array.isArray(entry) ? entry : [entry];
|
||||
for (const credential of credentials) {
|
||||
if (credential.type === "oauth") {
|
||||
accounts.push({
|
||||
provider,
|
||||
type: "oauth",
|
||||
email: credential.email,
|
||||
accountId: credential.accountId,
|
||||
projectId: credential.projectId,
|
||||
enterpriseUrl: credential.enterpriseUrl,
|
||||
});
|
||||
} else {
|
||||
accounts.push({ provider, type: "api_key" });
|
||||
}
|
||||
}
|
||||
}
|
||||
return accounts;
|
||||
}
|
||||
|
||||
/** Apply a redaction mask to an optional identity field. */
|
||||
function maskIdentity(redaction: Map<string, string>, value: string | undefined): string | undefined {
|
||||
return value === undefined ? undefined : (redaction.get(value) ?? value);
|
||||
}
|
||||
|
||||
const IDENTITY_METADATA_KEYS = ["email", "accountId", "projectId", "orgId"] as const;
|
||||
|
||||
/** Mask identity fields in a raw-stripped report for `--redact --json`. */
|
||||
function redactReportForJson(
|
||||
report: Omit<UsageReport, "raw">,
|
||||
redaction: Map<string, string>,
|
||||
): Omit<UsageReport, "raw"> {
|
||||
let metadata = report.metadata;
|
||||
if (metadata) {
|
||||
metadata = { ...metadata };
|
||||
for (const key of IDENTITY_METADATA_KEYS) {
|
||||
const value = metadata[key];
|
||||
if (typeof value === "string") metadata[key] = redaction.get(value) ?? value;
|
||||
}
|
||||
}
|
||||
const limits = report.limits.map(limit => ({
|
||||
...limit,
|
||||
scope: {
|
||||
...limit.scope,
|
||||
accountId: maskIdentity(redaction, limit.scope.accountId),
|
||||
projectId: maskIdentity(redaction, limit.scope.projectId),
|
||||
orgId: maskIdentity(redaction, limit.scope.orgId),
|
||||
},
|
||||
}));
|
||||
return { ...report, metadata, limits };
|
||||
}
|
||||
|
||||
export async function runUsageCommand(cmd: UsageCommandArgs): Promise<void> {
|
||||
const authStorage = await discoverAuthStorage();
|
||||
try {
|
||||
const modelRegistry = new ModelRegistry(authStorage);
|
||||
const reports =
|
||||
(await authStorage.fetchUsageReports({
|
||||
baseUrlResolver: provider => modelRegistry.getProviderBaseUrl(provider),
|
||||
})) ?? [];
|
||||
let accounts = collectStoredAccounts(authStorage);
|
||||
let filteredReports = reports;
|
||||
if (cmd.provider) {
|
||||
const wanted = cmd.provider.toLowerCase();
|
||||
filteredReports = reports.filter(report => report.provider.toLowerCase() === wanted);
|
||||
accounts = accounts.filter(account => account.provider.toLowerCase() === wanted);
|
||||
}
|
||||
|
||||
const redaction = cmd.redact ? buildRedactionMap(collectIdentityStrings(filteredReports, accounts)) : undefined;
|
||||
|
||||
if (cmd.json) {
|
||||
// Drop the heavy provider-specific `raw` payload — same shape as the
|
||||
// broker/gateway `/v1/usage` endpoints.
|
||||
let trimmed = filteredReports.map(({ raw: _raw, ...rest }) => rest);
|
||||
let unreportedAccounts = collectUnreportedAccounts(filteredReports, accounts);
|
||||
if (redaction) {
|
||||
trimmed = trimmed.map(report => redactReportForJson(report, redaction));
|
||||
unreportedAccounts = unreportedAccounts.map(account => ({
|
||||
...account,
|
||||
email: maskIdentity(redaction, account.email),
|
||||
accountId: maskIdentity(redaction, account.accountId),
|
||||
projectId: maskIdentity(redaction, account.projectId),
|
||||
enterpriseUrl: maskIdentity(redaction, account.enterpriseUrl),
|
||||
}));
|
||||
}
|
||||
const capacity: Record<string, ProviderWindowStat[]> = {};
|
||||
for (const report of filteredReports) {
|
||||
if (capacity[report.provider]) continue;
|
||||
const stats = computeProviderWindowStats(filteredReports.filter(peer => peer.provider === report.provider));
|
||||
if (stats.length > 0) capacity[report.provider] = stats;
|
||||
}
|
||||
const payload = {
|
||||
generatedAt: Date.now(),
|
||||
reports: trimmed,
|
||||
accountsWithoutUsage: unreportedAccounts,
|
||||
capacity,
|
||||
};
|
||||
process.stdout.write(`${JSON.stringify(payload, null, 2)}\n`);
|
||||
return;
|
||||
}
|
||||
|
||||
if (filteredReports.length === 0 && accounts.length === 0) {
|
||||
const scope = cmd.provider ? ` for provider "${cmd.provider}"` : "";
|
||||
process.stderr.write(
|
||||
chalk.yellow(`No credentials found${scope}. Run \`omp\` and use /login to add accounts.\n`),
|
||||
);
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
process.stdout.write(`${formatUsageBreakdown(filteredReports, accounts, Date.now(), redaction)}\n`);
|
||||
} finally {
|
||||
authStorage.close();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
/**
|
||||
* Show provider usage limits for every authenticated account.
|
||||
*/
|
||||
import { Command, Flags } from "@oh-my-pi/pi-utils/cli";
|
||||
import { runUsageCommand } from "../cli/usage-cli";
|
||||
|
||||
export default class Usage extends Command {
|
||||
static description = "Show provider usage limits for every authenticated account";
|
||||
|
||||
static flags = {
|
||||
json: Flags.boolean({ char: "j", description: "Output usage reports as JSON", default: false }),
|
||||
provider: Flags.string({ char: "p", description: "Only show usage for this provider id (e.g. anthropic)" }),
|
||||
redact: Flags.boolean({
|
||||
char: "r",
|
||||
description: "Redact account emails/ids (shortest unique prefix) for sharing screenshots",
|
||||
default: false,
|
||||
}),
|
||||
};
|
||||
|
||||
static examples = [
|
||||
"# Detailed per-account usage breakdown across all providers\n omp usage",
|
||||
"# Only Anthropic accounts\n omp usage --provider anthropic",
|
||||
"# Redact account identifiers for screenshots\n omp usage --redact",
|
||||
"# Machine-readable output\n omp usage --json",
|
||||
];
|
||||
|
||||
async run(): Promise<void> {
|
||||
const { flags } = await this.parse(Usage);
|
||||
await runUsageCommand({
|
||||
json: flags.json,
|
||||
provider: flags.provider,
|
||||
redact: flags.redact,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -428,6 +428,12 @@ export interface CanonicalModelQueryOptions {
|
||||
candidates?: readonly Model<Api>[];
|
||||
}
|
||||
|
||||
/** A canonical record (with query-filtered variants) plus the variant model selected for it. */
|
||||
export interface CanonicalModelSelection {
|
||||
record: CanonicalModelRecord;
|
||||
model: Model<Api>;
|
||||
}
|
||||
|
||||
/** Result of loading custom models from models.json */
|
||||
interface CustomModelsResult {
|
||||
models?: CustomModelOverlay[];
|
||||
@@ -768,8 +774,19 @@ function buildCustomReferenceSuffixAliasMap(exactReferences: ReadonlyMap<string,
|
||||
return aliases;
|
||||
}
|
||||
|
||||
const customReferenceMap = buildCustomReferenceMap();
|
||||
const customReferenceSuffixAliasMap = buildCustomReferenceSuffixAliasMap(customReferenceMap);
|
||||
// Lazy: building these maps walks every bundled model (~12K) and triggers
|
||||
// model enrichment in pi-ai; defer off module load until the first
|
||||
// custom-model reference lookup actually needs them.
|
||||
let customReferenceMap: Map<string, Model<Api>> | undefined;
|
||||
let customReferenceSuffixAliasMap: Map<string, Model<Api>> | undefined;
|
||||
|
||||
function getCustomReferenceMaps(): { exact: Map<string, Model<Api>>; suffixAlias: Map<string, Model<Api>> } {
|
||||
if (customReferenceMap === undefined || customReferenceSuffixAliasMap === undefined) {
|
||||
customReferenceMap = buildCustomReferenceMap();
|
||||
customReferenceSuffixAliasMap = buildCustomReferenceSuffixAliasMap(customReferenceMap);
|
||||
}
|
||||
return { exact: customReferenceMap, suffixAlias: customReferenceSuffixAliasMap };
|
||||
}
|
||||
|
||||
const CUSTOM_REFERENCE_TRAILING_MARKER_PATTERN =
|
||||
/[-:](?:thinking|customtools|high|low|medium|minimal|xhigh|free|cloud|exacto|nitro|original|optimized|nvfp4|fp8|fp4|bf16|int8|int4|search)$/i;
|
||||
@@ -824,9 +841,10 @@ function getCustomReferenceCandidateIds(modelId: string): string[] {
|
||||
}
|
||||
|
||||
function resolveCustomModelReference(modelId: string): Model<Api> | undefined {
|
||||
const { exact, suffixAlias } = getCustomReferenceMaps();
|
||||
for (const candidate of getCustomReferenceCandidateIds(modelId)) {
|
||||
const key = normalizeCustomReferenceKey(candidate);
|
||||
const reference = customReferenceMap.get(key) ?? customReferenceSuffixAliasMap.get(key);
|
||||
const reference = exact.get(key) ?? suffixAlias.get(key);
|
||||
if (reference) return reference;
|
||||
}
|
||||
return undefined;
|
||||
@@ -2217,48 +2235,81 @@ export class ModelRegistry {
|
||||
return this.#models;
|
||||
}
|
||||
|
||||
#isModelAvailable(model: Model<Api>): boolean {
|
||||
/**
|
||||
* Availability predicate with per-provider memoization. Auth lookups
|
||||
* (`authStorage.hasAuth`) and the disabled-provider set are resolved once
|
||||
* per provider instead of once per model, which matters when filtering the
|
||||
* full bundled catalog (thousands of models, ~50 providers).
|
||||
*/
|
||||
#createAvailabilityCheck(): (model: Model<Api>) => boolean {
|
||||
const disabledProviders = getDisabledProviderIdsFromSettings();
|
||||
return (
|
||||
!disabledProviders.has(model.provider) &&
|
||||
(this.#keylessProviders.has(model.provider) || this.authStorage.hasAuth(model.provider))
|
||||
);
|
||||
const byProvider = new Map<string, boolean>();
|
||||
return model => {
|
||||
let available = byProvider.get(model.provider);
|
||||
if (available === undefined) {
|
||||
available =
|
||||
!disabledProviders.has(model.provider) &&
|
||||
(this.#keylessProviders.has(model.provider) || this.authStorage.hasAuth(model.provider));
|
||||
byProvider.set(model.provider, available);
|
||||
}
|
||||
return available;
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the shared per-query filter state for canonical model queries.
|
||||
* Hoisted out of the per-record loop: building the candidate-selector set
|
||||
* and availability memo once per query instead of once per record is what
|
||||
* keeps `getCanonicalModelSelections` linear instead of O(records × candidates).
|
||||
*/
|
||||
#canonicalQueryFilters(options: CanonicalModelQueryOptions | undefined): {
|
||||
candidateKeys: Set<string> | undefined;
|
||||
isAvailable: ((model: Model<Api>) => boolean) | undefined;
|
||||
} {
|
||||
return {
|
||||
candidateKeys: options?.candidates
|
||||
? new Set(options.candidates.map(candidate => formatCanonicalVariantSelector(candidate)))
|
||||
: undefined,
|
||||
isAvailable: options?.availableOnly ? this.#createAvailabilityCheck() : undefined,
|
||||
};
|
||||
}
|
||||
|
||||
#filterCanonicalVariants(
|
||||
record: CanonicalModelRecord,
|
||||
options: CanonicalModelQueryOptions | undefined,
|
||||
candidateKeys: ReadonlySet<string> | undefined,
|
||||
isAvailable: ((model: Model<Api>) => boolean) | undefined,
|
||||
): CanonicalModelVariant[] {
|
||||
const candidateKeys = options?.candidates
|
||||
? new Set(options.candidates.map(candidate => formatCanonicalVariantSelector(candidate)))
|
||||
: undefined;
|
||||
return record.variants.filter(variant => {
|
||||
if (candidateKeys && !candidateKeys.has(variant.selector)) {
|
||||
return false;
|
||||
}
|
||||
if (options?.availableOnly && !this.#isModelAvailable(variant.model)) {
|
||||
if (isAvailable && !isAvailable(variant.model)) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
});
|
||||
}
|
||||
|
||||
#buildModelOrder(candidates: readonly Model<Api>[]): Map<string, number> {
|
||||
const modelOrder = new Map<string, number>();
|
||||
for (let index = 0; index < candidates.length; index += 1) {
|
||||
modelOrder.set(formatCanonicalVariantSelector(candidates[index]!), index);
|
||||
}
|
||||
return modelOrder;
|
||||
}
|
||||
|
||||
#providerRank(): Map<string, number> {
|
||||
return buildModelProviderPriorityRank(getConfiguredProviderOrderFromSettings());
|
||||
}
|
||||
|
||||
#resolveCanonicalVariant(
|
||||
variants: readonly CanonicalModelVariant[],
|
||||
allCandidates: readonly Model<Api>[],
|
||||
modelOrder: ReadonlyMap<string, number>,
|
||||
providerRank: ReadonlyMap<string, number>,
|
||||
): CanonicalModelVariant | undefined {
|
||||
if (variants.length === 0) {
|
||||
return undefined;
|
||||
}
|
||||
const providerRank = this.#providerRank();
|
||||
const modelOrder = new Map<string, number>();
|
||||
for (let index = 0; index < allCandidates.length; index += 1) {
|
||||
modelOrder.set(formatCanonicalVariantSelector(allCandidates[index]!), index);
|
||||
}
|
||||
const sourceRank: Record<CanonicalModelVariant["source"], number> = {
|
||||
override: 1,
|
||||
bundled: 1,
|
||||
@@ -2289,9 +2340,10 @@ export class ModelRegistry {
|
||||
}
|
||||
|
||||
getCanonicalModels(options?: CanonicalModelQueryOptions): CanonicalModelRecord[] {
|
||||
const { candidateKeys, isAvailable } = this.#canonicalQueryFilters(options);
|
||||
const records: CanonicalModelRecord[] = [];
|
||||
for (const record of this.#canonicalIndex.records) {
|
||||
const variants = this.#filterCanonicalVariants(record, options);
|
||||
const variants = this.#filterCanonicalVariants(record, candidateKeys, isAvailable);
|
||||
if (variants.length === 0) {
|
||||
continue;
|
||||
}
|
||||
@@ -2304,12 +2356,43 @@ export class ModelRegistry {
|
||||
return records;
|
||||
}
|
||||
|
||||
/**
|
||||
* One-pass equivalent of `getCanonicalModels` + `resolveCanonicalModel` per
|
||||
* record. The per-query state (candidate-selector set, availability memo,
|
||||
* provider rank, candidate order) is built once, so the whole catalog
|
||||
* resolves in O(records + candidates) instead of O(records × candidates).
|
||||
* This is the path the model selector hydrates from synchronously on open.
|
||||
*/
|
||||
getCanonicalModelSelections(options?: CanonicalModelQueryOptions): CanonicalModelSelection[] {
|
||||
const { candidateKeys, isAvailable } = this.#canonicalQueryFilters(options);
|
||||
const candidates = options?.candidates ?? (options?.availableOnly ? this.getAvailable() : this.getAll());
|
||||
const modelOrder = this.#buildModelOrder(candidates);
|
||||
const providerRank = this.#providerRank();
|
||||
const selections: CanonicalModelSelection[] = [];
|
||||
for (const record of this.#canonicalIndex.records) {
|
||||
const variants = this.#filterCanonicalVariants(record, candidateKeys, isAvailable);
|
||||
if (variants.length === 0) {
|
||||
continue;
|
||||
}
|
||||
const resolved = this.#resolveCanonicalVariant(variants, modelOrder, providerRank);
|
||||
if (!resolved) {
|
||||
continue;
|
||||
}
|
||||
selections.push({
|
||||
record: { id: record.id, name: record.name, variants },
|
||||
model: resolved.model,
|
||||
});
|
||||
}
|
||||
return selections;
|
||||
}
|
||||
|
||||
getCanonicalVariants(canonicalId: string, options?: CanonicalModelQueryOptions): CanonicalModelVariant[] {
|
||||
const record = this.#canonicalIndex.byId.get(canonicalId.trim().toLowerCase());
|
||||
if (!record) {
|
||||
return [];
|
||||
}
|
||||
return this.#filterCanonicalVariants(record, options);
|
||||
const { candidateKeys, isAvailable } = this.#canonicalQueryFilters(options);
|
||||
return this.#filterCanonicalVariants(record, candidateKeys, isAvailable);
|
||||
}
|
||||
|
||||
resolveCanonicalModel(canonicalId: string, options?: CanonicalModelQueryOptions): Model<Api> | undefined {
|
||||
@@ -2318,7 +2401,7 @@ export class ModelRegistry {
|
||||
return undefined;
|
||||
}
|
||||
const candidates = options?.candidates ?? (options?.availableOnly ? this.getAvailable() : this.getAll());
|
||||
return this.#resolveCanonicalVariant(variants, candidates)?.model;
|
||||
return this.#resolveCanonicalVariant(variants, this.#buildModelOrder(candidates), this.#providerRank())?.model;
|
||||
}
|
||||
|
||||
getCanonicalId(model: Model<Api>): string | undefined {
|
||||
@@ -2330,7 +2413,7 @@ export class ModelRegistry {
|
||||
* This is a fast check that doesn't refresh OAuth tokens.
|
||||
*/
|
||||
getAvailable(): Model<Api>[] {
|
||||
return this.#models.filter(model => this.#isModelAvailable(model));
|
||||
return this.#models.filter(this.#createAvailabilityCheck());
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -246,7 +246,11 @@ export const DEFAULT_BASH_INTERCEPTOR_RULES: BashInterceptorRule[] = [
|
||||
message: "Use the `edit` tool instead of awk -i inplace. It provides diff preview and fuzzy matching.",
|
||||
},
|
||||
{
|
||||
pattern: "^\\s*(echo|printf|cat\\s*<<)\\s+.*[^|]>\\s*\\S",
|
||||
// `>` must sit outside quoted regions (so `echo "a -> b"` passes) and be
|
||||
// followed by a plausible filename — including `$VAR` targets; `>|`
|
||||
// (clobber) counts as a redirect; `>&2`/`2>&1` style fd duplication is
|
||||
// not matched.
|
||||
pattern: "^\\s*(echo|printf|cat\\s*<<)\\s+(?:[^\"'>]|\"[^\"]*\"|'[^']*')*(?<!\\|)>{1,2}\\|?\\s*[$\\w./~\"'-]",
|
||||
tool: "write",
|
||||
message: "Use the `write` tool instead of echo/cat redirection. It handles encoding and provides confirmation.",
|
||||
},
|
||||
@@ -686,16 +690,6 @@ export const SETTINGS_SCHEMA = {
|
||||
ui: { tab: "appearance", label: "Show Hardware Cursor", description: "Show terminal cursor for IME support" },
|
||||
},
|
||||
|
||||
clearOnShrink: {
|
||||
type: "boolean",
|
||||
default: false,
|
||||
ui: {
|
||||
tab: "appearance",
|
||||
label: "Clear on Shrink",
|
||||
description: "Clear empty rows when content shrinks (may cause flicker)",
|
||||
},
|
||||
},
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Model
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -29,32 +29,67 @@ type DapReverseRequestHandler = (args: unknown) => unknown | Promise<unknown>;
|
||||
|
||||
const DEFAULT_REQUEST_TIMEOUT_MS = 30_000;
|
||||
|
||||
function findHeaderEnd(buffer: Uint8Array): number {
|
||||
for (let index = 0; index < buffer.length - 3; index += 1) {
|
||||
if (buffer[index] === 13 && buffer[index + 1] === 10 && buffer[index + 2] === 13 && buffer[index + 3] === 10) {
|
||||
return index;
|
||||
// Reused for all full decodes; each decode() resets state, so a single
|
||||
// instance is safe and avoids per-message TextDecoder allocation.
|
||||
const MESSAGE_DECODER = new TextDecoder("utf-8");
|
||||
|
||||
/**
|
||||
* Locate the `\r\n\r\n` header terminator across the pending chunk list.
|
||||
* Returns the absolute byte index of the first `\r`, or -1 when not present.
|
||||
* Equivalent to scanning the contiguous concatenation of the chunks.
|
||||
*/
|
||||
function findHeaderEndInChunks(chunks: Buffer[]): number {
|
||||
let global = 0;
|
||||
let b0 = -1;
|
||||
let b1 = -1;
|
||||
let b2 = -1;
|
||||
for (const chunk of chunks) {
|
||||
for (let i = 0; i < chunk.length; i++) {
|
||||
const b3 = chunk[i];
|
||||
if (b0 === 13 && b1 === 10 && b2 === 13 && b3 === 10) {
|
||||
return global - 3;
|
||||
}
|
||||
b0 = b1;
|
||||
b1 = b2;
|
||||
b2 = b3;
|
||||
global++;
|
||||
}
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
function parseMessage(
|
||||
buffer: Buffer,
|
||||
): { message: DapResponseMessage | DapEventMessage | DapRequestMessage; remaining: Buffer } | null {
|
||||
const headerEndIndex = findHeaderEnd(buffer);
|
||||
if (headerEndIndex === -1) return null;
|
||||
const headerText = new TextDecoder().decode(buffer.slice(0, headerEndIndex));
|
||||
const contentLengthMatch = headerText.match(/Content-Length: (\d+)/i);
|
||||
if (!contentLengthMatch) return null;
|
||||
const contentLength = Number.parseInt(contentLengthMatch[1], 10);
|
||||
const messageStart = headerEndIndex + 4;
|
||||
const messageEnd = messageStart + contentLength;
|
||||
if (buffer.length < messageEnd) return null;
|
||||
const messageText = new TextDecoder().decode(buffer.subarray(messageStart, messageEnd));
|
||||
return {
|
||||
message: JSON.parse(messageText) as DapResponseMessage | DapEventMessage | DapRequestMessage,
|
||||
remaining: buffer.subarray(messageEnd),
|
||||
};
|
||||
/** Copy the byte range [from, to) out of the pending chunk list into one Buffer. */
|
||||
function copyChunkRange(chunks: Buffer[], from: number, to: number): Buffer {
|
||||
const out = Buffer.allocUnsafe(to - from);
|
||||
let global = 0;
|
||||
let written = 0;
|
||||
for (const chunk of chunks) {
|
||||
const chunkEnd = global + chunk.length;
|
||||
if (chunkEnd > from && global < to) {
|
||||
const start = Math.max(from, global) - global;
|
||||
const end = Math.min(to, chunkEnd) - global;
|
||||
chunk.copy(out, written, start, end);
|
||||
written += end - start;
|
||||
}
|
||||
global = chunkEnd;
|
||||
if (global >= to) break;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Drop the first `count` bytes from the pending chunk list in place. */
|
||||
function dropChunkFront(chunks: Buffer[], count: number): void {
|
||||
let removed = 0;
|
||||
while (chunks.length > 0) {
|
||||
const head = chunks[0];
|
||||
if (removed + head.length <= count) {
|
||||
removed += head.length;
|
||||
chunks.shift();
|
||||
} else {
|
||||
chunks[0] = head.subarray(count - removed);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function writeMessage(sink: DapWriteSink, message: DapRequestMessage | DapResponseMessage): Promise<void> {
|
||||
@@ -81,7 +116,7 @@ export class DapClient {
|
||||
readonly #socket?: { end(): void };
|
||||
#requestSeq = 0;
|
||||
#pendingRequests = new Map<number, DapPendingRequest>();
|
||||
#messageBuffer = Buffer.alloc(0);
|
||||
#messageBuffer: Buffer = Buffer.alloc(0);
|
||||
#isReading = false;
|
||||
#disposed = false;
|
||||
#lastActivity = Date.now();
|
||||
@@ -416,32 +451,84 @@ export class DapClient {
|
||||
if (this.#isReading) return;
|
||||
this.#isReading = true;
|
||||
const reader = this.#readable.getReader();
|
||||
|
||||
// Incoming bytes are buffered as a list of chunks and only joined when a
|
||||
// full message is framed (mirrors the LSP reader) — concatenating the
|
||||
// accumulator on every read is O(n^2) for messages spanning many reads.
|
||||
const pendingChunks: Buffer[] = [];
|
||||
let pendingLen = 0;
|
||||
if (this.#messageBuffer.length > 0) {
|
||||
pendingChunks.push(this.#messageBuffer);
|
||||
pendingLen = this.#messageBuffer.length;
|
||||
}
|
||||
|
||||
try {
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
const currentBuffer = Buffer.concat([this.#messageBuffer, value]);
|
||||
this.#messageBuffer = currentBuffer;
|
||||
let workingBuffer = currentBuffer;
|
||||
let parsed = parseMessage(workingBuffer);
|
||||
while (parsed) {
|
||||
const { message, remaining } = parsed;
|
||||
workingBuffer = Buffer.from(remaining);
|
||||
this.#lastActivity = Date.now();
|
||||
if (message.type === "response") {
|
||||
this.#handleResponse(message);
|
||||
} else if (message.type === "event") {
|
||||
await this.#dispatchEvent(message);
|
||||
} else {
|
||||
await this.#handleAdapterRequest(message);
|
||||
|
||||
pendingChunks.push(Buffer.from(value));
|
||||
pendingLen += value.length;
|
||||
|
||||
// Drain every complete message currently buffered.
|
||||
while (true) {
|
||||
const headerEnd = findHeaderEndInChunks(pendingChunks);
|
||||
if (headerEnd === -1) break;
|
||||
|
||||
const headerText = MESSAGE_DECODER.decode(copyChunkRange(pendingChunks, 0, headerEnd));
|
||||
const contentLengthMatch = headerText.match(/Content-Length: (\d+)/i);
|
||||
if (!contentLengthMatch) {
|
||||
// Non-protocol bytes (e.g. an adapter printing to stdout).
|
||||
// Drop past the bogus terminator and resync instead of
|
||||
// stalling on the same junk header forever.
|
||||
logger.warn("DAP framing resync: header block without Content-Length", {
|
||||
adapter: this.adapter.name,
|
||||
header: headerText.slice(0, 200),
|
||||
});
|
||||
dropChunkFront(pendingChunks, headerEnd + 4);
|
||||
pendingLen -= headerEnd + 4;
|
||||
continue;
|
||||
}
|
||||
|
||||
const contentLength = Number.parseInt(contentLengthMatch[1], 10);
|
||||
const messageStart = headerEnd + 4; // Skip \r\n\r\n
|
||||
const messageEnd = messageStart + contentLength;
|
||||
if (pendingLen < messageEnd) break;
|
||||
|
||||
const messageText = MESSAGE_DECODER.decode(copyChunkRange(pendingChunks, messageStart, messageEnd));
|
||||
dropChunkFront(pendingChunks, messageEnd);
|
||||
pendingLen -= messageEnd;
|
||||
this.#lastActivity = Date.now();
|
||||
|
||||
// A malformed message must not kill the reader — later
|
||||
// messages are still well-framed.
|
||||
try {
|
||||
const message = JSON.parse(messageText) as DapResponseMessage | DapEventMessage | DapRequestMessage;
|
||||
if (message.type === "response") {
|
||||
this.#handleResponse(message);
|
||||
} else if (message.type === "event") {
|
||||
await this.#dispatchEvent(message);
|
||||
} else {
|
||||
await this.#handleAdapterRequest(message);
|
||||
}
|
||||
} catch (error) {
|
||||
logger.warn("DAP message handling failed", {
|
||||
adapter: this.adapter.name,
|
||||
error: toErrorMessage(error),
|
||||
});
|
||||
}
|
||||
parsed = parseMessage(workingBuffer);
|
||||
}
|
||||
this.#messageBuffer = workingBuffer;
|
||||
}
|
||||
} catch (error) {
|
||||
this.#rejectPendingRequests(new Error(`DAP connection closed: ${toErrorMessage(error)}`));
|
||||
} finally {
|
||||
// Persist any unparsed remainder so a restarted reader resumes mid-message.
|
||||
this.#messageBuffer =
|
||||
pendingChunks.length === 0
|
||||
? Buffer.alloc(0)
|
||||
: pendingChunks.length === 1
|
||||
? pendingChunks[0]
|
||||
: Buffer.concat(pendingChunks, pendingLen);
|
||||
reader.releaseLock();
|
||||
this.#isReading = false;
|
||||
}
|
||||
|
||||
@@ -76,8 +76,14 @@ interface DapSession {
|
||||
functionBreakpoints: DapFunctionBreakpointRecord[];
|
||||
instructionBreakpoints: DapInstructionBreakpoint[];
|
||||
dataBreakpoints: DapDataBreakpoint[];
|
||||
output: string;
|
||||
/** Serializes breakpoint mutations — see #serializeBreakpointMutation. */
|
||||
breakpointMutationQueue: Promise<void>;
|
||||
/** Recent output chunks; trimmed from the front when over MAX_OUTPUT_BYTES. */
|
||||
outputChunks: string[];
|
||||
/** Cumulative bytes of output ever received (reported in summaries). */
|
||||
outputBytes: number;
|
||||
/** Bytes currently buffered in outputChunks. */
|
||||
outputBufferedBytes: number;
|
||||
outputTruncated: boolean;
|
||||
stop: DapStopLocation;
|
||||
threads: DapThread[];
|
||||
@@ -175,10 +181,31 @@ function normalizePath(filePath: string): string {
|
||||
|
||||
function truncateOutput(session: DapSession, output: string): void {
|
||||
if (!output) return;
|
||||
session.output += output;
|
||||
session.outputBytes += Buffer.byteLength(output, "utf-8");
|
||||
while (Buffer.byteLength(session.output, "utf-8") > MAX_OUTPUT_BYTES) {
|
||||
session.output = session.output.slice(Math.min(1024, session.output.length));
|
||||
const bytes = Buffer.byteLength(output, "utf-8");
|
||||
session.outputChunks.push(output);
|
||||
session.outputBytes += bytes;
|
||||
session.outputBufferedBytes += bytes;
|
||||
// Trim whole chunks from the front, but only while the remainder still
|
||||
// holds a full MAX_OUTPUT_BYTES tail — dropping the front chunk whenever
|
||||
// the total exceeded the cap could retain far less than the cap (e.g.
|
||||
// [120KB, 10KB] would keep only 10KB). Recomputing one big string's byte
|
||||
// length per 1KB trim iteration was O(n^2) inside the event dispatch loop.
|
||||
while (session.outputChunks.length > 1) {
|
||||
const frontBytes = Buffer.byteLength(session.outputChunks[0], "utf-8");
|
||||
if (session.outputBufferedBytes - frontBytes < MAX_OUTPUT_BYTES) break;
|
||||
session.outputChunks.shift();
|
||||
session.outputBufferedBytes -= frontBytes;
|
||||
session.outputTruncated = true;
|
||||
}
|
||||
if (session.outputBufferedBytes > MAX_OUTPUT_BYTES) {
|
||||
// Byte-slice the front chunk's head so exactly the cap remains (a torn
|
||||
// code point at the cut decodes as U+FFFD, acceptable for log output).
|
||||
const front = session.outputChunks[0];
|
||||
const frontBytes = Buffer.byteLength(front, "utf-8");
|
||||
const excess = session.outputBufferedBytes - MAX_OUTPUT_BYTES;
|
||||
const kept = Buffer.from(front, "utf-8").subarray(excess).toString("utf-8");
|
||||
session.outputChunks[0] = kept;
|
||||
session.outputBufferedBytes += Buffer.byteLength(kept, "utf-8") - frontBytes;
|
||||
session.outputTruncated = true;
|
||||
}
|
||||
}
|
||||
@@ -368,6 +395,26 @@ export class DapSessionManager {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Serialize breakpoint mutations per session: every mutator does a
|
||||
* read-modify-write of session state around an await, and the adapter-side
|
||||
* set*Breakpoints request replaces the whole list — concurrent mutations
|
||||
* would silently drop each other's breakpoints on both sides.
|
||||
*/
|
||||
#serializeBreakpointMutation<T>(session: DapSession, mutate: () => Promise<T>, signal?: AbortSignal): Promise<T> {
|
||||
const run = session.breakpointMutationQueue.then(() => {
|
||||
// A mutation can sit behind several queued 30s predecessors; honor a
|
||||
// caller abort at dequeue instead of running a request nobody awaits.
|
||||
if (signal?.aborted) throw signal.reason instanceof Error ? signal.reason : new Error("Aborted");
|
||||
return mutate();
|
||||
});
|
||||
session.breakpointMutationQueue = run.then(
|
||||
() => undefined,
|
||||
() => undefined,
|
||||
);
|
||||
return run;
|
||||
}
|
||||
|
||||
async setBreakpoint(
|
||||
file: string,
|
||||
line: number,
|
||||
@@ -376,99 +423,123 @@ export class DapSessionManager {
|
||||
timeoutMs: number = 30_000,
|
||||
) {
|
||||
const session = this.#touchActiveSession();
|
||||
const sourcePath = normalizePath(file);
|
||||
const current = [...(session.breakpoints.get(sourcePath) ?? [])];
|
||||
const deduped = current.filter(entry => entry.line !== line);
|
||||
deduped.push({ verified: false, line, condition });
|
||||
deduped.sort((left, right) => left.line - right.line);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setBreakpoints",
|
||||
{
|
||||
source: { path: sourcePath, name: path.basename(sourcePath) },
|
||||
breakpoints: deduped.map<DapSourceBreakpoint>(entry => ({
|
||||
line: entry.line,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
async () => {
|
||||
const sourcePath = normalizePath(file);
|
||||
const current = [...(session.breakpoints.get(sourcePath) ?? [])];
|
||||
const deduped = current.filter(entry => entry.line !== line);
|
||||
deduped.push({ verified: false, line, condition });
|
||||
deduped.sort((left, right) => left.line - right.line);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setBreakpoints",
|
||||
{
|
||||
source: { path: sourcePath, name: path.basename(sourcePath) },
|
||||
breakpoints: deduped.map<DapSourceBreakpoint>(entry => ({
|
||||
line: entry.line,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.breakpoints.set(sourcePath, this.#mapSourceBreakpoints(deduped, response?.breakpoints));
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: session.breakpoints.get(sourcePath) ?? [],
|
||||
sourcePath,
|
||||
};
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.breakpoints.set(sourcePath, this.#mapSourceBreakpoints(deduped, response?.breakpoints));
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: session.breakpoints.get(sourcePath) ?? [],
|
||||
sourcePath,
|
||||
};
|
||||
}
|
||||
|
||||
async removeBreakpoint(file: string, line: number, signal?: AbortSignal, timeoutMs: number = 30_000) {
|
||||
const session = this.#touchActiveSession();
|
||||
const sourcePath = normalizePath(file);
|
||||
const current = [...(session.breakpoints.get(sourcePath) ?? [])].filter(entry => entry.line !== line);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setBreakpoints",
|
||||
{
|
||||
source: { path: sourcePath, name: path.basename(sourcePath) },
|
||||
breakpoints: current.map<DapSourceBreakpoint>(entry => ({
|
||||
line: entry.line,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
async () => {
|
||||
const sourcePath = normalizePath(file);
|
||||
const current = [...(session.breakpoints.get(sourcePath) ?? [])].filter(entry => entry.line !== line);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setBreakpoints",
|
||||
{
|
||||
source: { path: sourcePath, name: path.basename(sourcePath) },
|
||||
breakpoints: current.map<DapSourceBreakpoint>(entry => ({
|
||||
line: entry.line,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
if (current.length === 0) {
|
||||
session.breakpoints.delete(sourcePath);
|
||||
} else {
|
||||
session.breakpoints.set(sourcePath, this.#mapSourceBreakpoints(current, response?.breakpoints));
|
||||
}
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: session.breakpoints.get(sourcePath) ?? [],
|
||||
sourcePath,
|
||||
};
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
if (current.length === 0) {
|
||||
session.breakpoints.delete(sourcePath);
|
||||
} else {
|
||||
session.breakpoints.set(sourcePath, this.#mapSourceBreakpoints(current, response?.breakpoints));
|
||||
}
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: session.breakpoints.get(sourcePath) ?? [],
|
||||
sourcePath,
|
||||
};
|
||||
}
|
||||
|
||||
async setFunctionBreakpoint(name: string, condition?: string, signal?: AbortSignal, timeoutMs: number = 30_000) {
|
||||
const session = this.#touchActiveSession();
|
||||
const current = session.functionBreakpoints.filter(entry => entry.name !== name);
|
||||
current.push({ verified: false, name, condition });
|
||||
current.sort((left, right) => left.name.localeCompare(right.name));
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setFunctionBreakpoints",
|
||||
{
|
||||
breakpoints: current.map<DapFunctionBreakpoint>(entry => ({
|
||||
name: entry.name,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
async () => {
|
||||
const current = session.functionBreakpoints.filter(entry => entry.name !== name);
|
||||
current.push({ verified: false, name, condition });
|
||||
current.sort((left, right) => left.name.localeCompare(right.name));
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setFunctionBreakpoints",
|
||||
{
|
||||
breakpoints: current.map<DapFunctionBreakpoint>(entry => ({
|
||||
name: entry.name,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.functionBreakpoints = this.#mapFunctionBreakpoints(current, response?.breakpoints);
|
||||
return { snapshot: buildSummary(session), breakpoints: session.functionBreakpoints };
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.functionBreakpoints = this.#mapFunctionBreakpoints(current, response?.breakpoints);
|
||||
return { snapshot: buildSummary(session), breakpoints: session.functionBreakpoints };
|
||||
}
|
||||
|
||||
async removeFunctionBreakpoint(name: string, signal?: AbortSignal, timeoutMs: number = 30_000) {
|
||||
const session = this.#touchActiveSession();
|
||||
const current = session.functionBreakpoints.filter(entry => entry.name !== name);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setFunctionBreakpoints",
|
||||
{
|
||||
breakpoints: current.map<DapFunctionBreakpoint>(entry => ({
|
||||
name: entry.name,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
async () => {
|
||||
const current = session.functionBreakpoints.filter(entry => entry.name !== name);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setFunctionBreakpoints",
|
||||
{
|
||||
breakpoints: current.map<DapFunctionBreakpoint>(entry => ({
|
||||
name: entry.name,
|
||||
...(entry.condition ? { condition: entry.condition } : {}),
|
||||
})),
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.functionBreakpoints = this.#mapFunctionBreakpoints(current, response?.breakpoints);
|
||||
return { snapshot: buildSummary(session), breakpoints: session.functionBreakpoints };
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.functionBreakpoints = this.#mapFunctionBreakpoints(current, response?.breakpoints);
|
||||
return { snapshot: buildSummary(session), breakpoints: session.functionBreakpoints };
|
||||
}
|
||||
|
||||
async setInstructionBreakpoint(
|
||||
@@ -480,31 +551,37 @@ export class DapSessionManager {
|
||||
timeoutMs: number = 30_000,
|
||||
) {
|
||||
const session = this.#touchActiveSession();
|
||||
const current = session.instructionBreakpoints.filter(
|
||||
entry => entry.instructionReference !== instructionReference || entry.offset !== offset,
|
||||
);
|
||||
current.push({ instructionReference, offset, condition, hitCondition });
|
||||
current.sort((left, right) => {
|
||||
const referenceOrder = left.instructionReference.localeCompare(right.instructionReference);
|
||||
if (referenceOrder !== 0) {
|
||||
return referenceOrder;
|
||||
}
|
||||
return (left.offset ?? 0) - (right.offset ?? 0);
|
||||
});
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setInstructionBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetInstructionBreakpointsArguments,
|
||||
async () => {
|
||||
const current = session.instructionBreakpoints.filter(
|
||||
entry => entry.instructionReference !== instructionReference || entry.offset !== offset,
|
||||
);
|
||||
current.push({ instructionReference, offset, condition, hitCondition });
|
||||
current.sort((left, right) => {
|
||||
const referenceOrder = left.instructionReference.localeCompare(right.instructionReference);
|
||||
if (referenceOrder !== 0) {
|
||||
return referenceOrder;
|
||||
}
|
||||
return (left.offset ?? 0) - (right.offset ?? 0);
|
||||
});
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setInstructionBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetInstructionBreakpointsArguments,
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.instructionBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapInstructionBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.instructionBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapInstructionBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
}
|
||||
|
||||
async removeInstructionBreakpoint(
|
||||
@@ -514,29 +591,35 @@ export class DapSessionManager {
|
||||
timeoutMs: number = 30_000,
|
||||
) {
|
||||
const session = this.#touchActiveSession();
|
||||
const current = session.instructionBreakpoints.filter(entry => {
|
||||
if (entry.instructionReference !== instructionReference) {
|
||||
return true;
|
||||
}
|
||||
if (offset === undefined) {
|
||||
return false;
|
||||
}
|
||||
return entry.offset !== offset;
|
||||
});
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setInstructionBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetInstructionBreakpointsArguments,
|
||||
async () => {
|
||||
const current = session.instructionBreakpoints.filter(entry => {
|
||||
if (entry.instructionReference !== instructionReference) {
|
||||
return true;
|
||||
}
|
||||
if (offset === undefined) {
|
||||
return false;
|
||||
}
|
||||
return entry.offset !== offset;
|
||||
});
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setInstructionBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetInstructionBreakpointsArguments,
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.instructionBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapInstructionBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.instructionBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapInstructionBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
}
|
||||
|
||||
async dataBreakpointInfo(
|
||||
@@ -570,42 +653,54 @@ export class DapSessionManager {
|
||||
timeoutMs: number = 30_000,
|
||||
) {
|
||||
const session = this.#touchActiveSession();
|
||||
const current = session.dataBreakpoints.filter(entry => entry.dataId !== dataId);
|
||||
current.push({ dataId, accessType, condition, hitCondition });
|
||||
current.sort((left, right) => left.dataId.localeCompare(right.dataId));
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setDataBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetDataBreakpointsArguments,
|
||||
async () => {
|
||||
const current = session.dataBreakpoints.filter(entry => entry.dataId !== dataId);
|
||||
current.push({ dataId, accessType, condition, hitCondition });
|
||||
current.sort((left, right) => left.dataId.localeCompare(right.dataId));
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setDataBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetDataBreakpointsArguments,
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.dataBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapDataBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.dataBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapDataBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
}
|
||||
|
||||
async removeDataBreakpoint(dataId: string, signal?: AbortSignal, timeoutMs: number = 30_000) {
|
||||
const session = this.#touchActiveSession();
|
||||
const current = session.dataBreakpoints.filter(entry => entry.dataId !== dataId);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
return this.#serializeBreakpointMutation(
|
||||
session,
|
||||
"setDataBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetDataBreakpointsArguments,
|
||||
async () => {
|
||||
const current = session.dataBreakpoints.filter(entry => entry.dataId !== dataId);
|
||||
const response = await this.#sendRequestWithConfig<{ breakpoints?: DapBreakpoint[] }>(
|
||||
session,
|
||||
"setDataBreakpoints",
|
||||
{
|
||||
breakpoints: current,
|
||||
} satisfies DapSetDataBreakpointsArguments,
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.dataBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapDataBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
},
|
||||
signal,
|
||||
timeoutMs,
|
||||
);
|
||||
session.dataBreakpoints = current;
|
||||
return {
|
||||
snapshot: buildSummary(session),
|
||||
breakpoints: this.#mapDataBreakpoints(current, response?.breakpoints),
|
||||
};
|
||||
}
|
||||
|
||||
async disassemble(
|
||||
@@ -756,21 +851,25 @@ export class DapSessionManager {
|
||||
|
||||
async pause(signal?: AbortSignal, timeoutMs: number = 30_000): Promise<DapSessionSummary> {
|
||||
const session = this.#touchActiveSession();
|
||||
if (session.status === "stopped") {
|
||||
// status is mutated by the event reader between awaits; check through a
|
||||
// closure so TS does not carry stale narrowing from the early return.
|
||||
const isStopped = () => session.status === "stopped";
|
||||
if (isStopped()) {
|
||||
return buildSummary(session);
|
||||
}
|
||||
const threadId = await this.#resolveThreadId(session, signal, timeoutMs);
|
||||
// Subscribe BEFORE sending pause: the stopped event can arrive in the
|
||||
// same chunk as the response and would otherwise be dispatched before
|
||||
// the waiter subscribes, burning the whole timeout.
|
||||
const stoppedPromise = session.client.waitForEvent<DapStoppedEventBody>("stopped", undefined, signal, timeoutMs);
|
||||
stoppedPromise.catch(() => {});
|
||||
await this.#sendRequestWithConfig(session, "pause", { threadId } satisfies DapPauseArguments, signal, timeoutMs);
|
||||
// The stopped event may already have been processed by #handleStoppedEvent
|
||||
// between the request and here. Wait for it, but tolerate timeout if the
|
||||
// session already transitioned.
|
||||
try {
|
||||
await untilAborted(
|
||||
signal,
|
||||
session.client.waitForEvent<DapStoppedEventBody>("stopped", undefined, signal, timeoutMs),
|
||||
);
|
||||
} catch {
|
||||
// Timeout or abort — report current state regardless
|
||||
if (!isStopped()) {
|
||||
try {
|
||||
await untilAborted(signal, stoppedPromise);
|
||||
} catch {
|
||||
// Timeout or abort — report current state regardless
|
||||
}
|
||||
}
|
||||
return buildSummary(session);
|
||||
}
|
||||
@@ -884,16 +983,16 @@ export class DapSessionManager {
|
||||
|
||||
getOutput(limitBytes?: number): DapOutputSnapshot {
|
||||
const session = this.#touchActiveSession();
|
||||
if (!limitBytes || limitBytes <= 0 || Buffer.byteLength(session.output, "utf-8") <= limitBytes) {
|
||||
return { snapshot: buildSummary(session), output: session.output };
|
||||
const output = session.outputChunks.join("");
|
||||
if (!limitBytes || limitBytes <= 0 || session.outputBufferedBytes <= limitBytes) {
|
||||
return { snapshot: buildSummary(session), output };
|
||||
}
|
||||
let sliceStart = session.output.length;
|
||||
let remaining = limitBytes;
|
||||
while (sliceStart > 0 && remaining > 0) {
|
||||
sliceStart -= 1;
|
||||
remaining -= Buffer.byteLength(session.output[sliceStart] ?? "", "utf-8");
|
||||
// Byte-slice the tail once; a torn code point at the cut decodes as U+FFFD.
|
||||
const buffer = Buffer.from(output, "utf-8");
|
||||
if (buffer.length <= limitBytes) {
|
||||
return { snapshot: buildSummary(session), output };
|
||||
}
|
||||
return { snapshot: buildSummary(session), output: session.output.slice(sliceStart) };
|
||||
return { snapshot: buildSummary(session), output: buffer.subarray(buffer.length - limitBytes).toString("utf-8") };
|
||||
}
|
||||
|
||||
async terminate(signal?: AbortSignal, timeoutMs: number = 30_000): Promise<DapSessionSummary | null> {
|
||||
@@ -973,8 +1072,10 @@ export class DapSessionManager {
|
||||
functionBreakpoints: [],
|
||||
instructionBreakpoints: [],
|
||||
dataBreakpoints: [],
|
||||
output: "",
|
||||
breakpointMutationQueue: Promise.resolve(),
|
||||
outputChunks: [],
|
||||
outputBytes: 0,
|
||||
outputBufferedBytes: 0,
|
||||
outputTruncated: false,
|
||||
stop: {},
|
||||
threads: [],
|
||||
|
||||
@@ -36,7 +36,6 @@ export interface TerminalStateInfo {
|
||||
hyperlinks: boolean;
|
||||
deccara: boolean;
|
||||
screenToScrollback: boolean;
|
||||
eagerEraseScrollbackRisk: boolean;
|
||||
synchronizedOutput: boolean;
|
||||
multiplexer: string | null;
|
||||
env: { TERM?: string; TERM_PROGRAM?: string; TERM_PROGRAM_VERSION?: string; COLORTERM?: string };
|
||||
@@ -82,7 +81,6 @@ export function collectTerminalState(runtime: TerminalRuntimeState): TerminalSta
|
||||
hyperlinks: TERMINAL.hyperlinks,
|
||||
deccara: TERMINAL.deccara,
|
||||
screenToScrollback: TERMINAL.supportsScreenToScrollback,
|
||||
eagerEraseScrollbackRisk: TERMINAL.eagerEraseScrollbackRisk,
|
||||
synchronizedOutput: runtime.synchronizedOutput,
|
||||
multiplexer: detectMultiplexer(env),
|
||||
env: {
|
||||
@@ -115,7 +113,6 @@ export function formatTerminalState(info: TerminalStateInfo): string {
|
||||
"",
|
||||
"Scrollback",
|
||||
` Screen->history clear: ${info.screenToScrollback ? "CSI 22 J" : "CSI 2 J (redraw)"}`,
|
||||
` Eager-erase risk: ${yesNo(info.eagerEraseScrollbackRisk)} (ED3 may yank scrolled readers)`,
|
||||
"",
|
||||
"Detection signals",
|
||||
` TERM: ${info.env.TERM ?? "(unset)"}`,
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user