Merge remote-tracking branch 'upstream/main' into feat/secret-friendly-names

This commit is contained in:
Mathews-Tom
2026-06-19 21:26:25 +05:30
174 changed files with 4757 additions and 1207 deletions
+20 -18
View File
@@ -23,10 +23,10 @@
a-glapinski
ak4153
apoc
AsafMah
asafmah
azais-corentin
basedcorp99
BayLee4
baylee4
bjin
cagedbird043
cexll
@@ -36,16 +36,17 @@ ckumar1
daandden
daaximus
danzaio
DarkPhilosophy
DeprecatedLuke
darkphilosophy
deprecatedluke
djdembeck
dmarsh-gusto
elikoga
enieuwy
ForeverYoungPp
GratefulDave
H4vC
HabibPro1999
flare576
foreveryoungpp
gratefuldave
h4vc
habibpro1999
handlecusion
haosenwang1018
hezhiyang2000
@@ -55,34 +56,34 @@ itertea
jchristman
jiwangyihao
kamafozilov
KamijoToma
Kukkerem
kamijotoma
kukkerem
ldx
lederniermagicien
loftiskg
lyc-aon
makoMakoGo
makomakogo
masonc15
maximhar
maxvisionai
metaphorics
MikeeI
Mokto
mikeei
mokto
mouyase
muness
nnk97
ogrodev
oldschoola
Parsifa1
parsifa1
phanthh
pidevxplay
qfrtt
ravshansbox
rburketaylor
RensTillmann
renstillmann
riverpilot
romanalexander
RzNmKX
rznmkx
scarthread
segmentationf4u1t
shoucandanghehe
@@ -98,10 +99,11 @@ tsagi2045
turbomolli
usr-bin-roygbiv
vmcall
VoidChecksum
voidchecksum
voiys
watzon
WodenJay
wodenjay
wolfiesch
zakhar-kogan
zamorakpds
korri123
+6
View File
@@ -3,6 +3,12 @@ name: CI
on:
push:
branches: [main]
# Vouch bookkeeping commits (mitchellh/vouch writes VOUCHED.td back to
# main on !vouch/!denounce/!unvouch) only edit the vouch list and need no
# build. Skip CI when a main push changes nothing but the vouch file; a
# push that also touches anything else still runs the full matrix.
paths-ignore:
- .github/VOUCHED.td
pull_request:
branches: [main]
workflow_dispatch:
Generated
+6 -6
View File
@@ -2314,7 +2314,7 @@ dependencies = [
[[package]]
name = "pi-ast"
version = "16.1.2"
version = "16.1.3"
dependencies = [
"anyhow",
"ast-grep-core",
@@ -2384,7 +2384,7 @@ dependencies = [
[[package]]
name = "pi-iso"
version = "16.1.2"
version = "16.1.3"
dependencies = [
"async-trait",
"libc",
@@ -2396,7 +2396,7 @@ dependencies = [
[[package]]
name = "pi-natives"
version = "16.1.2"
version = "16.1.3"
dependencies = [
"anyhow",
"arboard",
@@ -2444,7 +2444,7 @@ dependencies = [
[[package]]
name = "pi-shell"
version = "16.1.2"
version = "16.1.3"
dependencies = [
"anyhow",
"brush-builtins",
@@ -4270,9 +4270,9 @@ dependencies = [
[[package]]
name = "wayland-protocols"
version = "0.32.12"
version = "0.32.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "563a85523cade2429938e790815fd7319062103b9f4a2dc806e9b53b95982d8f"
checksum = "23d0c813de3daa2ed6520af85a3bd49b0e722a3078506899aa9686fea58dc4b6"
dependencies = [
"bitflags 2.13.0",
"wayland-backend",
+1 -1
View File
@@ -4,7 +4,7 @@ exclude = ["crates/brush-core-vendored", "crates/brush-builtins-vendored"]
resolver = "3"
[workspace.package]
version = "16.1.2"
version = "16.1.3"
edition = "2024"
license = "MIT"
authors = ["Can Boluk"]
+26 -26
View File
@@ -21,7 +21,7 @@
},
"packages/agent": {
"name": "@oh-my-pi/pi-agent-core",
"version": "16.1.2",
"version": "16.1.3",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -39,7 +39,7 @@
},
"packages/ai": {
"name": "@oh-my-pi/pi-ai",
"version": "16.1.2",
"version": "16.1.3",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-catalog": "catalog:",
@@ -55,7 +55,7 @@
},
"packages/catalog": {
"name": "@oh-my-pi/pi-catalog",
"version": "16.1.2",
"version": "16.1.3",
"dependencies": {
"@bufbuild/protobuf": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -69,7 +69,7 @@
},
"packages/coding-agent": {
"name": "@oh-my-pi/pi-coding-agent",
"version": "16.1.2",
"version": "16.1.3",
"bin": {
"omp": "src/cli.ts",
},
@@ -137,7 +137,7 @@
},
"packages/hashline": {
"name": "@oh-my-pi/hashline",
"version": "16.1.2",
"version": "16.1.3",
"dependencies": {
"diff": "catalog:",
"lru-cache": "catalog:",
@@ -148,7 +148,7 @@
},
"packages/mnemopi": {
"name": "@oh-my-pi/pi-mnemopi",
"version": "16.1.2",
"version": "16.1.3",
"bin": {
"mnemopi": "src/cli.ts",
},
@@ -165,7 +165,7 @@
},
"peerDependencies": {
"fastembed": "2.1.0",
"onnxruntime-node": "1.26.0",
"onnxruntime-node": "1.21.0",
},
"optionalPeers": [
"fastembed",
@@ -174,7 +174,7 @@
},
"packages/natives": {
"name": "@oh-my-pi/pi-natives",
"version": "16.1.2",
"version": "16.1.3",
"devDependencies": {
"@napi-rs/cli": "catalog:",
"@types/bun": "catalog:",
@@ -182,7 +182,7 @@
},
"packages/snapcompact": {
"name": "@oh-my-pi/snapcompact",
"version": "16.1.2",
"version": "16.1.3",
"dependencies": {
"@oh-my-pi/pi-ai": "catalog:",
"@oh-my-pi/pi-natives": "catalog:",
@@ -195,7 +195,7 @@
},
"packages/stats": {
"name": "@oh-my-pi/omp-stats",
"version": "16.1.2",
"version": "16.1.3",
"bin": {
"omp-stats": "./src/index.ts",
},
@@ -221,7 +221,7 @@
},
"packages/swarm-extension": {
"name": "@oh-my-pi/swarm-extension",
"version": "16.1.2",
"version": "16.1.3",
"bin": {
"omp-swarm": "src/cli.ts",
},
@@ -237,7 +237,7 @@
},
"packages/tui": {
"name": "@oh-my-pi/pi-tui",
"version": "16.1.2",
"version": "16.1.3",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"@oh-my-pi/pi-utils": "catalog:",
@@ -278,7 +278,7 @@
},
"packages/utils": {
"name": "@oh-my-pi/pi-utils",
"version": "16.1.2",
"version": "16.1.3",
"dependencies": {
"@oh-my-pi/pi-natives": "catalog:",
"handlebars": "catalog:",
@@ -291,7 +291,7 @@
},
"packages/wire": {
"name": "@oh-my-pi/pi-wire",
"version": "16.1.2",
"version": "16.1.3",
"devDependencies": {
"@types/bun": "catalog:",
},
@@ -327,18 +327,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "16.1.2",
"@oh-my-pi/omp-stats": "16.1.2",
"@oh-my-pi/pi-agent-core": "16.1.2",
"@oh-my-pi/pi-ai": "16.1.2",
"@oh-my-pi/pi-catalog": "16.1.2",
"@oh-my-pi/pi-coding-agent": "16.1.2",
"@oh-my-pi/pi-mnemopi": "16.1.2",
"@oh-my-pi/pi-natives": "16.1.2",
"@oh-my-pi/pi-tui": "16.1.2",
"@oh-my-pi/pi-utils": "16.1.2",
"@oh-my-pi/pi-wire": "16.1.2",
"@oh-my-pi/snapcompact": "16.1.2",
"@oh-my-pi/hashline": "16.1.3",
"@oh-my-pi/omp-stats": "16.1.3",
"@oh-my-pi/pi-agent-core": "16.1.3",
"@oh-my-pi/pi-ai": "16.1.3",
"@oh-my-pi/pi-catalog": "16.1.3",
"@oh-my-pi/pi-coding-agent": "16.1.3",
"@oh-my-pi/pi-mnemopi": "16.1.3",
"@oh-my-pi/pi-natives": "16.1.3",
"@oh-my-pi/pi-tui": "16.1.3",
"@oh-my-pi/pi-utils": "16.1.3",
"@oh-my-pi/pi-wire": "16.1.3",
"@oh-my-pi/snapcompact": "16.1.3",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
+1 -1
View File
@@ -172,7 +172,7 @@ fn create_windows_napi_tokio_runtime() -> Option<tokio::runtime::Runtime> {
/// MUST stay in sync with `VERSION_SENTINEL_EXPORT` in
/// `packages/natives/native/index.js` (which derives the name from
/// `package.json#version`).
#[napi(js_name = "__piNativesV16_1_2")]
#[napi(js_name = "__piNativesV16_1_3")]
pub const fn pi_natives_version_sentinel() {}
/// Native module entry point: install crash diagnostics before any tool can
+1 -1
View File
@@ -22,7 +22,7 @@
| --- | --- | --- | --- |
| `pattern` | `string` | Yes | Regex pattern. `search.ts` rejects whitespace-only input but otherwise preserves the pattern verbatim (leading/trailing whitespace is meaningful in regexes). The native matcher enables multiline only when the pattern text contains a literal newline or the two-character sequence `\\n`. The native layer auto-escapes braces that cannot be valid repetition quantifiers, so patterns like `${platform}` stay searchable (see Notes). |
| `paths` | `string \| string[]` | No | One file path, directory path, glob-like path, archive member, internal URL, or an array of those. Omitted or empty defaults to `.` (the workspace root). Append a line-range selector such as `:50-100` or `:5-16,960-973` to a single file/archive/internal-resource input to constrain matches. Empty strings are rejected after trimming/quote stripping. Single entries accidentally joined with comma, semicolon, or whitespace are expanded only after existence validation; existing paths containing delimiters stay intact. Filesystem-backed internal URLs search their backing file; virtual internal resources search resolved text in memory. Internal URLs cannot contain glob characters. |
| `i` | `boolean` | No | Case-insensitive search. Defaults to `false`. Passed to native `ignoreCase` or JS `RegExp` flags for virtual resources. |
| `case` | `boolean` | No | Case-sensitive search. Defaults to `true`. Passed to native `ignoreCase` or JS `RegExp` flags for virtual resources. |
| `gitignore` | `boolean` | No | Respect `.gitignore` during directory scans. Defaults to `true`. Passed to native `gitignore`. |
| `skip` | `number` | No | File-page offset for multi-file results. Defaults to `0`; `search.ts` floors finite numbers and rejects negative or non-finite values. Single-file searches ignore it because they do not paginate by file. |
+12 -12
View File
@@ -24,18 +24,18 @@
"@huggingface/transformers": "^4.2.0",
"@mozilla/readability": "^0.6.0",
"@napi-rs/cli": "3.7.0",
"@oh-my-pi/hashline": "16.1.2",
"@oh-my-pi/omp-stats": "16.1.2",
"@oh-my-pi/pi-agent-core": "16.1.2",
"@oh-my-pi/pi-ai": "16.1.2",
"@oh-my-pi/pi-catalog": "16.1.2",
"@oh-my-pi/pi-coding-agent": "16.1.2",
"@oh-my-pi/pi-mnemopi": "16.1.2",
"@oh-my-pi/pi-natives": "16.1.2",
"@oh-my-pi/pi-tui": "16.1.2",
"@oh-my-pi/pi-utils": "16.1.2",
"@oh-my-pi/pi-wire": "16.1.2",
"@oh-my-pi/snapcompact": "16.1.2",
"@oh-my-pi/hashline": "16.1.3",
"@oh-my-pi/omp-stats": "16.1.3",
"@oh-my-pi/pi-agent-core": "16.1.3",
"@oh-my-pi/pi-ai": "16.1.3",
"@oh-my-pi/pi-catalog": "16.1.3",
"@oh-my-pi/pi-coding-agent": "16.1.3",
"@oh-my-pi/pi-mnemopi": "16.1.3",
"@oh-my-pi/pi-natives": "16.1.3",
"@oh-my-pi/pi-tui": "16.1.3",
"@oh-my-pi/pi-utils": "16.1.3",
"@oh-my-pi/pi-wire": "16.1.3",
"@oh-my-pi/snapcompact": "16.1.3",
"@opentelemetry/api": "^1.9.1",
"@opentelemetry/context-async-hooks": "^2.7.1",
"@opentelemetry/exporter-trace-otlp-proto": "^0.218.0",
+2
View File
@@ -6,6 +6,8 @@
### Fixed
- Prevented sensitive raw JSON payloads from leaking into agent events during tool validation
- Ensured tool validation errors are handled correctly for malformed JSON parse inputs
- Ensure deep-cloning of tool-call arguments respects own enumerable properties
- Prevent direct object references between agent message snapshots and streaming events
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-agent-core",
"version": "16.1.2",
"version": "16.1.3",
"description": "General-purpose agent with transport abstraction, state management, and attachment support",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+42 -16
View File
@@ -591,7 +591,7 @@ export function normalizeTools(
// specs without their descriptions (top-level + nested schema annotations)
// so they are not duplicated on the wire. Strip the STABLE wire schema (the
// memoized `stripSchemaDescriptions` result is reused across requests), then
// re-inject `_i` (without its hint, which `describeIntent: false` omits) so
// re-inject `i` (without its hint, which `describeIntent: false` omits) so
// intent tracing keeps the field while no descriptions ride the wire.
if (pruneDescriptions) {
let parameters = stripSchemaDescriptions(toolWireSchema(t)) as TSchema;
@@ -718,7 +718,7 @@ async function runLoopBody(
stepCounter: StepCounter,
streamFn?: StreamFn,
): Promise<void> {
let deadlineTimer: ReturnType<typeof setTimeout> | undefined;
let deadlineTimer: Timer | undefined;
if (config.deadline !== undefined) {
const deadlineAbortController = new AbortController();
const delay = config.deadline - Date.now();
@@ -1756,7 +1756,44 @@ async function executeToolCalls(
}
}
}
record.args = argsForExecution;
let effectiveArgs: Record<string, unknown>;
try {
if (!tool) throw new Error(`Tool ${toolCall.name} not found`);
effectiveArgs = validateToolArguments(tool, { ...toolCall, arguments: argsForExecution });
} catch (validationError) {
if (tool?.lenientArgValidation) {
effectiveArgs = { ...argsForExecution };
delete effectiveArgs.__parseError;
delete effectiveArgs.__rawJson;
} else {
if ("__parseError" in argsForExecution) {
record.args = {
__parseError: argsForExecution.__parseError,
};
} else {
record.args = argsForExecution;
}
emitToolResult(
record,
{
content: [
{
type: "text" as const,
text: validationError instanceof Error ? validationError.message : String(validationError),
},
],
details: {
isError: true,
error: validationError instanceof Error ? validationError.message : String(validationError),
},
},
true,
);
return;
}
}
record.args = effectiveArgs;
if (toolSignal.aborted) {
record.skipped = true;
recordSkippedTool(telemetry, {
@@ -1772,7 +1809,7 @@ async function executeToolCalls(
type: "tool_execution_start",
toolCallId: toolCall.id,
toolName: toolCall.name,
args: argsForExecution,
args: effectiveArgs,
intent: toolCall.intent,
});
@@ -1780,7 +1817,7 @@ async function executeToolCalls(
tool,
toolName: toolCall.name,
toolCallId: toolCall.id,
args: argsForExecution,
args: effectiveArgs,
parent: invokeAgentSpan,
});
if (toolSpan && toolCall.intent) {
@@ -1801,17 +1838,6 @@ async function executeToolCalls(
return;
}
let effectiveArgs: Record<string, unknown>;
try {
effectiveArgs = validateToolArguments(tool, { ...toolCall, arguments: argsForExecution });
} catch (validationError) {
if (tool.lenientArgValidation) {
effectiveArgs = argsForExecution;
} else {
throw validationError;
}
}
if (beforeToolCall) {
const beforeResult = await beforeToolCall(
{
+1 -1
View File
@@ -32,7 +32,7 @@ export interface StablePrefixSnapshot {
/** Options threaded through `build()` so the snapshot reflects loop-time settings. */
export interface BuildOptions {
/** Inject the `_i` intent field into tool schemas (must match agent-loop's normalizeTools). */
/** Inject the `i` intent field into tool schemas (must match agent-loop's normalizeTools). */
intentTracing: boolean;
exampleDialect?: Dialect;
/** Strip tool descriptions from the provider-bound specs (must match normalizeTools). */
+6 -6
View File
@@ -273,7 +273,7 @@ export interface AgentLoopConfig extends SimpleStreamOptions {
*
* When set, the loop reads messages from the append-only log (stable
* byte prefix) and caches system prompt + tools. Tools exclude per-turn
* `_i` intent fields.
* `i` intent fields.
*/
appendOnlyContext?: AppendOnlyContextManager;
@@ -584,11 +584,11 @@ export interface AgentTool<TParameters extends TSchema = TSchema, TDetails = any
*/
interruptible?: boolean;
/**
* Controls how the INTENT_FIELD (`_i`) is handled for this tool.
* - `"require"` (default): `_i` is injected and required in the parameter schema.
* - `"optional"`: `_i` is injected as an optional/nullable field.
* - `"omit"`: `_i` is NOT injected. Use for tools where intent is obvious (yield, resolve, todo, …).
* - function: `_i` is NOT injected; intent is derived dynamically from (potentially partial / streaming) args.
* Controls how the INTENT_FIELD (`i`) is handled for this tool.
* - `"require"` (default): `i` is injected and required in the parameter schema.
* - `"optional"`: `i` is injected as an optional/nullable field.
* - `"omit"`: `i` is NOT injected. Use for tools where intent is obvious (yield, resolve, todo, …).
* - function: `i` is NOT injected; intent is derived dynamically from (potentially partial / streaming) args.
*/
intent?: "omit" | "optional" | "require" | ((args: Partial<Static<TParameters>>) => string | undefined);
+62
View File
@@ -503,6 +503,68 @@ describe("agentLoop with AgentMessage", () => {
}
});
it("surfaces validation error for malformed JSON parse sentinels without leaking __rawJson", async () => {
const toolSchema = type({ value: "string" });
const tool: AgentTool<typeof toolSchema, { value: string }> = {
name: "echo",
label: "Echo",
description: "Echo tool",
parameters: toolSchema,
async execute() {
return { content: [] };
},
};
const context: AgentContext = { systemPrompt: [""], messages: [], tools: [tool] };
const rawJsonPayload = `{"i": Finding getAvailable definition, "value": "hello"}${"A".repeat(1000)}`;
const mock = createMockModel({
responses: [
{
content: [
{
type: "toolCall",
id: "tool-1",
name: "echo",
arguments: {
__parseError: "Unexpected token F in JSON at position 6",
__rawJson: rawJsonPayload,
},
},
],
},
{ content: ["done"] },
],
});
const config: AgentLoopConfig = { model: mock.model, convertToLlm: identityConverter };
const events: AgentEvent[] = [];
const stream = agentLoop([createUserMessage("run echo")], context, config, undefined, mock.stream);
for await (const event of stream) {
events.push(event);
}
// Validation should have failed and reported the parse error & truncated JSON
const toolResultMsg = events.find(e => e.type === "message_start" && e.message.role === "toolResult") as any;
expect(toolResultMsg).toBeDefined();
const resultText = toolResultMsg.message.content[0].text;
expect(resultText).toContain("Tool call arguments are not valid JSON.");
expect(resultText).toContain("Unexpected token F");
expect(resultText).toContain("[truncated");
// Should have paired start and end events
const toolStart = events.find(e => e.type === "tool_execution_start") as any;
const toolEnd = events.find(e => e.type === "tool_execution_end") as any;
expect(toolStart).toBeDefined();
expect(toolEnd).toBeDefined();
// Start args must not include __rawJson
expect(toolStart.args).toBeDefined();
expect(toolStart.args.__rawJson).toBeUndefined();
expect(toolStart.args.__parseError).toBeDefined(); // keeps __parseError for visibility of parse failure
});
it("injects and strips intent when intent tracing is enabled", async () => {
const toolSchema = type({ value: "string" });
const executedParams: Record<string, unknown>[] = [];
@@ -592,7 +592,7 @@ describe("message sync", () => {
// ---------------------------------------------------------------------------
describe("intent injection through build()", () => {
it("injects required `_i` into tool schemas when intentTracing is true", () => {
it("injects required `i` into tool schemas when intentTracing is true", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("read", "Read", {
type: "object",
@@ -608,20 +608,20 @@ describe("intent injection through build()", () => {
expect(params!.required).toContain(INTENT_FIELD);
});
it("materializes ArkType params and keeps `_i` first in authored order", () => {
it("materializes ArkType params and keeps `i` first in authored order", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("write", "Write", type({ path: "string", content: "string" }));
const ctx = makeContext({ tools: [tool] });
const result = mgr.build(ctx, { intentTracing: true });
const params = result.tools?.[0]?.parameters as { properties?: Record<string, unknown>; required?: string[] };
// `_i` must lead; authored order (path before content) is preserved rather
// `i` must lead; authored order (path before content) is preserved rather
// than ArkType's alphabetized-by-hash order (content, path).
expect(Object.keys(params.properties ?? {})).toEqual([INTENT_FIELD, "path", "content"]);
expect(params.required).toContain(INTENT_FIELD);
});
it("omits `_i` when intentTracing is false", () => {
it("omits `i` when intentTracing is false", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("read", "Read", {
type: "object",
@@ -679,7 +679,7 @@ describe("tool examples injection through build()", () => {
expect(desc).toBe("Find files.");
});
it("injects the `_i` placeholder into examples when intentTracing is on", () => {
it("injects the `i` placeholder into examples when intentTracing is on", () => {
const mgr = new AppendOnlyContextManager();
const tool = makeTool("find", "Find files.", findParams, findExamples);
const ctx = makeContext({ tools: [tool] });
+18
View File
@@ -2,14 +2,32 @@
## [Unreleased]
## [16.1.3] - 2026-06-19
### Added
- Added regression test pinning that `openai-completions` emits a `thinking` block for `reasoning_content` deltas even when `delta.content` is explicitly JSON `null` (the DeepSeek-format dual-key pattern used by custom GLM/Qwen reasoning providers). See [#2996](https://github.com/can1357/oh-my-pi/issues/2996).
### Changed
- Improved the thinking loop guard to treat assistant text loops as retryable errors
- Refined text normalization logic to reduce false positives in the thinking loop detector
### Fixed
- Fixed Ollama chat requests sending image payloads to text-only models. Image blocks are now omitted and replaced with the standard non-vision placeholder for models without vision support, while vision-capable Ollama models continue to receive images. ([#3009](https://github.com/can1357/oh-my-pi/pull/3009) by [@serverinspector](https://github.com/serverinspector))
- Fixed `SqliteAuthCredentialStore.close()` leaking one-off prepared statements created by inline `this.#db.prepare()` calls in `#authCredentialsTableExists`, `#readAuthSchemaVersion`, `#inferAuthSchemaVersion`, `#migrateAuthSchemaV0ToV1`, `#backfillCredentialIdentityKeys`, and `updateAuthCredential`. Each statement is now wrapped in `try/finally` with `stmt.finalize()`, and the `close()` method finalizes `#insertUsageCostStmt` and `#listUsageCostsStmt` which were previously missed. This caused EBUSY on Windows when tests tried to delete temp dirs containing open SQLite handles.
## [16.1.2] - 2026-06-19
### Added
- Added improved JSON repair capabilities for Anthropic tool arguments
- Added authentication broker discovery to sync credentials between local SQLite and remote state
### Fixed
- Improved error feedback and transparency for malformed Anthropic tool call arguments
- Added automatic fallback for unsupported OpenAI reasoning effort levels
- Improved reliability when handling invalid reasoning parameter errors across OpenAI-compatible APIs
- Fixed OpenAI-compatible Chat Completions, Responses, and Azure Responses requests to retry once with the nearest provider-supported reasoning effort when an endpoint rejects `xhigh`/`minimal`-style effort values.
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-ai",
"version": "16.1.2",
"version": "16.1.3",
"description": "Unified LLM API with automatic model discovery and provider configuration",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+69 -27
View File
@@ -4742,25 +4742,47 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
}
#authCredentialsTableExists(): boolean {
const row = this.#db
.prepare("SELECT 1 AS present FROM sqlite_master WHERE type = 'table' AND name = 'auth_credentials'")
.get() as { present?: number } | undefined;
return row?.present === 1;
const stmt = this.#db.prepare(
"SELECT 1 AS present FROM sqlite_master WHERE type = 'table' AND name = 'auth_credentials'",
);
try {
const row = stmt.get() as { present?: number } | undefined;
return row?.present === 1;
} finally {
stmt.finalize();
}
}
#readAuthSchemaVersion(): number | null {
const row = this.#db.prepare("SELECT version FROM auth_schema_version WHERE id = 1").get() as
| { version?: number }
| undefined;
return typeof row?.version === "number" ? row.version : null;
const stmt = this.#db.prepare("SELECT version FROM auth_schema_version WHERE id = 1");
try {
const row = stmt.get() as { version?: number } | undefined;
return typeof row?.version === "number" ? row.version : null;
} finally {
stmt.finalize();
}
}
#writeAuthSchemaVersion(version: number): void {
this.#db.prepare("INSERT OR REPLACE INTO auth_schema_version(id, version) VALUES (1, ?)").run(version);
const stmt = this.#db.prepare("INSERT OR REPLACE INTO auth_schema_version(id, version) VALUES (1, ?)");
try {
stmt.run(version);
} finally {
stmt.finalize();
}
}
#inferAuthSchemaVersion(): number {
const cols = this.#db.prepare("PRAGMA table_info(auth_credentials)").all() as Array<{ name?: string }>;
const stmt = this.#db.prepare("PRAGMA table_info(auth_credentials)");
try {
const cols = stmt.all() as Array<{ name?: string }>;
return this.#inferAuthSchemaVersionFromColumns(cols);
} finally {
stmt.finalize();
}
}
#inferAuthSchemaVersionFromColumns(cols: Array<{ name?: string }>): number {
const hasDisabledCause = cols.some(column => column.name === "disabled_cause");
const hasIdentityKey = cols.some(column => column.name === "identity_key");
const hasAccountId = cols.some(column => column.name === "account_id");
@@ -4808,8 +4830,14 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
#migrateAuthSchemaV0ToV1(): void {
const migrate = this.#db.transaction(() => {
const v0Cols = this.#db.prepare("PRAGMA table_info(auth_credentials)").all() as Array<{ name?: string }>;
const hasDisabled = v0Cols.some(col => col.name === "disabled");
const stmt = this.#db.prepare("PRAGMA table_info(auth_credentials)");
let hasDisabled = false;
try {
const v0Cols = stmt.all() as Array<{ name?: string }>;
hasDisabled = v0Cols.some(col => col.name === "disabled");
} finally {
stmt.finalize();
}
this.#db.run("ALTER TABLE auth_credentials RENAME TO auth_credentials_v0");
this.#db.run(`
@@ -4885,21 +4913,29 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
}
#backfillCredentialIdentityKeys(): void {
const rows = this.#db
.prepare(
"SELECT id, provider, credential_type, data, disabled_cause, identity_key FROM auth_credentials WHERE identity_key IS NULL ORDER BY id ASC",
)
.all() as AuthRow[];
const selectRowsStmt = this.#db.prepare(
"SELECT id, provider, credential_type, data, disabled_cause, identity_key FROM auth_credentials WHERE identity_key IS NULL ORDER BY id ASC",
);
let rows: AuthRow[];
try {
rows = selectRowsStmt.all() as AuthRow[];
} finally {
selectRowsStmt.finalize();
}
if (rows.length === 0) return;
let updateIdentity: Statement | null = null;
for (const row of rows) {
const identityKey = resolveRowCredentialIdentityKey(row.provider, row);
// Rows whose identity cannot be derived stay NULL; writing NULL over
// NULL would just burn a write transaction on every boot.
if (identityKey === null) continue;
updateIdentity ??= this.#db.prepare("UPDATE auth_credentials SET identity_key = ? WHERE id = ?");
updateIdentity.run(identityKey, row.id);
try {
for (const row of rows) {
const identityKey = resolveRowCredentialIdentityKey(row.provider, row);
// Rows whose identity cannot be derived stay NULL; writing NULL over
// NULL would just burn a write transaction on every boot.
if (identityKey === null) continue;
updateIdentity ??= this.#db.prepare("UPDATE auth_credentials SET identity_key = ? WHERE id = ?");
updateIdentity.run(identityKey, row.id);
}
} finally {
updateIdentity?.finalize();
}
}
@@ -5063,9 +5099,13 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
updateAuthCredential(id: number, credential: AuthCredential): void {
try {
const providerRow = this.#db.prepare("SELECT provider FROM auth_credentials WHERE id = ?").get(id) as
| { provider?: string }
| undefined;
const providerStmt = this.#db.prepare("SELECT provider FROM auth_credentials WHERE id = ?");
let providerRow: { provider?: string } | undefined;
try {
providerRow = providerStmt.get(id) as { provider?: string } | undefined;
} finally {
providerStmt.finalize();
}
const provider = providerRow?.provider ?? "";
const serialized = serializeCredential(provider, credential);
if (!serialized) return;
@@ -5334,6 +5374,8 @@ export class SqliteAuthCredentialStore implements AuthCredentialStore {
this.#lastUsageHistoryStmt.finalize();
this.#listUsageHistoryStmt.finalize();
this.#updateUsageHistoryStmt.finalize();
this.#insertUsageCostStmt.finalize();
this.#listUsageCostsStmt.finalize();
this.#db.close();
}
}
+1 -1
View File
@@ -9,7 +9,7 @@ export function renderToolExamples(tool: InbandTool, dialect: Dialect, intentFie
if (!examples?.length) return "";
const definition = getDialectDefinition(dialect);
const renderCall = (args: Record<string, unknown>): string => {
// When intent tracing injects `_i` into the schema, examples must show a
// When intent tracing injects `i` into the schema, examples must show a
// placeholder so the model learns to emit it. Keep it first, matching the
// schema injection order.
const finalArgs = intentField ? { [intentField]: INTENT_PLACEHOLDER, ...args } : args;
+1 -1
View File
@@ -23,7 +23,7 @@ Call with a verbatim body — everything between `«` and `»` is taken literall
Argument values:
- Strings are written bare and verbatim (`path=src/a.ts`). Quote with `"…"` only when the value contains spaces or starts with `"`, `[`, or `{` (`_i="run the tests"`).
- Strings are written bare and verbatim (`path=src/a.ts`). Quote with `"…"` only when the value contains spaces or starts with `"`, `[`, or `{` (`i="run the tests"`).
- Numbers, booleans, and `null` are JSON literals (`offset=50`, `force=true`).
- Arrays and objects are inline JSON (`paths=["src","test"]`).
- The body fence holds the call's first long/multi-line string parameter; its key is implied, never written.
+14 -3
View File
@@ -50,7 +50,7 @@ import { AssistantMessageEventStream } from "../utils/event-stream";
import { isFoundryEnabled } from "../utils/foundry";
import { finalizeErrorMessage, type RawHttpRequestDump, rewriteCopilotError } from "../utils/http-inspector";
import { getStreamFirstEventTimeoutMs, getStreamIdleTimeoutMs, iterateWithIdleTimeout } from "../utils/idle-iterator";
import { parseStreamingJsonThrottled } from "../utils/json-parse";
import { parseJsonWithRepair, parseStreamingJsonThrottled } from "../utils/json-parse";
import { notifyProviderResponse } from "../utils/provider-response";
import { isCopilotTransientModelError } from "../utils/retry";
import { COMBINATOR_KEYS, NO_STRICT, toolWireSchema } from "../utils/schema";
@@ -1752,14 +1752,25 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
const finalJson =
block.partialJson.length > 0 ? block.partialJson : JSON.stringify(block.arguments ?? {});
try {
block.arguments = JSON.parse(finalJson) as ToolCall["arguments"];
block.arguments = parseJsonWithRepair(finalJson) as ToolCall["arguments"];
} catch (parseError) {
// Non-fatal: keep the best-effort arguments recovered by the throttled streaming
// parser instead of failing the turn on malformed/truncated tool-argument JSON.
reportAnthropicEnvelopeAnomaly(
`tool_use ${block.id} arguments are not valid JSON: ${parseError instanceof Error ? parseError.message : String(parseError)}`,
);
block.arguments = (block.arguments ?? {}) as ToolCall["arguments"];
const recoveredKeys = Object.keys(block.arguments ?? {});
if (recoveredKeys.length === 0) {
const maxLen = 512;
const truncatedJson =
finalJson.length <= maxLen
? finalJson
: `${finalJson.slice(0, maxLen)}… [truncated ${finalJson.length - maxLen} chars]`;
block.arguments = {
__parseError: parseError instanceof Error ? parseError.message : String(parseError),
__rawJson: truncatedJson,
};
}
}
delete (block as { partialJson?: string }).partialJson;
delete (block as { lastParseLen?: number }).lastParseLen;
+17 -21
View File
@@ -5,15 +5,14 @@ import type {
Api,
AssistantMessage,
Context,
DeveloperMessage,
ImageContent,
Message,
Model,
StreamFunction,
StreamOptions,
TextContent,
Tool,
ToolChoice,
ToolResultMessage,
UserMessage,
} from "../types";
import { normalizeSystemPrompts } from "../utils";
import { AssistantMessageEventStream } from "../utils/event-stream";
@@ -32,6 +31,7 @@ import {
type StreamMarkupHealingEvent,
} from "../utils/stream-markup-healing";
import { transformMessages } from "./transform-messages";
import { joinTextWithImagePlaceholder, partitionVisionContent } from "./vision-guard";
/** Non-2xx response from the Ollama `/api/chat` endpoint. */
export class OllamaApiError extends ProviderHttpError {
@@ -174,40 +174,35 @@ function selectToolsForToolChoice(tools: Tool[] | undefined, toolChoice: ToolCho
return [];
}
function toPlainContent(content: string | Array<{ type: "text" | "image"; text?: string; data?: string }>): {
function toPlainContent(
content: string | ReadonlyArray<TextContent | ImageContent>,
supportsImages: boolean,
): {
content: string;
images?: string[];
} {
if (typeof content === "string") {
return { content };
}
const textParts: string[] = [];
const images: string[] = [];
for (const block of content) {
if (block.type === "text" && typeof block.text === "string") {
textParts.push(block.text);
}
if (block.type === "image" && typeof block.data === "string") {
images.push(block.data);
}
}
const { textBlocks, imageBlocks, omittedImages } = partitionVisionContent(content, supportsImages);
const text = textBlocks.map(block => block.text).join("\n");
return {
content: textParts.join("\n"),
...(images.length > 0 ? { images } : {}),
content: joinTextWithImagePlaceholder(text, omittedImages),
...(imageBlocks.length > 0 ? { images: imageBlocks.map(block => block.data) } : {}),
};
}
function convertMessage(message: Message): OllamaMessage {
function convertMessage(message: Message, supportsImages: boolean): OllamaMessage {
if (message.role === "user") {
const converted = toPlainContent(message.content as UserMessage["content"]);
const converted = toPlainContent(message.content, supportsImages);
return { role: "user", ...converted };
}
if (message.role === "developer") {
const converted = toPlainContent(message.content as DeveloperMessage["content"]);
const converted = toPlainContent(message.content, supportsImages);
return { role: "system", ...converted };
}
if (message.role === "toolResult") {
const converted = toPlainContent(message.content as ToolResultMessage["content"]);
const converted = toPlainContent(message.content, supportsImages);
return {
role: "tool",
tool_name: message.toolName,
@@ -259,8 +254,9 @@ function convertMessages(model: Model<"ollama-chat">, context: Context): OllamaM
}
messages.push(...context.messages);
const isCloud = model.provider === "ollama-cloud";
const supportsImages = model.input.includes("image");
return transformMessages(messages, model).map(msg => {
const converted = convertMessage(msg);
const converted = convertMessage(msg, supportsImages);
// Ollama cloud rejects requests when assistant history messages contain the `thinking`
// field — it's valid in model responses but not accepted as a history input. Strip it
// to prevent HTTP 400 errors. Local Ollama instances are unaffected.
+2 -2
View File
@@ -656,8 +656,8 @@ export interface Tool<TParameters extends TSchema = TSchema> {
* Illustrative calls/notes; the AI layer renders them into an `<examples>`
* block in the model's native tool-call syntax and appends to the wire
* description. Author `call`/`bad`/`good` as plain argument objects WITHOUT
* `_i` — when intent tracing injects `_i` into the schema, the renderer adds
* a placeholder `_i` automatically. Type each tool's `examples` against its
* `i` — when intent tracing injects `i` into the schema, the renderer adds
* a placeholder `i` automatically. Type each tool's `examples` against its
* own schema (e.g. `readonly ToolExample<typeof schema["type"]>[]`).
*/
examples?: readonly ToolExample[];
+18 -19
View File
@@ -25,11 +25,11 @@
* 13.5k non-loop thinking blocks (zero false positives; hardest negative
* scored 3 against the trigger of 4).
*
* Scope is deliberately narrow: **thinking only**. Answer text is left
* untouched so the guard can never discard already-streamed visible output. The
* guard is gated to Gemini models and wraps the provider stream, so it works
* across every Gemini transport (OpenRouter `openai-completions`, direct
* `google-generative-ai` / `google-gemini-cli`, Vertex). Disable with
* Scope is narrow: guarded Gemini/DeepSeek streams before any tool call. Native
* thinking is checked first; assistant text can also be checked for providers
* that surface reasoning as visible prose. On a hit, the failed turn is emitted
* as an empty retryable stream-stall error so the session drops and re-samples
* it instead of committing the runaway transcript. Disable with
* `PI_NO_THINKING_LOOP_GUARD=1`.
*/
import { logger } from "@oh-my-pi/pi-utils";
@@ -208,7 +208,6 @@ export function guardThinkingLoopStream(
void (async () => {
let thinkingArmed = true;
let textArmed = checkAssistantContent;
let accumulatedText = "";
try {
for await (const event of inner) {
let detail: string | null = null;
@@ -221,7 +220,6 @@ export function guardThinkingLoopStream(
thinkingArmed = false;
if (textArmed && event.type === "text_delta") {
detail = textDetector.push(event.delta);
accumulatedText += event.delta;
}
} else if (event.type === "toolcall_start" || event.type === "toolcall_delta") {
thinkingArmed = false;
@@ -244,7 +242,7 @@ export function guardThinkingLoopStream(
outer.push({
type: "error",
reason: "error",
error: buildThinkingLoopError(model, detail, accumulatedText),
error: buildThinkingLoopError(model, detail),
});
return;
}
@@ -289,13 +287,13 @@ export function withGeminiThinkingLoopGuard<
return guardThinkingLoopStream(dispatch(merged), model, controller, options);
}
function buildThinkingLoopError(model: Model<Api>, detail: string, accumulatedText?: string): AssistantMessage {
const hasText = Boolean(accumulatedText);
function buildThinkingLoopError(model: Model<Api>, detail: string): AssistantMessage {
return {
role: "assistant",
// Empty content is load-bearing: a contentful error stop is replay-unsafe
// and would NOT be auto-retried by the session.
content: hasText ? [{ type: "text", text: accumulatedText! }] : [],
// Empty content is load-bearing: loop-guard output is replay garbage, even
// when it arrived as assistant text instead of native thinking. Keeping it
// would persist the failed attempt before AgentSession retries.
content: [],
api: model.api,
provider: model.provider,
model: model.id,
@@ -310,7 +308,7 @@ function buildThinkingLoopError(model: Model<Api>, detail: string, accumulatedTe
stopReason: "error",
// "stream stall" makes the transport/session retry classifiers treat this
// as a transient (retryable) failure with no bespoke rule.
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content (${detail}).${hasText ? " Non-retryable because output was already streamed." : " Treating as a stream stall and retrying."}`,
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content (${detail}). Treating as a stream stall and retrying.`,
timestamp: Date.now(),
};
}
@@ -345,14 +343,15 @@ function detectVerbatimRepetition(text: string): [unit: string, count: number] |
return null;
}
/** Lowercase, drop code spans / paths / digits, keep only letter words. */
/** Lowercase and tokenize prose plus code/path payloads, dropping pure numbers. */
function normalizeSegment(segment: string): string {
return segment
.toLowerCase()
.replace(/`[^`]*`/g, " ")
.replace(/\/[^\s`]+/g, " ")
.replace(/\d+/g, " ")
.replace(/[^a-z]+/g, " ")
.replace(/`([^`]*)`/g, " $1 ")
.replace(/[^a-z0-9]+/g, " ")
.split(/\s+/)
.filter(token => /[a-z]/.test(token))
.join(" ")
.trim();
}
+12
View File
@@ -1291,6 +1291,18 @@ function truncateArgsForError(value: unknown): unknown {
*/
export function validateToolArguments(tool: Tool, toolCall: ToolCall): ToolCall["arguments"] {
const originalArgs = toolCall.arguments;
if (originalArgs && typeof originalArgs === "object" && "__parseError" in originalArgs) {
const parseError = originalArgs.__parseError;
const rawJson = String(originalArgs.__rawJson ?? "");
const maxLen = 512;
const truncatedRawJson =
rawJson.length <= maxLen
? rawJson
: `${rawJson.slice(0, maxLen)}… [truncated ${rawJson.length - maxLen} chars]`;
throw new Error(
`Validation failed for tool "${toolCall.name}": Tool call arguments are not valid JSON.\nParse Error: ${parseError}\nRaw JSON:\n${truncatedRawJson}`,
);
}
const ctx = getValidationContext(tool);
const { json } = ctx;
@@ -260,6 +260,34 @@ function createMalformedToolUseEvents(): MockAnthropicEvent[] {
];
}
function createGenuinelyMalformedToolUseEvents(): MockAnthropicEvent[] {
return [
{
type: "message_start",
message: {
id: "msg_tool_broken",
usage: {
input_tokens: 12,
output_tokens: 0,
cache_read_input_tokens: 0,
cache_creation_input_tokens: 0,
},
},
},
{
type: "content_block_start",
index: 0,
content_block: { type: "tool_use", id: "tool_broken", name: "lookup_weather", input: {} },
},
{
type: "content_block_delta",
index: 0,
delta: { type: "input_json_delta", partial_json: '{"city": Par' },
},
{ type: "content_block_stop", index: 0 },
];
}
function createUnterminatedToolUseSplicedReconnectEvents(): MockAnthropicEvent[] {
return [
{
@@ -1021,6 +1049,32 @@ describe("anthropic stream envelope handling", () => {
expect("partialJson" in toolCall).toBe(false);
});
it("records __parseError and pre-truncated __rawJson when partialParse fails on malformed JSON", async () => {
let attempt = 0;
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => {
attempt += 1;
return createMockRequest(createGenuinelyMalformedToolUseEvents()) as never;
});
const stream = streamAnthropic(model, context, { apiKey: "sk-ant-test" });
const events: AssistantMessageEvent[] = [];
for await (const event of stream) {
events.push(event);
}
const result = await stream.result();
expect(attempt).toBe(1);
const toolCall = result.content[0];
expect(toolCall?.type).toBe("toolCall");
if (toolCall?.type !== "toolCall") {
throw new Error("Expected toolCall content");
}
expect(toolCall.arguments.__parseError).toBeDefined();
expect(toolCall.arguments.__rawJson).toBeDefined();
expect(toolCall.arguments.__rawJson).toContain('{"city": Par');
});
it("finalizes a tool call left open by a spliced reconnect instead of erroring", async () => {
let attempt = 0;
vi.spyOn(AnthropicMessages.prototype, "create").mockImplementation(() => {
+124
View File
@@ -0,0 +1,124 @@
/**
* Regression guard for the openai-completions streaming reasoning contract.
*
* Some OpenAI-compatible hosts (GLM, Qwen reasoning variants behind custom
* `openai-completions` providers) stream the DeepSeek-format dual-key pattern:
*
* {"delta":{"content":null,"reasoning_content":"..."}}
*
* where `content` is explicitly JSON `null` (not absent) while `reasoning_content`
* carries the thinking text. The provider must emit a `thinking` block for that
* text — the null `content` must not cause the reasoning delta to be dropped,
* and it must not be coerced into an empty text block either.
*
* This pins the behavior so a future change that introduces an `if (delta.content)`
* guard (or routes the reasoning path behind a content-presence check) is caught.
* See issue #2996 for the reported (non-reproducing) scenario this defends.
*/
import { describe, expect, it } from "bun:test";
import { streamOpenAICompletions } from "@oh-my-pi/pi-ai/providers/openai-completions";
import type { Context, FetchImpl, Model, ModelSpec } from "@oh-my-pi/pi-ai/types";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
function createSseResponse(events: unknown[]): Response {
const payload = `${events
.map(event => `data: ${typeof event === "string" ? event : JSON.stringify(event)}`)
.join("\n\n")}\n\n`;
return new Response(payload, {
status: 200,
headers: { "content-type": "text/event-stream" },
});
}
function createMockFetch(events: unknown[]): FetchImpl {
async function mockFetch(_input: string | URL | Request, _init?: RequestInit): Promise<Response> {
return createSseResponse(events);
}
return Object.assign(mockFetch, { preconnect: fetch.preconnect });
}
function baseContext(): Context {
return {
messages: [{ role: "user", content: "1+1=?", timestamp: Date.now() }],
};
}
/** A custom openai-completions provider model (e.g. yunwu/glm-5.2) with reasoning enabled. */
function customReasoningModel(id = "glm-5.2"): Model<"openai-completions"> {
const base = getBundledModel("openai", "gpt-4o-mini");
return buildModel({
...base,
api: "openai-completions",
provider: "yunwu",
baseUrl: "https://yunwu.ai/v1",
id,
reasoning: true,
compat: base.compatConfig,
} as ModelSpec<"openai-completions">);
}
function deltaChunk(model: Model<"openai-completions">, delta: Record<string, unknown>): unknown {
return {
id: "x",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta }],
};
}
describe("openai-completions keeps reasoning_content when delta.content is null", () => {
it("emits a thinking block for reasoning_content deltas paired with content:null", async () => {
const model = customReasoningModel();
const fetchMock = createMockFetch([
deltaChunk(model, { content: null, reasoning_content: "分析" }),
deltaChunk(model, { content: null, reasoning_content: "步骤" }),
deltaChunk(model, { content: "2", reasoning_content: null }),
{
id: "x",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
},
"[DONE]",
]);
const result = await streamOpenAICompletions(model, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
reasoning: "high",
}).result();
expect(result.content).toEqual([
{ type: "thinking", thinking: "分析步骤", thinkingSignature: "reasoning_content" },
{ type: "text", text: "2" },
]);
});
it("does not coerce a content:null delta into a text block when only reasoning_content is present", async () => {
const model = customReasoningModel();
const fetchMock = createMockFetch([
deltaChunk(model, { content: null, reasoning_content: "only thinking" }),
deltaChunk(model, { content: "answer", reasoning_content: null }),
{
id: "x",
object: "chat.completion.chunk",
created: 0,
model: model.id,
choices: [{ index: 0, delta: {}, finish_reason: "stop" }],
},
"[DONE]",
]);
const result = await streamOpenAICompletions(model, baseContext(), {
apiKey: "test-key",
fetch: fetchMock,
reasoning: "high",
}).result();
const textBlocks = result.content.filter(b => b.type === "text");
expect(textBlocks).toEqual([{ type: "text", text: "answer" }]);
});
});
@@ -1,17 +1,36 @@
import { describe, expect, it } from "bun:test";
import type { Context } from "@oh-my-pi/pi-ai";
import type { AssistantMessage, Context, ToolResultMessage, Usage } from "@oh-my-pi/pi-ai";
import { streamOllama } from "@oh-my-pi/pi-ai/providers/ollama";
import { NON_VISION_IMAGE_PLACEHOLDER } from "@oh-my-pi/pi-ai/providers/vision-guard";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
interface OllamaChatMessagePayload {
role?: unknown;
content?: unknown;
images?: unknown;
}
interface OllamaChatRequestPayload {
think?: unknown;
messages?: OllamaChatMessagePayload[];
}
function isOllamaChatRequestPayload(value: unknown): value is OllamaChatRequestPayload {
return value !== null && typeof value === "object";
if (value === null || typeof value !== "object") return false;
const payload = value as { messages?: unknown };
return payload.messages === undefined || Array.isArray(payload.messages);
}
function createReasoningOllamaModel() {
const emptyUsage: Usage = {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
function createReasoningOllamaModel(input: Array<"text" | "image"> = ["text"]) {
return buildModel({
id: "deepseek-v4-flash",
name: "DeepSeek V4 Flash",
@@ -19,7 +38,7 @@ function createReasoningOllamaModel() {
provider: "ollama-cloud",
baseUrl: "https://ollama.com",
reasoning: true,
input: ["text"],
input,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 1_000_000,
maxTokens: 8192,
@@ -51,4 +70,108 @@ describe("Ollama chat thinking controls", () => {
expect(payload?.think).toBe(false);
});
it("omits tool-result images for text-only Ollama chat models", async () => {
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (!isOllamaChatRequestPayload(parsed)) {
throw new Error("Expected Ollama payload object");
}
payload = parsed;
return new Response('{"message":{"content":"ok"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
status: 200,
});
};
const now = Date.now();
const assistantMessage: AssistantMessage = {
role: "assistant",
content: [{ type: "toolCall", id: "tool-1", name: "browser", arguments: { action: "screenshot" } }],
api: "ollama-chat",
provider: "ollama-cloud",
model: "deepseek-v4-flash",
usage: emptyUsage,
stopReason: "toolUse",
timestamp: now,
};
const toolResult: ToolResultMessage = {
role: "toolResult",
toolCallId: "tool-1",
toolName: "browser",
content: [
{ type: "text", text: "Screenshot captured" },
{ type: "image", data: "ZmFrZQ==", mimeType: "image/png" },
],
isError: false,
timestamp: now + 1,
};
const context: Context = {
messages: [{ role: "user", content: "Inspect the page", timestamp: now - 1 }, assistantMessage, toolResult],
};
await streamOllama(createReasoningOllamaModel(), context, {
apiKey: "test-key",
fetch: fetchMock,
}).result();
const toolMessage = payload?.messages?.find(message => message.role === "tool");
if (!toolMessage) {
throw new Error("Expected converted Ollama tool message");
}
expect("images" in toolMessage).toBe(false);
expect(toolMessage.content).toContain("Screenshot captured");
expect(toolMessage.content).toContain(NON_VISION_IMAGE_PLACEHOLDER);
});
it("keeps tool-result images for vision-capable Ollama chat models", async () => {
let payload: OllamaChatRequestPayload | undefined;
const fetchMock = async (_input: string | URL | Request, init?: RequestInit): Promise<Response> => {
const parsed: unknown = JSON.parse(String(init?.body));
if (!isOllamaChatRequestPayload(parsed)) {
throw new Error("Expected Ollama payload object");
}
payload = parsed;
return new Response('{"message":{"content":"ok"},"done":true,"prompt_eval_count":1,"eval_count":1}\n', {
status: 200,
});
};
const now = Date.now();
const assistantMessage: AssistantMessage = {
role: "assistant",
content: [{ type: "toolCall", id: "tool-1", name: "browser", arguments: { action: "screenshot" } }],
api: "ollama-chat",
provider: "ollama-cloud",
model: "deepseek-v4-flash",
usage: emptyUsage,
stopReason: "toolUse",
timestamp: now,
};
const toolResult: ToolResultMessage = {
role: "toolResult",
toolCallId: "tool-1",
toolName: "browser",
content: [
{ type: "text", text: "Screenshot captured" },
{ type: "image", data: "ZmFrZQ==", mimeType: "image/png" },
],
isError: false,
timestamp: now + 1,
};
const context: Context = {
messages: [{ role: "user", content: "Inspect the page", timestamp: now - 1 }, assistantMessage, toolResult],
};
await streamOllama(createReasoningOllamaModel(["text", "image"]), context, {
apiKey: "test-key",
fetch: fetchMock,
}).result();
const toolMessage = payload?.messages?.find(message => message.role === "tool");
if (!toolMessage) {
throw new Error("Expected converted Ollama tool message");
}
expect(toolMessage.images).toEqual(["ZmFrZQ=="]);
expect(toolMessage.content).toContain("Screenshot captured");
expect(toolMessage.content).not.toContain(NON_VISION_IMAGE_PLACEHOLDER);
});
});
@@ -65,8 +65,8 @@ describe("processResponsesStream: parallel function_call items", () => {
const emitted: EmittedEvent[] = [];
const stream = { push: (e: unknown) => emitted.push(e as EmittedEvent), end: () => {} } as never;
const argsA = JSON.stringify({ _i: "Reading test", path: "test.txt" });
const argsB = JSON.stringify({ _i: "Reading test", path: "test.md" });
const argsA = JSON.stringify({ i: "Reading test", path: "test.txt" });
const argsB = JSON.stringify({ i: "Reading test", path: "test.md" });
await processResponsesStream(
makeStream([
@@ -125,8 +125,8 @@ describe("processResponsesStream: parallel function_call items", () => {
expect(blockA?.type).toBe("toolCall");
expect(blockB?.type).toBe("toolCall");
if (blockA?.type !== "toolCall" || blockB?.type !== "toolCall") throw new Error("expected toolCalls");
expect(blockA.arguments).toEqual({ _i: "Reading test", path: "test.txt" });
expect(blockB.arguments).toEqual({ _i: "Reading test", path: "test.md" });
expect(blockA.arguments).toEqual({ i: "Reading test", path: "test.txt" });
expect(blockB.arguments).toEqual({ i: "Reading test", path: "test.md" });
const ends = emitted.filter(e => e.type === "toolcall_end") as Array<{
toolCall: { id: string; arguments: Record<string, unknown> };
@@ -134,8 +134,8 @@ describe("processResponsesStream: parallel function_call items", () => {
}>;
expect(ends).toHaveLength(2);
const byCallId = new Map(ends.map(e => [e.toolCall.id.split("|")[0], e]));
expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ _i: "Reading test", path: "test.txt" });
expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ _i: "Reading test", path: "test.md" });
expect(byCallId.get("call_a")?.toolCall.arguments).toEqual({ i: "Reading test", path: "test.txt" });
expect(byCallId.get("call_b")?.toolCall.arguments).toEqual({ i: "Reading test", path: "test.md" });
expect(byCallId.get("call_a")?.contentIndex).toBe(0);
expect(byCallId.get("call_b")?.contentIndex).toBe(1);
+2 -2
View File
@@ -1,4 +1,4 @@
import { afterEach, describe, expect, it, mock, spyOn } from "bun:test";
import { afterEach, describe, expect, it, type Mock, mock, spyOn } from "bun:test";
import { streamPiNative } from "@oh-my-pi/pi-ai/providers/pi-native-client";
import type {
AssistantMessage,
@@ -288,7 +288,7 @@ describe("streamPiNative event flow", () => {
await expect(stream.result()).rejects.toThrow(/pre-aborted/);
// fetch was never called — short-circuit happened in the abort guard
expect((fetchImpl as unknown as ReturnType<typeof spyOn>).mock.calls.length).toBe(0);
expect((fetchImpl as unknown as Mock<typeof globalThis.fetch>).mock.calls.length).toBe(0);
});
it("forwards the caller's AbortSignal to the underlying fetch", async () => {
@@ -72,7 +72,7 @@ function chunk(model: string, delta: SseChoiceDelta, finish: SseChunk["choices"]
const REPORTED_DSML_LEAK =
"<|DSML|tool_calls>\n" +
' <|DSML|invoke name="bash">\n' +
' <|DSML|parameter name="_i" string="true">Check Fedora 42 available packages</|DSML|parameter>\n' +
' <|DSML|parameter name="i" string="true">Check Fedora 42 available packages</|DSML|parameter>\n' +
' <|DSML|parameter name="command" string="true">docker run --rm --platform linux/arm64 fedora:42 bash -c \'type python3; type git; type sed; type cp; ls /usr/bin/python3 2>/dev/null; rpm -qa | grep -E "^python3|^git-|^sed-|^bash-" | sort\'</|DSML|parameter>\n' +
' <|DSML|parameter name="timeout" string="false">15</|DSML|parameter>\n' +
" </|DSML|invoke>\n" +
@@ -616,7 +616,7 @@ describe("Ollama provider DSML envelope healing", () => {
expect(toolCalls).toHaveLength(1);
expect(toolCalls[0].name).toBe("bash");
expect(toolCalls[0].arguments).toMatchObject({
_i: "Check Fedora 42 available packages",
[INTENT_FIELD]: "Check Fedora 42 available packages",
timeout: 15,
});
expect(String(toolCalls[0].arguments.command)).toContain("docker run");
+75 -7
View File
@@ -2,7 +2,7 @@ import { describe, expect, test } from "bun:test";
import { clearCustomApis } from "@oh-my-pi/pi-ai/api-registry";
import { createMockModel, type MockContent, registerMockApi } from "@oh-my-pi/pi-ai/providers/mock";
import { stream, streamSimple } from "@oh-my-pi/pi-ai/stream";
import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model, TextContent } from "@oh-my-pi/pi-ai/types";
import type { Api, AssistantMessage, AssistantMessageEvent, Context, Model } from "@oh-my-pi/pi-ai/types";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import {
isGeminiThinkingLoopModel,
@@ -120,6 +120,75 @@ describe("ThinkingLoopDetector", () => {
expect(detail).toBeNull();
});
test("does not collapse distinct per-file assignment templates into a loop", () => {
const detector = new ThinkingLoopDetector();
const text = [
`1. Subagent ApprovalModeTest:
- target: packages/coding-agent/test/tools/approval-mode.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/tools/approval-mode.test.ts
# Change
Replace the type annotation:
\`let session: Awaited<ReturnType<typeof createAgentSession>>["session"];\`
with \`AgentSession\`.
Verify where \`AgentSession\` is imported from.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
`2. Subagent GhTest:
- target: packages/coding-agent/test/tools/gh.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/tools/gh.test.ts
# Change
Replace the type annotations:
\`let tempHome: Awaited<ReturnType<typeof setupTempHome>>;\`
with \`TempDir\`.
Verify where \`TempDir\` is imported from.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
`3. Subagent TodoTest:
- target: packages/coding-agent/test/tools/todo.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/tools/todo.test.ts
# Change
Locate line 438 containing:
\`function innerLines(component: ReturnType<typeof todoToolRenderer.renderResult>): string[] {\`
Replace \`ReturnType<typeof todoToolRenderer.renderResult>\` with the explicit return type.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
`4. Subagent HookEditorTest:
- target: packages/coding-agent/test/hook-editor.test.ts
- role: "Test-file refactoring specialist"
- assignment:
\`\`\`markdown
# Target
packages/coding-agent/test/hook-editor.test.ts
# Change
Replace \`setFocus: ReturnType<typeof vi.fn>;\` and \`requestRender: ReturnType<typeof vi.fn>;\`
with Bun's explicit mock type from \`bun:test\`.
Run \`biome check --write --unsafe\` on the file.
\`\`\``,
].join("\n\n");
let detail: string | null = null;
for (let i = 0; i < text.length && !detail; i += 37) {
detail = detector.push(text.slice(i, i + 37));
}
detail ??= detector.flush();
expect(detail).toBeNull();
});
test("flush() catches a final unterminated duplicate paragraph", () => {
const detector = new ThinkingLoopDetector();
// Seven blank-line-separated dupes leave the eighth (cluster-completing)
@@ -326,13 +395,12 @@ describe("loop guard assistant prose/text loops", () => {
const result = await guarded.result();
expect(result.stopReason).toBe("error");
// Content must hold the text streamed BEFORE the loop detector tripped.
expect(result.content.length).toBe(1);
expect(result.content[0].type).toBe("text");
expect((result.content[0] as TextContent).text).toContain("First healthy text");
// Loop-guard output is replay garbage even when it came through text_delta:
// drop it so AgentSession can retry with a clean assistant turn.
expect(result.content).toEqual([]);
expect(result.errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
// Since some text was forwarded, it is replay-unsafe, so isRetryableError should return false.
expect(isRetryableError(new Error(result.errorMessage))).toBe(false);
expect(result.errorMessage).toContain("stream stall");
expect(isRetryableError(new Error(result.errorMessage))).toBe(true);
});
test("does not trip on assistant text loop when checkAssistantContent is false", async () => {
+1 -1
View File
@@ -186,6 +186,6 @@ describe("renderToolExamples", () => {
examples: [{ caption: "Find files", call: { paths: ["src/**/*.ts"] } }],
};
expect(renderToolExamples(tool, "anthropic")).not.toContain(INTENT_FIELD);
expect(renderToolExamples(tool, "anthropic")).not.toContain(`<parameter name="${INTENT_FIELD}"`);
});
});
+7
View File
@@ -2,6 +2,13 @@
## [Unreleased]
## [16.1.3] - 2026-06-19
### Fixed
- Marked Ollama Cloud catalog models to omit on-the-wire output-token caps, preventing context-window-sized `num_predict` values from causing HTTP 400s for models whose true output cap is not discoverable. ([#2984](https://github.com/can1357/oh-my-pi/issues/2984))
- Fixed `readModelCache`/`writeModelCache` using a process-global shared database even when a custom `dbPath` was provided. Custom-path cache operations now open and close a per-call database via `withModelCacheDb`, preventing leaked SQLite handles on Windows
## [16.1.2] - 2026-06-19
### Added
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-catalog",
"version": "16.1.2",
"version": "16.1.3",
"description": "Model catalog for omp: bundled model database, provider discovery descriptors, model identity, classification, and equivalence",
"homepage": "https://omp.sh",
"author": "Can Boluk",
@@ -208,6 +208,10 @@ function applyGeneratedModelPolicy(model: ModelSpec<Api>): void {
model.maxTokens = copilotLimits.maxTokens;
}
if (model.provider === "ollama-cloud") {
model.omitMaxOutputTokens = true;
}
// GLM Coding Plan: GLM-5.2 is the selectable 1M served id; pin it so
// endpoint discovery or older bundled fallbacks cannot regress to 200k.
if ((model.provider === "zai" || model.provider === "zhipu-coding-plan") && model.id === "glm-5.2") {
+65 -39
View File
@@ -46,14 +46,7 @@ interface CacheEntry<TApi extends Api = Api> {
let sharedDb: Database | null = null;
let sharedDbPath: string | null = null;
function getDb(dbPath?: string): Database {
const resolvedPath = dbPath ?? getModelDbPath();
if (sharedDb && sharedDbPath === resolvedPath) {
return sharedDb;
}
if (sharedDb) {
sharedDb.close();
}
function openDb(resolvedPath: string): Database {
const db = new Database(resolvedPath, { create: true });
// Install the busy handler BEFORE any lock-taking statement. See
// https://github.com/can1357/oh-my-pi/issues/2421.
@@ -70,16 +63,42 @@ function getDb(dbPath?: string): Database {
)
`);
migrateCacheSchema(db);
return db;
}
function getSharedDb(): Database {
const resolvedPath = getModelDbPath();
if (sharedDb && sharedDbPath === resolvedPath) {
return sharedDb;
}
if (sharedDb) {
sharedDb.close();
}
const db = openDb(resolvedPath);
sharedDb = db;
sharedDbPath = resolvedPath;
return db;
}
function withModelCacheDb<T>(dbPath: string | undefined, useDb: (db: Database) => T): T {
if (!dbPath) return useDb(getSharedDb());
const db = openDb(dbPath);
try {
return useDb(db);
} finally {
db.close();
}
}
function migrateCacheSchema(db: Database): void {
const columns = db.prepare("PRAGMA table_info(model_cache)").all() as TableInfoRow[];
if (!columns.some(column => column.name === "static_fingerprint")) {
db.run("ALTER TABLE model_cache ADD COLUMN static_fingerprint TEXT NOT NULL DEFAULT ''");
const stmt = db.prepare("PRAGMA table_info(model_cache)");
try {
const columns = stmt.all() as TableInfoRow[];
if (!columns.some(column => column.name === "static_fingerprint")) {
db.run("ALTER TABLE model_cache ADD COLUMN static_fingerprint TEXT NOT NULL DEFAULT ''");
}
} finally {
stmt.finalize();
}
db.run("UPDATE model_cache SET version = ? WHERE version = 2", [CACHE_SCHEMA_VERSION]);
}
@@ -91,21 +110,27 @@ export function readModelCache<TApi extends Api>(
dbPath?: string,
): CacheEntry<TApi> | null {
try {
const db = getDb(dbPath);
const row = db.query<CacheRow, [string]>("SELECT * FROM model_cache WHERE provider_id = ?").get(providerId);
if (!row || row.version !== CACHE_SCHEMA_VERSION) {
return null;
}
const models = JSON.parse(row.models) as ModelSpec<TApi>[];
const ageMs = now() - row.updated_at;
const fresh = Number.isFinite(ageMs) && ageMs >= 0 && ageMs <= ttlMs;
return {
models,
fresh,
authoritative: row.authoritative === 1,
updatedAt: row.updated_at,
staticFingerprint: row.static_fingerprint ?? "",
};
return withModelCacheDb(dbPath, db => {
const stmt = db.query<CacheRow, [string]>("SELECT * FROM model_cache WHERE provider_id = ?");
try {
const row = stmt.get(providerId);
if (!row || row.version !== CACHE_SCHEMA_VERSION) {
return null;
}
const models = JSON.parse(row.models) as ModelSpec<TApi>[];
const ageMs = now() - row.updated_at;
const fresh = Number.isFinite(ageMs) && ageMs >= 0 && ageMs <= ttlMs;
return {
models,
fresh,
authoritative: row.authoritative === 1,
updatedAt: row.updated_at,
staticFingerprint: row.static_fingerprint ?? "",
};
} finally {
stmt.finalize();
}
});
} catch {
return null;
}
@@ -120,19 +145,20 @@ export function writeModelCache<TApi extends Api>(
dbPath?: string,
): void {
try {
const db = getDb(dbPath);
db.run(
`INSERT OR REPLACE INTO model_cache (provider_id, version, updated_at, authoritative, static_fingerprint, models)
VALUES (?, ?, ?, ?, ?, ?)`,
[
providerId,
CACHE_SCHEMA_VERSION,
updatedAt,
authoritative ? 1 : 0,
staticFingerprint,
JSON.stringify(models.map(model => ({ ...model, compat: model.compatConfig, compatConfig: undefined }))),
],
);
withModelCacheDb(dbPath, db => {
db.run(
`INSERT OR REPLACE INTO model_cache (provider_id, version, updated_at, authoritative, static_fingerprint, models)
VALUES (?, ?, ?, ?, ?, ?)`,
[
providerId,
CACHE_SCHEMA_VERSION,
updatedAt,
authoritative ? 1 : 0,
staticFingerprint,
JSON.stringify(models.map(model => ({ ...model, compat: model.compatConfig, compatConfig: undefined }))),
],
);
});
} catch {
// Cache writes are best-effort; failures should not break model resolution.
}
+52 -12
View File
@@ -53624,6 +53624,7 @@
},
"contextWindow": 163840,
"maxTokens": 32000,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53652,6 +53653,7 @@
},
"contextWindow": 163840,
"maxTokens": 163840,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53680,6 +53682,7 @@
},
"contextWindow": 163840,
"maxTokens": 65536,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53708,6 +53711,7 @@
},
"contextWindow": 1048576,
"maxTokens": 1048576,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53737,6 +53741,7 @@
},
"contextWindow": 1048576,
"maxTokens": 1048576,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53765,7 +53770,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144
"maxTokens": 262144,
"omitMaxOutputTokens": true
},
"devstral-small-2:24b": {
"id": "devstral-small-2:24b",
@@ -53785,7 +53791,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144
"maxTokens": 262144,
"omitMaxOutputTokens": true
},
"gemini-3-flash-preview": {
"id": "gemini-3-flash-preview",
@@ -53806,6 +53813,7 @@
},
"contextWindow": 1048576,
"maxTokens": 65536,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53836,6 +53844,7 @@
},
"contextWindow": 262144,
"maxTokens": 262144,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53864,6 +53873,7 @@
},
"contextWindow": 202752,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53892,6 +53902,7 @@
},
"contextWindow": 202752,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53920,6 +53931,7 @@
},
"contextWindow": 202752,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53948,6 +53960,7 @@
},
"contextWindow": 202752,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -53976,6 +53989,7 @@
},
"contextWindow": 976000,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54005,6 +54019,7 @@
},
"contextWindow": 131072,
"maxTokens": 32768,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54033,6 +54048,7 @@
},
"contextWindow": 131072,
"maxTokens": 32768,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54061,6 +54077,7 @@
},
"contextWindow": 262144,
"maxTokens": 262144,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54089,7 +54106,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144
"maxTokens": 262144,
"omitMaxOutputTokens": true
},
"kimi-k2.5": {
"id": "kimi-k2.5",
@@ -54110,6 +54128,7 @@
},
"contextWindow": 262144,
"maxTokens": 262144,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54139,6 +54158,7 @@
},
"contextWindow": 262144,
"maxTokens": 262144,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54168,6 +54188,7 @@
},
"contextWindow": 262144,
"maxTokens": 262144,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54196,6 +54217,7 @@
},
"contextWindow": 204800,
"maxTokens": 128000,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54225,6 +54247,7 @@
},
"contextWindow": 204800,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54254,6 +54277,7 @@
},
"contextWindow": 204800,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54283,6 +54307,7 @@
},
"contextWindow": 196608,
"maxTokens": 196608,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54313,6 +54338,7 @@
},
"contextWindow": 512000,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54341,7 +54367,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 128000
"maxTokens": 128000,
"omitMaxOutputTokens": true
},
"ministral-3:3b": {
"id": "ministral-3:3b",
@@ -54361,7 +54388,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 128000
"maxTokens": 128000,
"omitMaxOutputTokens": true
},
"ministral-3:8b": {
"id": "ministral-3:8b",
@@ -54381,7 +54409,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 128000
"maxTokens": 128000,
"omitMaxOutputTokens": true
},
"mistral-large-3:675b": {
"id": "mistral-large-3:675b",
@@ -54401,7 +54430,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 262144
"maxTokens": 262144,
"omitMaxOutputTokens": true
},
"nemotron-3-nano:30b": {
"id": "nemotron-3-nano:30b",
@@ -54421,6 +54451,7 @@
},
"contextWindow": 1048576,
"maxTokens": 131072,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54449,6 +54480,7 @@
},
"contextWindow": 262144,
"maxTokens": 65536,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54477,6 +54509,7 @@
},
"contextWindow": 262144,
"maxTokens": 128000,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54504,7 +54537,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536
"maxTokens": 65536,
"omitMaxOutputTokens": true
},
"qwen3-coder:480b": {
"id": "qwen3-coder:480b",
@@ -54523,7 +54557,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 65536
"maxTokens": 65536,
"omitMaxOutputTokens": true
},
"qwen3-next:80b": {
"id": "qwen3-next:80b",
@@ -54543,6 +54578,7 @@
},
"contextWindow": 262144,
"maxTokens": 32768,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54572,6 +54608,7 @@
},
"contextWindow": 262144,
"maxTokens": 32768,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54600,7 +54637,8 @@
"cacheWrite": 0
},
"contextWindow": 262144,
"maxTokens": 131072
"maxTokens": 131072,
"omitMaxOutputTokens": true
},
"qwen3.5:397b": {
"id": "qwen3.5:397b",
@@ -54621,6 +54659,7 @@
},
"contextWindow": 262144,
"maxTokens": 65536,
"omitMaxOutputTokens": true,
"thinking": {
"mode": "effort",
"efforts": [
@@ -54648,7 +54687,8 @@
"cacheWrite": 0
},
"contextWindow": 32768,
"maxTokens": 4096
"maxTokens": 4096,
"omitMaxOutputTokens": true
}
},
"openai": {
@@ -84815,4 +84855,4 @@
}
}
}
}
}
@@ -159,6 +159,7 @@ export function ollamaCloudModelManagerOptions(
discoveredContextWindow !== null && discoveredContextWindow !== undefined
? (providerReference?.maxTokens ?? Math.min(contextWindow, 8192))
: Math.min(contextWindow, 8192),
omitMaxOutputTokens: true,
};
}),
);
+5
View File
@@ -594,6 +594,11 @@ export interface Model<TApi extends Api = Api> {
baseUrl: string;
reasoning: boolean;
input: ("text" | "image")[];
/**
* Decoder family used for image inputs when it has narrower format support
* than OMP's general image pipeline. `stb` local backends reject WebP.
*/
imageInputDecoder?: "stb";
/**
* Native provider tool-call support. `false` is the only unsupported signal:
* `true` and `undefined` both mean callers may use native tools. Catalog and
@@ -196,6 +196,30 @@ describe("generated model policies", () => {
expect(models[2]?.maxTokens).toBe(64000);
});
it("marks Ollama Cloud generated rows to omit max output tokens", () => {
const models: ModelSpec<Api>[] = [
createSpec({
id: "deepseek-v4-flash",
api: "ollama-chat",
provider: "ollama-cloud",
contextWindow: 1048576,
maxTokens: 1048576,
}),
createSpec({
id: "deepseek-v4-flash",
api: "ollama-chat",
provider: "ollama",
contextWindow: 1048576,
maxTokens: 1048576,
}),
];
applyGeneratedModelPolicies(models);
expect(models[0]?.omitMaxOutputTokens).toBe(true);
expect(models[1]?.omitMaxOutputTokens).toBeUndefined();
});
it("marks OpenCode Go MiMo models as not supporting tool_choice", () => {
const models: ModelSpec<"openai-completions">[] = [
createSpec({
@@ -48,6 +48,40 @@ test("ollama-cloud discovery does not inherit unsafe cross-provider maxTokens",
expect(model?.maxTokens).toBe(8192);
});
test("ollama-cloud discovery always omits max output tokens", async () => {
const fetchMock: FetchImpl = vi.fn(async (input, _init) => {
const url = String(input);
if (url === "https://ollama.com/api/tags") {
return new Response(JSON.stringify({ models: [{ name: "deepseek-v4-flash" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
if (url === "https://ollama.com/api/show") {
return new Response(
JSON.stringify({
capabilities: ["completion", "thinking"],
model_info: { "deepseek4.context_length": 1048576 },
}),
{
status: 200,
headers: { "Content-Type": "application/json" },
},
);
}
throw new Error(`Unexpected URL: ${url}`);
});
const options = ollamaCloudModelManagerOptions({ apiKey: "cloud-test-key", fetch: fetchMock });
const models = await options.fetchDynamicModels?.();
const model = models?.find(candidate => candidate.id === "deepseek-v4-flash");
expect(model?.provider).toBe("ollama-cloud");
expect(model?.contextWindow).toBe(1048576);
expect(model?.maxTokens).toBe(1048576);
expect(model?.omitMaxOutputTokens).toBe(true);
});
test("ollama-chat omits num_predict when model opts out of max output tokens", async () => {
let requestBody: Record<string, unknown> | undefined;
const fetchMock: FetchImpl = vi.fn(async (_input, init) => {
+32
View File
@@ -6,6 +6,35 @@
- Added `friendlyName` support for hidden secrets so model-visible placeholders can carry sanitized semantic labels, content-derived hashes, and case hints while preserving exact deobfuscation ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)).
## [16.1.3] - 2026-06-19
### Changed
- Refactored Perplexity authentication logic to prioritize cookies over OAuth in search operations
- Updated `token` command to correctly display active Perplexity OAuth tokens when present
### Fixed
- Enabled auto-retry for AI "thinking loop" errors encountered during model inference
- Cleared stale error banners automatically when triggered by an auto-retry recovery phase
- Preserved bundled `omitMaxOutputTokens` policy when fresh cached provider discovery rows replace Ollama Cloud catalog models, so stale `models.db` entries cannot re-enable context-window-sized `num_predict` values. ([#2984](https://github.com/can1357/oh-my-pi/issues/2984))
- Normalized cached-only Ollama Cloud discovery rows to omit on-the-wire output-token caps even when the cached model id has no bundled catalog entry. ([#2984](https://github.com/can1357/oh-my-pi/issues/2984))
- Fixed Ollama, LM Studio, and llama.cpp (plus loopback vLLM / sglang servers) reprocessing the full prompt on every turn because `provider.appendOnlyContext: auto` only recognized DeepSeek and Xiaomi as prefix-cache providers. The auto-detect now enables append-only mode for `ollama`, `ollama-cloud`, `lm-studio`, `llama.cpp`, and any baseUrl resolving to a loopback/RFC1918/`.local` host, so the system prompt + tool catalogue + prior-turn message bytes stay byte-stable across turns and llama.cpp's KV-cache prefix reuse can hit ([#3033](https://github.com/can1357/oh-my-pi/issues/3033)).
- Isolated mnemopi's local embedding provider in a dedicated `Bun.spawn` subprocess so `onnxruntime-node` and `fastembed` never load into the main agent process. Previously `memory.backend: mnemopi` crashed Bun on Windows — standalone binaries faulted in the NAPI `process.dlopen` constructor at session start, npm installs faulted in the NAPI finalizer at process teardown. Mirrors the tiny-model isolation pattern from [#1607](https://github.com/can1357/oh-my-pi/pull/1607); the parent SIGKILLs the child on dispose so the destructor never runs in either address space ([#3031](https://github.com/can1357/oh-my-pi/issues/3031)).
- Fixed image tool registration resolving image provider credentials during session startup, so broken or slow `google-antigravity` OAuth state no longer blocks sessions that never invoke `generate_image` ([#3036](https://github.com/can1357/oh-my-pi/issues/3036)).
- Fixed LSP client returning `-32601 Method not found` for defined server→client requests (`window/showMessageRequest`, `window/showDocument`, `workspace/{semanticTokens,inlayHint,codeLens,codeAction,diagnostic}/refresh`). Servers that stall waiting for a real reply (same failure mode as #3029) now receive the spec no-op result ([#3044](https://github.com/can1357/oh-my-pi/issues/3044)).
- Fixed WebP images being sent unchanged to `local-server` vision models, which can fail through llama.cpp/STB-backed decoders that do not support WebP ([#2922](https://github.com/can1357/oh-my-pi/issues/2922)).
- Made `getSettingsListTheme`, `getEditorTheme`, `getSelectListTheme`, and `getSymbolTheme` return a plain ASCII fallback instead of crashing with "undefined is not an object (evaluating 'theme.fg')" when the global `theme` is undefined — e.g. when a plugin calls them before `initTheme()` completes or from a separate module instance under npm-global installs. ([#2998](https://github.com/can1357/oh-my-pi/issues/2998))
- Hardened TTS, STT, and tiny-title worker IPC `send()` paths against async EPIPE rejections: `Subprocess.send()` is now wrapped so neither a synchronous "process exited" throw nor an asynchronous EPIPE rejection (when the pipe breaks between exit being observed and the next send) can escape as a fatal unhandled rejection. A dying Kokoro/TTS/STT worker can no longer crash the whole agent session mid-task. ([#2997](https://github.com/can1357/oh-my-pi/issues/2997))
- Fixed Windows test failures caused by path handling: tests now use `pathToFileURL`, `path.resolve`, and `path.join` instead of hard-coded POSIX paths; `shortenPath()` normalizes backslashes to forward slashes after `~` and respects home directory boundaries; shell-escaped interpolated paths in bash tool tests to prevent Git Bash eating backslashes
- Fixed `HistoryStorage.resetInstance()` leaking its SQLite database handle on Windows by adding a `#close()` method that finalizes all prepared statements and closes the database; `AgentStorage` gained the same `resetInstance()`/`#close()` pattern
- Fixed `createAgentSession` leaking the internally-created `AuthStorage` when session construction fails before the session takes ownership, causing EBUSY on Windows temp dir cleanup
- Fixed `MnemopiBackend.removeDbFiles()` throwing on Windows when the database handle is still being released; it is now truly best-effort (logs failures instead of silently swallowing)
- Fixed Windows EBUSY test failures by replacing raw `fs.rmSync`/`fs.rm` cleanup with `TempDir` (which retries) and best-effort `.catch(() => {})` where SQLite handles outlive the test
- Fixed `TempDir` prefix convention: non-`@` prefixes created temp dirs relative to cwd instead of `os.tmpdir()`, causing module resolution failures on Windows
- Fixed git line-ending mismatches in autoresearch tests by setting `core.autocrlf false` in test repo initialization
- Fixed Bedrock inference-profile ARN models being dropped from the allowed-model set when models were scoped via `enabledModels`, the SDK, or ACP, so an accepted ARN no longer resolves to an empty selection. ([#3006](https://github.com/can1357/oh-my-pi/pull/3006))
## [16.1.2] - 2026-06-19
### Added
@@ -14,12 +43,15 @@
### Changed
- Renamed the search tool's `i` parameter to `case` and inverted its semantics to represent case-sensitive search.
- Improved session history to export empty objects as `{}` instead of empty strings
- Refined system prompt and tool documentation to improve conciseness and clarity
- Simplified tool input descriptions for browser, eval, find, and memory-edit operations
- Refactored authentication storage discovery to share logic with other pi-ai tools
### Fixed
- Fixed `omp bench` resolving an ambiguous model selector — a bare or canonical id shared by several providers (e.g. `gpt-oss-20b` or `openai/gpt-oss-20b`) — to a provider you have no credentials for. Bench resolves against the full catalog (credentials are ignored), so the default pick was decided by provider-priority order alone. It now redirects such selectors to an equivalent model under a provider with configured auth (honoring `modelProviderOrder` and canonical cross-provider variants), while an explicit `provider/id` selector is still benchmarked verbatim so forced/unauthenticated runs keep working.
- Resuming a session whose project directory no longer exists (deleted or renamed worktree) no longer crashes with an unhandled `ENOENT … chdir` rejection. The resume now keeps the current working directory instead of trying to `chdir` into the missing path, across the in-session selector, the `--resume` startup picker, and `SessionManager.open`/`continueRecent`.
- Fixed streaming reflowing Markdown — a fenced mermaid diagram or a GFM table — stranding stale fragments in native scrollback once the reply scrolled past the viewport (cleared only by a full repaint / Ctrl+L). While streaming, the assistant block defaulted to commit-stable, so the transcript advertised its scrolled-off rows as durable snapshot content and the renderer committed an intermediate layout to immutable terminal history; the later re-layout (a diagram reshaping, a table re-aligning its columns) then froze that superseded fragment in scrollback. A still-streaming reply whose Markdown carries a mermaid fence or a table — detected outside fenced code blocks so ordinary code snippets are unaffected — is now commit-unstable, so it stays wholly in the repaintable live region and commits once, at its final layout, when the turn finalizes.
- Fixed `SYSTEM.md` prompt customization going through the raw system prompt override path, which dropped sections rendered by `custom-system-prompt.md` such as skills and rules ([#3014](https://github.com/can1357/oh-my-pi/issues/3014)).
+1 -1
View File
@@ -1,7 +1,7 @@
{
"type": "module",
"name": "@oh-my-pi/pi-coding-agent",
"version": "16.1.2",
"version": "16.1.3",
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
"homepage": "https://omp.sh",
"author": "Can Boluk",
+8
View File
@@ -68,6 +68,7 @@ async function runSmokeTest(): Promise<void> {
const { smokeTestTinyTitleWorker } = await import("./tiny/title-client");
const { smokeTestSttWorker } = await import("./stt/asr-client");
const { smokeTestTtsWorker } = await import("./tts/tts-client");
const { smokeTestMnemopiEmbedWorker } = await import("./mnemopi/embed-client");
const { smokeTestJsEvalWorker } = await import("./eval/js/context-manager");
await smokeTestSyncWorker();
@@ -87,6 +88,7 @@ async function runSmokeTest(): Promise<void> {
await smokeTestSttWorker();
await smokeTestJsEvalWorker();
await smokeTestTtsWorker();
await smokeTestMnemopiEmbedWorker();
process.stdout.write("smoke-test: ok\n");
}
@@ -96,6 +98,7 @@ const TAB_WORKER_ARG = "__omp_worker_tab";
const JS_EVAL_WORKER_ARG = "__omp_worker_js_eval";
const STT_WORKER_ARG = "__omp_worker_stt";
const TTS_WORKER_ARG = "__omp_worker_tts";
const MNEMOPI_EMBED_WORKER_ARG = "__omp_worker_mnemopi_embed";
async function runWorkerEntrypoint(arg: string | undefined): Promise<boolean> {
if (arg === TINY_WORKER_ARG) {
@@ -151,6 +154,11 @@ async function runWorkerEntrypoint(arg: string | undefined): Promise<boolean> {
await runIpcSubprocessWorker(startTtsWorker);
return true;
}
if (arg === MNEMOPI_EMBED_WORKER_ARG) {
const { startMnemopiEmbedWorker } = await import("./mnemopi/embed-worker");
await runIpcSubprocessWorker(startMnemopiEmbedWorker);
return true;
}
return false;
}
+64 -3
View File
@@ -11,7 +11,7 @@ import type {
SimpleStreamOptions,
} from "@oh-my-pi/pi-ai";
import { streamSimple } from "@oh-my-pi/pi-ai";
import type { CanonicalModelVariant } from "@oh-my-pi/pi-catalog/identity";
import { buildModelProviderPriorityRank, type CanonicalModelVariant } from "@oh-my-pi/pi-catalog/identity";
import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui";
import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils";
import chalk from "chalk";
@@ -50,6 +50,7 @@ export interface BenchModelRegistry {
resolveCanonicalModel?(canonicalId: string, options?: CanonicalModelQueryOptions): Model<Api> | undefined;
getCanonicalVariants?(canonicalId: string, options?: CanonicalModelQueryOptions): CanonicalModelVariant[];
getCanonicalId?(model: Model<Api>): string | undefined;
hasConfiguredAuth?(model: Model<Api>): boolean;
}
export interface BenchRuntime {
@@ -346,6 +347,56 @@ interface BenchTarget {
thinking: ResolvedThinkingLevel | undefined;
}
/** Highest-priority provider variant: native/OAuth transports outrank mirrors. */
function pickHighestPriorityProvider(models: Model<Api>[], providerOrder?: readonly string[]): Model<Api> | undefined {
if (models.length <= 1) return models[0];
const priority = buildModelProviderPriorityRank(providerOrder);
return [...models].sort((a, b) => {
const aRank = priority.get(a.provider.toLowerCase()) ?? Number.POSITIVE_INFINITY;
const bRank = priority.get(b.provider.toLowerCase()) ?? Number.POSITIVE_INFINITY;
return aRank - bRank;
})[0];
}
/**
* Bench resolves selectors against the entire catalog (credentials are ignored),
* so an ambiguous id shared by several providers can land on one the user never
* authenticated. For non-pinned selectors, redirect to an equivalent model under
* a provider with configured auth. An explicit `provider/id` selector is honored
* verbatim — even unauthenticated — so forced benchmarking keeps working.
*/
function resolveAuthenticatedAlternative(
selector: string,
model: Model<Api>,
modelRegistry: BenchModelRegistry,
providerOrder?: readonly string[],
): Model<Api> | undefined {
if (!modelRegistry.hasConfiguredAuth) return undefined;
// A pinned `provider/...` selector is authoritative; never redirect off it.
if (selector.trim().toLowerCase().startsWith(`${model.provider.toLowerCase()}/`)) return undefined;
if (modelRegistry.hasConfiguredAuth(model)) return undefined;
const seen = new Set<string>();
const authenticated: Model<Api>[] = [];
const consider = (candidate: Model<Api>): void => {
const key = `${candidate.provider}/${candidate.id}`;
if (seen.has(key)) return;
seen.add(key);
if (modelRegistry.hasConfiguredAuth?.(candidate)) authenticated.push(candidate);
};
// Canonical variants link the same logical model across providers even when
// ids differ (e.g. fireworks `gpt-oss-20b` <-> openrouter `openai/gpt-oss-20b`).
const canonicalId = modelRegistry.getCanonicalId?.(model);
if (canonicalId) {
for (const variant of modelRegistry.getCanonicalVariants?.(canonicalId) ?? []) consider(variant.model);
}
// Same-id fallback for entries outside the canonical index.
for (const candidate of modelRegistry.getAll()) {
if (candidate.id === model.id) consider(candidate);
}
return pickHighestPriorityProvider(authenticated, providerOrder);
}
function resolveBenchModels(
selectors: string[],
modelRegistry: BenchModelRegistry,
@@ -366,10 +417,20 @@ function resolveBenchModels(
continue;
}
if (result.warning) writeStderr(`${chalk.yellow(`Warning: ${result.warning}`)}\n`);
let model = result.model;
const authenticated = resolveAuthenticatedAlternative(selector, model, modelRegistry, preferences.providerOrder);
if (authenticated) {
writeStderr(
`${chalk.yellow(
`Warning: no credentials for "${model.provider}"; benchmarking ${formatModelString(authenticated)} instead. Pin "${formatModelString(model)}" to force it.`,
)}\n`,
);
model = authenticated;
}
resolved.push({
selector,
model: result.model,
thinking: resolveThinkingLevelForModel(result.model, result.thinkingLevel),
model,
thinking: resolveThinkingLevelForModel(model, result.thinkingLevel),
});
}
if (errors.length > 0) {
+56 -37
View File
@@ -7,6 +7,7 @@ import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
import chalk from "chalk";
import { isAuthenticated, ModelRegistry } from "../config/model-registry";
import { discoverAuthStorage } from "../sdk";
import { getAvailableAuthMethods } from "../web/search/providers/perplexity-auth";
export default class Token extends Command {
static description = "Get the API key or OAuth token for a provider";
@@ -41,49 +42,67 @@ export default class Token extends Command {
const provider = providerName.toLowerCase();
const authStorage = await discoverAuthStorage();
const modelRegistry = new ModelRegistry(authStorage);
try {
const modelRegistry = new ModelRegistry(authStorage);
// Resolve the API key / token
const apiKey = await modelRegistry.getApiKeyForProvider(provider, undefined, {
forceRefresh: flags["force-refresh"],
});
// Resolve the API key / token
let apiKey: string | undefined;
if (!isAuthenticated(apiKey)) {
// Find all active/configured providers
const activeProviders = new Set<string>();
for (const p of PROVIDER_REGISTRY) {
if (authStorage.hasAuth(p.id)) {
activeProviders.add(p.id);
}
}
const all = authStorage.getAll();
for (const p in all) {
if (authStorage.hasAuth(p)) {
activeProviders.add(p);
if (provider === "perplexity") {
const methods = await getAvailableAuthMethods(authStorage, undefined, {
forceRefresh: flags["force-refresh"],
});
const printable = methods.find(m => m.type === "oauth" || m.type === "api_key");
if (printable) {
apiKey = printable.type === "oauth" ? printable.access.accessToken : printable.apiKey;
}
}
const msg = `No active credential found for provider "${providerName}".`;
process.stderr.write(`${chalk.red(msg)}\n`);
if (activeProviders.size > 0) {
process.stderr.write(`Configured providers: ${Array.from(activeProviders).sort().join(", ")}\n`);
if (!apiKey) {
apiKey = await modelRegistry.getApiKeyForProvider(provider, undefined, {
forceRefresh: flags["force-refresh"],
});
}
process.exitCode = 1;
return;
if (!isAuthenticated(apiKey)) {
// Find all active/configured providers
const activeProviders = new Set<string>();
for (const p of PROVIDER_REGISTRY) {
if (authStorage.hasAuth(p.id)) {
activeProviders.add(p.id);
}
}
const all = authStorage.getAll();
for (const p in all) {
if (authStorage.hasAuth(p)) {
activeProviders.add(p);
}
}
const msg = `No active credential found for provider "${providerName}".`;
process.stderr.write(`${chalk.red(msg)}\n`);
if (activeProviders.size > 0) {
process.stderr.write(`Configured providers: ${Array.from(activeProviders).sort().join(", ")}\n`);
}
process.exitCode = 1;
return;
}
if (!flags.raw) {
try {
const parsed = JSON.parse(apiKey);
if (parsed && typeof parsed === "object" && typeof parsed.token === "string") {
process.stdout.write(`${parsed.token}\n`);
return;
}
} catch {
// Not a JSON string, print as-is
}
}
process.stdout.write(`${apiKey}\n`);
} finally {
authStorage.close();
}
if (!flags.raw) {
try {
const parsed = JSON.parse(apiKey);
if (parsed && typeof parsed === "object" && typeof parsed.token === "string") {
process.stdout.write(`${parsed.token}\n`);
return;
}
} catch {
// Not a JSON string, print as-is
}
}
process.stdout.write(`${apiKey}\n`);
}
}
@@ -8,10 +8,55 @@ export interface AppendOnlyContextModel {
compatConfig?: object;
}
/**
* Local model servers (Ollama, LM Studio, llama.cpp, vLLM, sglang, …) all
* rely on llama.cpp-style prefix KV-cache reuse: identical leading tokens
* skip re-prefill on the next request. Append-only mode is the only way to
* guarantee byte-stable bytes across turns, since the live system prompt,
* tool catalogue, and message log all flow through fresh allocations every
* step (see `agent-loop.ts` `streamAssistantResponse` fallback path).
*/
const LOCAL_INFERENCE_PROVIDERS = new Set(["ollama", "ollama-cloud", "lm-studio", "llama.cpp"]);
/** True when `baseUrl` resolves to a loopback or RFC1918 host — covers
* llama.cpp/vLLM/sglang servers registered under a user-defined provider id
* via `models.yaml`. Built-in local provider ids (`ollama`, `lm-studio`,
* `llama.cpp`) are already handled by `LOCAL_INFERENCE_PROVIDERS`.
* Substring match on the parsed hostname only; ports, paths, and unparseable
* URLs return false.
*/
function hasLocalLoopbackBaseUrl(baseUrl: string | undefined): boolean {
if (!baseUrl) return false;
let hostname: string;
try {
hostname = new URL(baseUrl).hostname.toLowerCase();
} catch {
return false;
}
if (
hostname === "localhost" ||
hostname === "127.0.0.1" ||
hostname === "0.0.0.0" ||
hostname === "::1" ||
hostname === "[::1]"
) {
return true;
}
// RFC1918 private IPv4 ranges.
if (/^10\./.test(hostname)) return true;
if (/^192\.168\./.test(hostname)) return true;
if (/^172\.(1[6-9]|2[0-9]|3[01])\./.test(hostname)) return true;
// Common ".local" mDNS hostnames used for home-LAN llama.cpp boxes.
if (hostname.endsWith(".local")) return true;
return false;
}
function shouldAutoEnableAppendOnlyContext(model: AppendOnlyContextModel | null | undefined): boolean {
if (!model) return false;
if (model.provider === "deepseek") return true;
if (LOCAL_INFERENCE_PROVIDERS.has(model.provider)) return true;
if (hostMatchesUrl(model.baseUrl, "xiaomi")) return true;
if (hasLocalLoopbackBaseUrl(model.baseUrl)) return true;
return !!model.compatConfig && "supportsStore" in model.compatConfig && model.compatConfig.supportsStore === true;
}
@@ -275,6 +275,7 @@ export async function discoverOllamaModels(
baseUrl: `${endpoint}/v1`,
reasoning: metadata?.reasoning ?? false,
input: metadata?.input ?? ["text"],
imageInputDecoder: "stb",
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: metadata?.contextWindow ?? 128000,
maxTokens: Math.min(metadata?.contextWindow ?? Number.POSITIVE_INFINITY, DISCOVERY_DEFAULT_MAX_TOKENS),
@@ -352,6 +353,7 @@ export async function discoverLlamaCppModels(
baseUrl,
reasoning: false,
input: serverMetadata?.input ?? ["text"],
imageInputDecoder: "stb",
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: serverMetadata?.contextWindow ?? 128000,
maxTokens: Math.min(
@@ -424,6 +426,7 @@ export async function discoverOpenAIModelsList(
baseUrl,
reasoning: false,
input: nativeMetadataForModel?.input ?? ["text"],
...(providerConfig.discovery.type === "lm-studio" ? { imageInputDecoder: "stb" as const } : {}),
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow,
maxTokens: Math.min(contextWindow, discoveryDefaultMaxTokens(providerConfig.api)),
@@ -900,6 +900,7 @@ export class ModelRegistry {
...replacementModel,
contextWindow: replacementModel.contextWindow ?? existing.contextWindow,
maxTokens: replacementModel.maxTokens ?? existing.maxTokens,
omitMaxOutputTokens: replacementModel.omitMaxOutputTokens ?? existing.omitMaxOutputTokens,
...(supportsTools !== undefined ? { supportsTools } : {}),
};
});
@@ -1023,12 +1024,21 @@ export class ModelRegistry {
}
#normalizeDiscoverableModels(providerConfig: DiscoveryProviderConfig, models: Model<Api>[]): Model<Api>[] {
const withDecoderMetadata =
providerConfig.discovery.type === "ollama" ||
providerConfig.discovery.type === "llama.cpp" ||
providerConfig.discovery.type === "lm-studio"
? models.map(model =>
buildModel({ ...model, imageInputDecoder: "stb", compat: model.compatConfig } as ModelSpec<Api>),
)
: models;
if (providerConfig.provider !== "ollama" || providerConfig.api !== "openai-responses") {
return models;
return withDecoderMetadata;
}
const contextLengthOverride = getOllamaContextLengthOverride();
return models.map(model => {
return withDecoderMetadata.map(model => {
const normalized =
model.api === "openai-completions"
? buildModel({
@@ -1269,7 +1279,12 @@ export class ModelRegistry {
models: cached?.models.map(model => model.id) ?? [],
});
this.#lastDiscoveryWarnings.delete(providerConfig.provider);
return cached ? cached.models.map(model => buildModel(model)) : [];
return cached
? this.#normalizeDiscoverableModels(
providerConfig,
cached.models.map(model => buildModel(model)),
)
: [];
}
}
@@ -1569,6 +1584,9 @@ export class ModelRegistry {
}
#applyHardcodedModelPolicies(models: Model<Api>[]): Model<Api>[] {
return models.map(model => {
if (model.provider === "ollama-cloud" && model.omitMaxOutputTokens !== true) {
model = applyModelOverride(model, { omitMaxOutputTokens: true });
}
if (model.id !== "gpt-5.4" || model.provider === "github-copilot") {
return model;
}
@@ -556,6 +556,27 @@ function isAlias(id: string): boolean {
return !datePattern.test(id);
}
function includeSyntheticAllowedModels(available: Model<Api>[], allowedModels: Iterable<Model<Api>>): Model<Api>[] {
const allowedByKey = new Map<string, Model<Api>>();
for (const model of allowedModels) {
const key = formatModelString(model);
if (!allowedByKey.has(key)) {
allowedByKey.set(key, model);
}
}
if (allowedByKey.size === 0) return [];
const result: Model<Api>[] = [];
for (const model of available) {
if (allowedByKey.delete(formatModelString(model))) {
result.push(model);
}
}
result.push(...allowedByKey.values());
return result;
}
/**
* Find an exact explicit provider/model match.
* Bare model ids are handled separately so canonical ids can coalesce variants.
@@ -1335,9 +1356,9 @@ export async function resolveModelScope(
* the result to models matching those patterns.
*
* Returns the unfiltered available list when `enabledModels` is empty.
* Returns an empty list when `enabledModels` is configured but no available
* model matches any pattern — callers MUST treat this as "no usable model"
* rather than falling back to the global default (see issue #1022).
* Returns an empty list when `enabledModels` is configured but no model matches
* any pattern — callers MUST treat this as "no usable model" rather than
* falling back to the global default (see issue #1022).
*/
export async function resolveAllowedModels(
modelRegistry: Pick<ModelRegistry, "getAvailable" | "getCanonicalVariants">,
@@ -1353,8 +1374,10 @@ export async function resolveAllowedModels(
if (scoped.length === 0) {
return [];
}
const allowed = new Set(scoped.map(entry => `${entry.model.provider}/${entry.model.id}`));
return available.filter(model => allowed.has(`${model.provider}/${model.id}`));
return includeSyntheticAllowedModels(
available,
scoped.map(entry => entry.model),
);
}
/**
@@ -1382,9 +1405,9 @@ export function filterAvailableModelsByEnabledPatterns(
if (patterns.length === 0) return available;
const context = buildPreferenceContext(available, undefined);
const allowed = new Set<string>();
const allowedModels: Model<Api>[] = [];
const addAllowed = (model: Model<Api>) => {
allowed.add(`${model.provider}/${model.id}`);
allowedModels.push(model);
};
for (const pattern of patterns) {
@@ -1409,7 +1432,7 @@ export function filterAvailableModelsByEnabledPatterns(
}
}
return allowed.size === 0 ? [] : available.filter(model => allowed.has(`${model.provider}/${model.id}`));
return includeSyntheticAllowedModels(available, allowedModels);
}
export interface ResolveCliModelResult {
+1 -1
View File
@@ -181,7 +181,7 @@ export class CursorExecHandlers implements ICursorExecHandlers {
const toolResultMessage = await executeTool(this.options, "search", toolCallId, {
pattern: args.pattern,
paths: [searchPath],
i: args.caseInsensitive || undefined,
case: args.caseInsensitive === true ? false : undefined,
});
return toolResultMessage;
}
@@ -39,7 +39,6 @@ import type { LoadedConfig } from "./config";
## Exceptions
- Timer handles: `ReturnType<typeof setTimeout>` / `setInterval`.
- Generic type utilities where the function is a type parameter.
Concrete function? Export a concrete type.
+1 -1
View File
@@ -5,7 +5,7 @@ if "__omp_prelude_loaded__" not in globals():
from pathlib import Path
import os, json, math, re
from urllib.parse import unquote
INTENT_FIELD = "_i"
INTENT_FIELD = "i"
# __omp_display is injected by runner.py before the prelude executes; it
# mirrors IPython's display() semantics with the same MIME bundle output.
+24
View File
@@ -482,6 +482,30 @@ async function handleServerRequest(client: LspClient, message: LspJsonRpcRequest
await sendResponse(client, message.id, null, message.method);
return;
}
if (message.method === "window/showMessageRequest") {
// Headless: no UI to surface the prompt. Spec says null = "no action selected".
await sendResponse(client, message.id, null, message.method);
return;
}
if (message.method === "window/showDocument") {
// Headless: nothing to display. Spec result is `{ success: boolean }`.
await sendResponse(client, message.id, { success: false }, message.method);
return;
}
if (
message.method === "workspace/semanticTokens/refresh" ||
message.method === "workspace/inlayHint/refresh" ||
message.method === "workspace/codeLens/refresh" ||
message.method === "workspace/codeAction/refresh" ||
message.method === "workspace/inlineValue/refresh" ||
message.method === "workspace/foldingRange/refresh" ||
message.method === "workspace/diagnostic/refresh"
) {
// Void acknowledgement per spec; servers that stall waiting for a reply
// (same failure mode as the dynamic-registration hang in #3029) move on.
await sendResponse(client, message.id, null, message.method);
return;
}
await sendResponse(client, message.id, null, message.method, {
code: -32601,
message: `Method not found: ${message.method}`,
+49 -3
View File
@@ -120,6 +120,14 @@ export const mnemopiBackend: MemoryBackend = {
const config = previous?.config ?? (session ? loadMnemopiConfig(session.settings, agentDir) : undefined);
if (!config) return;
await loadMnemopiCore();
// Close the cached default Mnemopi instance so its SQLite handle doesn't
// keep the DB files locked on Windows when removeDbFiles tries to delete.
// Use the core module (already awaited via loadMnemopiCore above):
// requireMnemopi() throws "module not loaded" when clear() runs before the
// fire-and-forget start() has awaited loadMnemopi() (autolearn disabled, or
// taskDepth > 0). resetMemoryForTests is re-exported identically from core.
requireMnemopiCore().resetMemoryForTests();
await Bun.sleep(0);
await removeDbFiles(getMnemopiScopedDbPaths(config));
},
@@ -557,10 +565,48 @@ export function getMnemopiDbDirForTests(session: AgentSession): string | undefin
return state ? path.dirname(state.config.dbPath) : undefined;
}
/**
* Best-effort removal of a SQLite DB file and its WAL/SHM sidecars.
*
* Windows keeps `-wal`/`-shm` busy briefly after the DB handle closes, so a
* single `rm` races with EBUSY/EPERM. Retry a handful of times before giving
* up; `force: true` already makes "missing" a non-error.
*/
async function removeDbFiles(dbPaths: readonly string[]): Promise<void> {
for (const dbPath of dbPaths) {
await rm(dbPath, { force: true });
await rm(`${dbPath}-wal`, { force: true });
await rm(`${dbPath}-shm`, { force: true });
for (const suffix of ["", "-wal", "-shm"]) {
await removeWithRetries(`${dbPath}${suffix}`).catch(error => {
// `force: true` already makes ENOENT a non-error; anything else
// after the full retry window means the DB is genuinely locked and
// the user's "Memory cleared" message would be misleading. Log so
// the failure is diagnosable without blocking the clear flow.
const code = typeof error === "object" && error !== null && "code" in error ? error.code : undefined;
if (code !== "ENOENT") {
logger.warn("Mnemopi: failed to remove DB file after retries", { path: `${dbPath}${suffix}`, code });
}
});
}
}
}
const kRemoveRetries = 40;
const kRemoveRetryDelayMs = 25;
const kRetryableRemoveErrorCodes = new Set(["EBUSY", "EPERM", "ENOTEMPTY"]);
async function removeWithRetries(target: string): Promise<void> {
for (let attempt = 0; ; attempt++) {
try {
await rm(target, { force: true });
return;
} catch (err) {
const retryable =
typeof err === "object" &&
err !== null &&
"code" in err &&
typeof err.code === "string" &&
kRetryableRemoveErrorCodes.has(err.code);
if (!retryable || attempt >= kRemoveRetries) throw err;
await Bun.sleep(kRemoveRetryDelayMs);
}
}
}
@@ -0,0 +1,401 @@
import * as path from "node:path";
import { $env, isBunTestRuntime, isCompiledBinary, logger, workerHostEntry } from "@oh-my-pi/pi-utils";
import type { Subprocess } from "bun";
import type { MnemopiEmbedModelId, MnemopiEmbedWorkerInbound, MnemopiEmbedWorkerOutbound } from "./embed-protocol";
/**
* Abstraction over the mnemopi embeddings subprocess. The runtime
* implementation is a Bun child process so `onnxruntime-node`'s NAPI
* constructor + finalizer never run inside the main agent address space —
* those destructors segfault Bun on Windows when mnemopi's local embedding
* provider loads fastembed in the main process (issue #3031; the mnemopi
* sibling of the tiny-model fix from #1606 / #1607).
*/
export interface MnemopiEmbedWorkerHandle {
send(message: MnemopiEmbedWorkerInbound): void;
onMessage(handler: (message: MnemopiEmbedWorkerOutbound) => void): () => void;
onError(handler: (error: Error) => void): () => void;
terminate(): Promise<void>;
}
type PendingRequest =
| { kind: "init"; model: MnemopiEmbedModelId; resolve: (ok: boolean) => void }
| { kind: "embed"; model: MnemopiEmbedModelId; resolve: (vectors: number[][] | Error) => void };
// Cold-starting the worker from a compiled binary (decompress + module graph load)
// is slow on contended CI runners; the probe only proves the worker spawns and
// ponges, so a generous bound removes flakes without weakening the check.
const SMOKE_TEST_TIMEOUT_MS = 30_000;
/**
* Hidden subcommand on the main CLI that boots the mnemopi embeddings worker
* in the spawned subprocess. Kept in sync with the dispatch in `cli.ts`.
*/
export const MNEMOPI_EMBED_WORKER_ARG = "__omp_worker_mnemopi_embed";
/**
* Env handed to the embeddings subprocess. The child inherits the parent's
* environment verbatim — fastembed honours `HF_HUB_*`, `HTTPS_PROXY`, etc.,
* and our `loadFastembed()` reads the same `OMP_*` runtime-install knobs the
* parent uses. `process.env` carries `undefined` slots that Bun.spawn rejects;
* filter them out.
*/
function mnemopiEmbedWorkerEnv(): Record<string, string> {
const base = $env as Record<string, string | undefined>;
const merged: Record<string, string> = {};
for (const key in base) {
const value = base[key];
if (typeof value === "string") merged[key] = value;
}
return merged;
}
interface MnemopiEmbedWorkerSpawnCommand {
cmd: string[];
cwd?: string;
}
/**
* Resolve the command used to relaunch the agent CLI into mnemopi-embed-worker
* mode. In a compiled binary the entry point is the binary itself; otherwise
* re-enter the declared worker-host entry (cwd-relative for reliable Bun IPC),
* falling back to this package's own `src/cli.ts` when no host entry is
* declared (bun test, SDK embedding).
*/
function mnemopiEmbedWorkerSpawnCmd(): MnemopiEmbedWorkerSpawnCommand {
if (isCompiledBinary()) return { cmd: [process.execPath, MNEMOPI_EMBED_WORKER_ARG] };
const hostEntry = workerHostEntry();
if (hostEntry) {
return {
cmd: [process.execPath, path.basename(hostEntry), MNEMOPI_EMBED_WORKER_ARG],
cwd: path.dirname(hostEntry),
};
}
const packageRoot = path.resolve(import.meta.dir, "..", "..");
return { cmd: [process.execPath, "src/cli.ts", MNEMOPI_EMBED_WORKER_ARG], cwd: packageRoot };
}
interface SpawnedSubprocess {
proc: Subprocess<"ignore", "ignore", "ignore">;
inbound: Set<(message: MnemopiEmbedWorkerOutbound) => void>;
errors: Set<(error: Error) => void>;
/**
* Flipped to `true` right before the deliberate SIGKILL so `onExit` can
* distinguish the expected hard-kill from a crash (SIGSEGV from a native
* fault, OOM SIGKILL, operator `kill -9`). Only the latter surfaces as a
* worker error so callers don't await forever.
*/
intentionalExit: { value: boolean };
}
/**
* Spawn the mnemopi embeddings worker as a subprocess. Exported for tests and
* the smoke probe; production callers go through {@link spawnMnemopiEmbedWorker}.
*/
export function createMnemopiEmbedSubprocess(): SpawnedSubprocess {
const inbound = new Set<(message: MnemopiEmbedWorkerOutbound) => void>();
const errors = new Set<(error: Error) => void>();
const intentionalExit = { value: false };
const spawnCommand = mnemopiEmbedWorkerSpawnCmd();
const proc = Bun.spawn({
cmd: spawnCommand.cmd,
cwd: spawnCommand.cwd,
env: mnemopiEmbedWorkerEnv(),
stdin: "ignore",
stdout: "ignore",
stderr: "ignore",
serialization: "advanced",
windowsHide: true,
ipc(message) {
for (const handler of inbound) handler(message as MnemopiEmbedWorkerOutbound);
},
onExit(_proc, exitCode, signalCode) {
if (exitCode === 0) return;
if (exitCode === null && intentionalExit.value) return;
const reason = exitCode !== null ? `code ${exitCode}` : `signal ${signalCode ?? "unknown"}`;
const err = new Error(`mnemopi embed subprocess exited with ${reason}`);
for (const handler of errors) handler(err);
},
});
// Don't keep the parent event loop alive on an idle worker; the agent
// dispose path calls `terminate()` explicitly. Bun's test runner starves
// IPC for unref'd subprocesses, so keep it referenced only under tests.
if (!isBunTestRuntime()) proc.unref();
return { proc, inbound, errors, intentionalExit };
}
function wrapSubprocess({ proc, inbound, errors, intentionalExit }: SpawnedSubprocess): MnemopiEmbedWorkerHandle {
return {
send(message) {
try {
proc.send(message);
} catch (error) {
logger.debug("mnemopi-embed: send to subprocess failed", {
error: error instanceof Error ? error.message : String(error),
});
}
},
onMessage(handler) {
inbound.add(handler);
return () => inbound.delete(handler);
},
onError(handler) {
errors.add(handler);
return () => errors.delete(handler);
},
async terminate() {
// SIGKILL: the point of subprocess isolation is that the parent
// never runs `onnxruntime-node`'s NAPI finalizer (it crashes Bun
// on Windows). Hard-kill instead; the OS reclaims the model
// memory. Flip the intentional-exit flag *before* killing so
// `onExit` can tell this apart from a native crash.
intentionalExit.value = true;
try {
proc.kill("SIGKILL");
} catch {
// Already gone.
}
},
};
}
function spawnInlineUnavailableWorker(error: unknown): MnemopiEmbedWorkerHandle {
const listeners = new Set<(message: MnemopiEmbedWorkerOutbound) => void>();
const errorMessage = error instanceof Error ? error.message : String(error);
const emit = (message: MnemopiEmbedWorkerOutbound): void => {
for (const listener of listeners) listener(message);
};
return {
send(message) {
queueMicrotask(() => {
if (message.type === "ping") {
emit({ type: "pong", id: message.id });
return;
}
emit({ type: "error", id: message.id, error: errorMessage });
});
},
onMessage(handler) {
listeners.add(handler);
return () => listeners.delete(handler);
},
onError() {
return () => {};
},
async terminate() {
listeners.clear();
},
};
}
function spawnMnemopiEmbedWorker(): MnemopiEmbedWorkerHandle {
try {
return wrapSubprocess(createMnemopiEmbedSubprocess());
} catch (error) {
logger.warn("mnemopi embed worker spawn failed; local embeddings disabled", {
error: error instanceof Error ? error.message : String(error),
});
return spawnInlineUnavailableWorker(error);
}
}
function logWorkerMessage(message: Extract<MnemopiEmbedWorkerOutbound, { type: "log" }>): void {
if (message.level === "debug") logger.debug(message.msg, message.meta);
else if (message.level === "warn") logger.warn(message.msg, message.meta);
else logger.error(message.msg, message.meta);
}
/**
* Per-model wrapper produced by {@link MnemopiEmbedClient.initialize}.
* `embed()` round-trips one batch of texts through the worker subprocess and
* yields the resulting vectors in a single asynchronous batch — fastembed's
* own iterator was emitting batches that we collect on the child side anyway,
* and serializing per-batch over IPC would not improve throughput.
*/
export interface MnemopiSubprocessEmbeddingModel {
embed(texts: string[], batchSize?: number): AsyncIterable<number[][]>;
}
export class MnemopiEmbedClient {
#worker: MnemopiEmbedWorkerHandle | null = null;
#unsubscribeMessage: (() => void) | null = null;
#unsubscribeError: (() => void) | null = null;
#pending = new Map<string, PendingRequest>();
#nextRequestId = 0;
#spawnWorker: () => MnemopiEmbedWorkerHandle;
constructor(spawnWorker: () => MnemopiEmbedWorkerHandle = spawnMnemopiEmbedWorker) {
this.#spawnWorker = spawnWorker;
}
/**
* Load the named fastembed model inside the subprocess. Resolves to a
* thin wrapper whose `embed()` round-trips through the same worker, or
* `null` when the worker cannot init the model (missing peer, native
* load failure, etc.). Multiple calls with the same model reuse the
* single in-flight worker; calling with a different model loads it on
* the child without restarting the process.
*/
async initialize(
model: MnemopiEmbedModelId,
cacheDir: string | undefined,
): Promise<MnemopiSubprocessEmbeddingModel | null> {
try {
const worker = this.#ensureWorker();
const id = String(++this.#nextRequestId);
const { promise, resolve } = Promise.withResolvers<boolean>();
this.#pending.set(id, { kind: "init", model, resolve });
try {
worker.send({ type: "init", id, model, cacheDir });
const ok = await promise;
if (!ok) return null;
} finally {
this.#pending.delete(id);
}
} catch (error) {
logger.debug("mnemopi-embed: init failed", {
model,
error: error instanceof Error ? error.message : String(error),
});
return null;
}
return { embed: (texts, batchSize) => this.#streamEmbed(model, cacheDir, texts, batchSize) };
}
async terminate(): Promise<void> {
const worker = this.#worker;
this.#worker = null;
this.#unsubscribeMessage?.();
this.#unsubscribeMessage = null;
this.#unsubscribeError?.();
this.#unsubscribeError = null;
for (const pending of this.#pending.values()) {
if (pending.kind === "init") pending.resolve(false);
else pending.resolve(new Error("mnemopi embed worker terminated"));
}
this.#pending.clear();
try {
await worker?.terminate();
} catch {
// Already gone.
}
}
async #embed(
model: MnemopiEmbedModelId,
cacheDir: string | undefined,
texts: string[],
batchSize: number | undefined,
): Promise<number[][]> {
const worker = this.#ensureWorker();
const id = String(++this.#nextRequestId);
const { promise, resolve } = Promise.withResolvers<number[][] | Error>();
this.#pending.set(id, { kind: "embed", model, resolve });
try {
// Carry the (model, cacheDir) the wrapper was bound to in every
// embed message: dispose + respawn between two embeds on the same
// `LocalEmbeddingModel` handle would otherwise hit a fresh
// worker's "embed before init" guard. Worker `ensureLoaded` is
// idempotent so steady-state embeds pay no extra cost.
worker.send({ type: "embed", id, model, cacheDir, texts, batchSize });
const result = await promise;
if (result instanceof Error) throw result;
return result;
} finally {
this.#pending.delete(id);
}
}
async *#streamEmbed(
model: MnemopiEmbedModelId,
cacheDir: string | undefined,
texts: string[],
batchSize: number | undefined,
): AsyncIterable<number[][]> {
const vectors = await this.#embed(model, cacheDir, texts, batchSize);
// Mnemopi's `collectMatrix` re-batches via async iteration anyway; yield
// a single batch carrying the full result so the caller's drain loop
// behaves identically to the in-process fastembed iterator (one yield
// per `embed()` call) without paying extra IPC round-trips.
yield vectors;
}
#ensureWorker(): MnemopiEmbedWorkerHandle {
if (this.#worker) return this.#worker;
const worker = this.#spawnWorker();
this.#worker = worker;
this.#unsubscribeMessage = worker.onMessage(message => this.#handleMessage(message));
this.#unsubscribeError = worker.onError(error => this.#handleWorkerError(error));
return worker;
}
#handleMessage(message: MnemopiEmbedWorkerOutbound): void {
if (message.type === "log") {
logWorkerMessage(message);
return;
}
if (message.type === "pong") return;
const pending = this.#pending.get(message.id);
if (!pending) return;
this.#pending.delete(message.id);
if (message.type === "ready") {
if (pending.kind === "init") pending.resolve(true);
return;
}
if (message.type === "vectors") {
if (pending.kind === "embed") pending.resolve(message.vectors);
return;
}
logger.debug("mnemopi-embed: worker returned error", { error: message.error });
if (pending.kind === "init") pending.resolve(false);
else pending.resolve(new Error(message.error));
}
#handleWorkerError(error: Error): void {
logger.warn("mnemopi-embed: worker error", { error: error.message });
for (const pending of this.#pending.values()) {
if (pending.kind === "init") pending.resolve(false);
else pending.resolve(error);
}
this.#pending.clear();
void this.terminate();
}
}
export const mnemopiEmbedClient = new MnemopiEmbedClient();
export async function shutdownMnemopiEmbedClient(): Promise<void> {
await mnemopiEmbedClient.terminate();
}
export async function smokeTestMnemopiEmbedWorker({
timeoutMs = SMOKE_TEST_TIMEOUT_MS,
}: {
timeoutMs?: number;
} = {}): Promise<void> {
const handle = wrapSubprocess(createMnemopiEmbedSubprocess());
const { promise, resolve, reject } = Promise.withResolvers<void>();
const timer = setTimeout(
() => reject(new Error(`mnemopi embed worker did not pong within ${timeoutMs}ms`)),
timeoutMs,
);
const unsubscribeMessage = handle.onMessage(message => {
if (message.type === "pong") {
resolve();
return;
}
if (message.type === "log") return;
reject(new Error(`mnemopi embed worker: expected pong, got ${JSON.stringify(message)}`));
});
const unsubscribeError = handle.onError(reject);
try {
handle.send({ type: "ping", id: "smoke" } satisfies MnemopiEmbedWorkerInbound);
await promise;
} finally {
clearTimeout(timer);
unsubscribeMessage();
unsubscribeError();
await handle.terminate();
}
}
@@ -0,0 +1,35 @@
/**
* Wire types between the parent (`MnemopiEmbedClient`) and the local
* embeddings subprocess. The parent owns the subprocess lifecycle (graceful
* work, hard `SIGKILL` on shutdown); the protocol carries no explicit close
* handshake — once the parent decides to terminate, it signals the OS to reap
* the child so `onnxruntime-node`'s NAPI finalizer never runs in the main
* agent address space (it crashes Bun on Windows shutdown — issue #3031, the
* mnemopi sibling of the tiny-model fix from #1606/#1607). See
* `embed-client.ts` for the spawn/kill glue.
*/
/** Identifier of the fastembed model the worker should load (e.g. `fast-bge-base-en-v1.5`). */
export type MnemopiEmbedModelId = string;
export type MnemopiEmbedWorkerInbound =
| { type: "ping"; id: string }
| { type: "init"; id: string; model: MnemopiEmbedModelId; cacheDir?: string }
// `embed` always carries the same `model` / `cacheDir` the wrapper was
// initialized with so a fresh subprocess (after the parent SIGKILLed the
// previous one but mnemopi still holds the cached `LocalEmbeddingModel`)
// can lazily reload the model on demand instead of returning
// "embed before init".
| { type: "embed"; id: string; model: MnemopiEmbedModelId; cacheDir?: string; texts: string[]; batchSize?: number };
export type MnemopiEmbedWorkerOutbound =
| { type: "pong"; id: string }
| { type: "ready"; id: string }
| { type: "vectors"; id: string; vectors: number[][] }
| { type: "error"; id: string; error: string }
| { type: "log"; level: "debug" | "warn" | "error"; msg: string; meta?: Record<string, unknown> };
export interface MnemopiEmbedTransport {
send(message: MnemopiEmbedWorkerOutbound): void;
onMessage(handler: (message: MnemopiEmbedWorkerInbound) => void): () => void;
}
@@ -0,0 +1,113 @@
/**
* Mnemopi local-embeddings worker. Loaded inside the dedicated subprocess
* spawned by `embed-client.ts` (re-entered through the agent CLI's hidden
* `__omp_worker_mnemopi_embed` selector). The whole point of this module is
* that `loadFastembed()` — and therefore `onnxruntime-node`'s NAPI
* constructor + finalizer — only ever runs in this child address space. The
* parent `SIGKILL`s us on shutdown so the destructor that crashes Bun on
* Windows shutdown (issue #3031, mnemopi sibling of #1606/#1607) never runs
* in either process.
*/
import type { StandardEmbeddingModel } from "@oh-my-pi/pi-mnemopi/core";
import { loadFastembed } from "@oh-my-pi/pi-mnemopi/core/fastembed-runtime";
import type { MnemopiEmbedModelId, MnemopiEmbedTransport, MnemopiEmbedWorkerInbound } from "./embed-protocol";
interface LoadedModel {
model: MnemopiEmbedModelId;
cacheDir: string | undefined;
instance: {
embed(texts: string[], batchSize?: number): AsyncIterable<number[][]> | Iterable<number[][]>;
};
}
let loaded: Promise<LoadedModel> | null = null;
let loadedKey = "";
async function loadModel(model: MnemopiEmbedModelId, cacheDir: string | undefined): Promise<LoadedModel> {
const { FlagEmbedding } = await loadFastembed();
// Cast: `model` arrives as a string from the parent (resolved by
// mnemopi's `fastembedModelName`). Cast to the non-CUSTOM overload's
// argument so TypeScript picks the standard-model branch — the parent
// only ever passes pre-vetted fast-* identifiers.
const instance = await FlagEmbedding.init({
model: model as StandardEmbeddingModel,
cacheDir,
showDownloadProgress: false,
});
return { model, cacheDir, instance };
}
function ensureLoaded(model: MnemopiEmbedModelId, cacheDir: string | undefined): Promise<LoadedModel> {
const key = `${model}\u0000${cacheDir ?? ""}`;
if (loaded !== null && loadedKey === key) return loaded;
const loading = loadModel(model, cacheDir).catch(error => {
// Failed loads must not poison the cache — a retry with the same key
// should re-attempt the load.
if (loaded === loading) {
loaded = null;
loadedKey = "";
}
throw error;
});
loaded = loading;
loadedKey = key;
return loading;
}
async function handleEmbed(
transport: MnemopiEmbedTransport,
message: Extract<MnemopiEmbedWorkerInbound, { type: "embed" }>,
): Promise<void> {
try {
// Each `embed` carries the model + cacheDir the wrapper was bound to.
// `ensureLoaded` is idempotent for the same key, so this is a no-op
// once the model is in memory — and it transparently re-loads after
// the parent SIGKILLed the previous subprocess but mnemopi still
// holds the cached `LocalEmbeddingModel` wrapper from before.
const { instance } = await ensureLoaded(message.model, message.cacheDir);
const vectors: number[][] = [];
const batches = instance.embed([...message.texts], message.batchSize);
for await (const batch of batches) {
for (const row of batch) vectors.push(row);
}
transport.send({ type: "vectors", id: message.id, vectors });
} catch (error) {
transport.send({
type: "error",
id: message.id,
error: error instanceof Error ? error.message : String(error),
});
}
}
async function handleInit(
transport: MnemopiEmbedTransport,
message: Extract<MnemopiEmbedWorkerInbound, { type: "init" }>,
): Promise<void> {
try {
await ensureLoaded(message.model, message.cacheDir);
transport.send({ type: "ready", id: message.id });
} catch (error) {
transport.send({
type: "error",
id: message.id,
error: error instanceof Error ? error.message : String(error),
});
}
}
export function startMnemopiEmbedWorker(transport: MnemopiEmbedTransport): void {
transport.onMessage(message => {
switch (message.type) {
case "ping":
transport.send({ type: "pong", id: message.id });
return;
case "init":
void handleInit(transport, message);
return;
case "embed":
void handleEmbed(transport, message);
return;
}
});
}
+29 -1
View File
@@ -3,6 +3,7 @@ import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
import type * as MnemopiNs from "@oh-my-pi/pi-mnemopi";
import type { Mnemopi, RecallResult } from "@oh-my-pi/pi-mnemopi";
import type * as MnemopiCoreNs from "@oh-my-pi/pi-mnemopi/core";
import type { LocalModelInitializer } from "@oh-my-pi/pi-mnemopi/core";
import { logger } from "@oh-my-pi/pi-utils";
import {
composeRecallQuery,
@@ -13,16 +14,42 @@ import {
import { extractMessages } from "../hindsight/transcript";
import type { AgentSession, AgentSessionEvent } from "../session/agent-session";
import type { MnemopiBackendConfig, MnemopiScoping } from "./config";
import { mnemopiEmbedClient } from "./embed-client";
// The mnemopi package pulls the embeddings stack; keep it off the CLI startup
// module graph by loading it lazily at the async boundaries that need it.
let mnemopiMod: typeof MnemopiNs | undefined;
let mnemopiCoreMod: typeof MnemopiCoreNs | undefined;
/** Lazily load `@oh-my-pi/pi-mnemopi` (memoized). */
// `setLocalModelInitializer` writes a single module-level slot shared by
// both the root and `/core` re-exports, so install at most once across both
// loaders. Either entry point is enough to wire up the override.
let localModelInitializerInstalled = false;
function installLocalModelInitializer(setInitializer: (initializer: LocalModelInitializer) => void): void {
if (localModelInitializerInstalled) return;
localModelInitializerInstalled = true;
setInitializer(({ model, cacheDir }) =>
mnemopiEmbedClient.initialize(model, cacheDir).then(handle => {
if (handle) return handle;
throw new Error("mnemopi embed subprocess unavailable");
}),
);
}
/**
* Lazily load `@oh-my-pi/pi-mnemopi` (memoized) and route fastembed loads
* through the dedicated embeddings subprocess. The override is installed once
* — before any consumer gets the chance to call `embed()` — so
* `onnxruntime-node`'s NAPI constructor + finalizer never run inside the
* agent's address space (issue #3031). Test seams that swap the initializer
* with `setLocalModelInitializerForTests` still win because both go through
* the same module-level slot.
*/
export async function loadMnemopi(): Promise<typeof MnemopiNs> {
if (!mnemopiMod) {
mnemopiMod = await import("@oh-my-pi/pi-mnemopi");
installLocalModelInitializer(mnemopiMod.setLocalModelInitializer);
}
return mnemopiMod;
}
@@ -31,6 +58,7 @@ export async function loadMnemopi(): Promise<typeof MnemopiNs> {
export async function loadMnemopiCore(): Promise<typeof MnemopiCoreNs> {
if (!mnemopiCoreMod) {
mnemopiCoreMod = await import("@oh-my-pi/pi-mnemopi/core");
installLocalModelInitializer(mnemopiCoreMod.setLocalModelInitializer);
}
return mnemopiCoreMod;
}
@@ -177,7 +177,7 @@ export class CustomEditor extends Editor {
/** Per-render scratch flag: did any layout line in this render contain a magic
* keyword that should shimmer? Reset by {@link #scheduleShimmerIfNeeded} each
* time a frame is queued. */
#shimmerTimer: ReturnType<typeof setTimeout> | undefined;
#shimmerTimer: Timer | undefined;
/** Repaint hook the host wires once at construction. Called from the shimmer
* timer to request the next animation frame. Undefined when nobody is
* listening (tests, headless callers); the timer chain still self-cleans. */
@@ -179,9 +179,9 @@ export class ModelSelectorComponent extends Container {
#providers: ProviderTabState[] = STATIC_PROVIDER_TABS;
#activeTabIndex: number = 0;
#refreshingProviders: Set<string> = new Set();
#scheduledProviderRefreshes: Map<string, ReturnType<typeof setTimeout>> = new Map();
#scheduledProviderRefreshes: Map<string, Timer> = new Map();
#refreshSpinnerFrame: number = 0;
#refreshSpinnerInterval?: NodeJS.Timeout;
#refreshSpinnerInterval?: Timer;
// Context menu state
#isMenuOpen: boolean = false;
@@ -142,7 +142,7 @@ export interface LspServerInfo {
*/
export class WelcomeComponent implements Component {
#animStart: number | null = null;
#animTimer: ReturnType<typeof setInterval> | null = null;
#animTimer: Timer | null = null;
#selectedTip: string | undefined;
// Render cache: the welcome box is the first transcript-area component, so
// returning a stable array reference keeps the whole frame prefix stable.
@@ -1,4 +1,5 @@
import type { ImageContent } from "@oh-my-pi/pi-ai";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { type Component, Loader, TERMINAL } from "@oh-my-pi/pi-tui";
import { INTENT_FIELD } from "@oh-my-pi/pi-wire";
import { extractTextContent } from "../../commit/utils";
@@ -186,7 +187,7 @@ export class EventController {
}
#updateWorkingMessageFromIntent(intent: unknown): void {
if (this.ctx.session.isAborting) return;
// Streamed JSON can deliver non-string `_i` (object, number, boolean) before
// Streamed JSON can deliver non-string `i` (object, number, boolean) before
// schema validation; `?.` only guards null/undefined, so guard the type too.
if (typeof intent !== "string") return;
const trimmed = intent.trim();
@@ -1014,6 +1015,13 @@ export class EventController {
async #handleAutoRetryStart(event: Extract<AgentSessionEvent, { type: "auto_retry_start" }>): Promise<void> {
this.#stopWorkingLoader();
this.ctx.statusContainer.clear();
if (event.errorMessage?.includes(THINKING_LOOP_ERROR_MARKER)) {
// The retry path drops the failed assistant from runtime context. Do not
// restore its inline Error row; just unpin the fixed-region banner so the
// retry UI is the visible state.
this.#pinnedErrorComponent = undefined;
this.ctx.clearPinnedError();
}
const delaySeconds = Math.round(event.delayMs / 1000);
this.ctx.retryLoader = new Loader(
this.ctx.ui,
@@ -27,7 +27,7 @@ import {
theme,
} from "../../modes/theme/theme";
import type { InteractiveModeContext } from "../../modes/types";
import type { ResetCreditRedeemOutcome } from "../../session/auth-storage";
import type { ResetCreditAccountStatus, ResetCreditRedeemOutcome } from "../../session/auth-storage";
import type { SessionInfo } from "../../session/session-listing";
import { SessionManager } from "../../session/session-manager";
import { FileSessionStorage } from "../../session/session-storage";
@@ -1161,7 +1161,7 @@ export class SelectorController {
async showResetUsageSelector(): Promise<void> {
const session = this.ctx.session;
this.ctx.showStatus("Checking saved rate-limit resets…", { dim: true });
let statuses: Awaited<ReturnType<typeof session.listResetCredits>>;
let statuses: ResetCreditAccountStatus[];
try {
statuses = await session.listResetCredits();
} catch (error) {
@@ -2743,6 +2743,35 @@ export function highlightCode(code: string, lang?: string, highlightTheme: Theme
}
export function getSymbolTheme(): SymbolTheme {
// Guard against `theme` being undefined (pre-init or cross-module-instance
// plugin calls). Fall back to the ASCII preset so the returned symbols are
// usable instead of crashing. See #2998.
if (typeof theme === "undefined") {
const box = {
topLeft: "+",
topRight: "+",
bottomLeft: "+",
bottomRight: "+",
horizontal: "-",
vertical: "|",
cross: "+",
teeDown: "+",
teeUp: "+",
teeLeft: "+",
teeRight: "+",
};
return {
cursor: ">",
inputCursor: "|",
boxRound: box,
boxSharp: box,
table: box,
quoteBorder: "|",
hrChar: "-",
colorSwatch: "[]",
spinnerFrames: ["-", "\\", "|", "/"],
};
}
const preset = theme.getSymbolPreset();
return {
@@ -2808,6 +2837,19 @@ export function getMarkdownTheme(): MarkdownTheme {
}
export function getSelectListTheme(): SelectListTheme {
// Guard against `theme` being undefined (pre-init or cross-module-instance
// plugin calls). See #2998.
if (typeof theme === "undefined") {
return {
selectedPrefix: (text: string) => text,
selectedText: (text: string) => text,
description: (text: string) => text,
scrollInfo: (text: string) => text,
noMatch: (text: string) => text,
symbols: getSymbolTheme(),
hovered: (text: string) => text,
};
}
return {
selectedPrefix: (text: string) => theme.fg("accent", text),
selectedText: (text: string) => theme.fg("accent", text),
@@ -2820,6 +2862,16 @@ export function getSelectListTheme(): SelectListTheme {
}
export function getEditorTheme(): EditorTheme {
// Guard against `theme` being undefined (pre-init or cross-module-instance
// plugin calls). See #2998.
if (typeof theme === "undefined") {
return {
borderColor: (text: string) => text,
selectList: getSelectListTheme(),
symbols: getSymbolTheme(),
hintStyle: (text: string) => text,
};
}
return {
borderColor: (text: string) => theme.fg("borderMuted", text),
selectList: getSelectListTheme(),
@@ -2829,6 +2881,23 @@ export function getEditorTheme(): EditorTheme {
}
export function getSettingsListTheme(): SettingsListTheme {
// Plugins (e.g. pi-rtk-optimizer) may call this before `initTheme()` assigns
// the global `theme`, or from a separate module instance under npm-global
// installs where the live binding was never initialized. Fall back to plain
// text so the call returns a usable (unstyled) theme instead of crashing with
// "undefined is not an object (evaluating 'theme.fg')". See #2998.
if (typeof theme === "undefined") {
return {
label: (text: string) => text,
value: (text: string) => text,
description: (text: string) => text,
cursor: "> ",
hint: (text: string) => text,
heading: (text: string) => text,
section: (text: string) => text,
hovered: (text: string) => text,
};
}
return {
label: (text: string, selected: boolean, changed: boolean) =>
changed ? theme.fg("statusLineGitDirty", text) : selected ? theme.fg("accent", text) : text,
+4
View File
@@ -1053,6 +1053,9 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
const modelRegistry =
options.modelRegistry ??
new ModelRegistry(options.authStorage ?? (await logger.time("discoverModels", discoverAuthStorage, agentDir)));
// Track whether we internally created the authStorage so we can close it
// if construction fails before the session takes ownership.
const ownsAuthStorage = !options.authStorage && !options.modelRegistry;
const authStorage = modelRegistry.authStorage;
if (options.authStorage && options.authStorage !== authStorage) {
throw new Error(
@@ -2870,6 +2873,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
await asyncJobManager.dispose({ timeoutMs: 3_000 });
}
await disposeKernelSessionsByOwner(evalKernelOwnerId);
if (ownsAuthStorage) authStorage.close();
}
} catch (cleanupError) {
logger.warn("Failed to clean up createAgentSession resources after startup error", {
@@ -104,6 +104,7 @@ import {
streamSimple,
} from "@oh-my-pi/pi-ai";
import { stripToolDescriptions } from "@oh-my-pi/pi-ai/utils/schema";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { getSupportedEfforts } from "@oh-my-pi/pi-catalog/model-thinking";
import { modelsAreEqual } from "@oh-my-pi/pi-catalog/models";
import { MacOSPowerAssertion } from "@oh-my-pi/pi-natives";
@@ -205,6 +206,7 @@ import type { HindsightSessionState } from "../hindsight/state";
import { type LocalProtocolOptions, resolveLocalUrlToPath } from "../internal-urls";
import { IrcBus, type IrcMessage } from "../irc/bus";
import { resolveMemoryBackend } from "../memory-backend";
import { shutdownMnemopiEmbedClient } from "../mnemopi/embed-client";
import { getMnemopiSessionState, type MnemopiSessionState, setMnemopiSessionState } from "../mnemopi/state";
import { containsOrchestrate, ORCHESTRATE_NOTICE } from "../modes/orchestrate";
import { getCurrentThemeName, theme } from "../modes/theme/theme";
@@ -4209,6 +4211,11 @@ export class AgentSession {
hindsightState?.dispose();
const mnemopiState = setMnemopiSessionState(this, undefined);
await mnemopiState?.dispose();
// Tear down the embeddings subprocess AFTER mnemopi state.dispose:
// consolidate-on-dispose may still call `embed()` to store the final
// memories, and that round-trips through the worker we are about to
// hard-kill (issue #3031).
await shutdownMnemopiEmbedClient();
this.#disconnectFromAgent();
if (this.#unsubscribeAppendOnly) {
this.#unsubscribeAppendOnly();
@@ -9983,6 +9990,7 @@ export class AgentSession {
if (this.#isProviderErrorFinishReasonBeforeToolUse(message)) return true;
if (this.#isMalformedFunctionCallError(message)) return true;
if (this.#hasReplayUnsafeToolOutput(message)) return false;
if (message.errorMessage.includes(THINKING_LOOP_ERROR_MARKER)) return true;
if (this.#isStaleOpenAIResponsesReplayError(message)) return true;
const err = message.errorMessage;
@@ -247,6 +247,20 @@ FROM model_usage_legacy
{ cause: lastError },
);
}
/** @internal Reset all singletons and close their databases — test-only. */
static resetInstance(): void {
for (const storage of instances.values()) storage.#close();
instances.clear();
}
#close(): void {
this.#listSettingsStmt.finalize();
this.#upsertModelUsageStmt.finalize();
this.#listModelUsageStmt.finalize();
// SqliteAuthCredentialStore.close() finalizes its own statements and
// closes the shared #db handle — must run after our statements finalize.
this.#authStore.close();
}
/**
* Reads legacy settings persisted in the agent.db `settings` table.
@@ -30,6 +30,7 @@ import {
} from "@oh-my-pi/pi-ai/auth-broker/discover";
import { getAgentDir } from "@oh-my-pi/pi-utils";
import { resolveConfigValue } from "../config/resolve-config-value";
import type { AuthStorage } from "./auth-storage";
export { type AuthBrokerClientConfig, getAuthBrokerTokenFilePath };
@@ -82,7 +83,7 @@ export function resolveAuthBrokerConfig(): Promise<AuthBrokerClientConfig | null
export function discoverAuthStorage(
agentDir: string = getAgentDir(),
options?: Omit<DiscoverAuthStorageOptions, "agentDir" | "configValueResolver">,
): ReturnType<typeof discoverAuthStorageShared> {
): Promise<AuthStorage> {
return discoverAuthStorageShared({
...options,
agentDir,
@@ -145,9 +145,21 @@ CREATE TRIGGER IF NOT EXISTS history_ai AFTER INSERT ON history BEGIN
return HistoryStorage.#instance;
}
/** @internal Reset the singleton — test-only. */
/** @internal Reset the singleton and close its database — test-only. */
static resetInstance(): void {
const instance = HistoryStorage.#instance;
HistoryStorage.#instance = undefined;
if (instance) instance.#close();
}
#close(): void {
for (const stmt of this.#substringStmts.values()) stmt.finalize();
this.#substringStmts.clear();
this.#insertRowStmt.finalize();
this.#recentStmt.finalize();
this.#searchStmt.finalize();
this.#lastPromptStmt.finalize();
this.#db.close();
}
#insertBatch(rows: Array<Pick<HistoryEntry, "prompt" | "cwd" | "sessionId">>): void {
@@ -102,7 +102,7 @@ function primaryArg(name: string, args: Record<string, unknown> | undefined): st
rest[key] = value;
restCount++;
}
if (restCount === 0) return "";
if (restCount === 0) return "{}";
try {
return oneLine(JSON.stringify(rest));
} catch {
+2 -7
View File
@@ -3,6 +3,7 @@ import { $env, isBunTestRuntime, isCompiledBinary, logger, workerHostEntry } fro
import type { Subprocess } from "bun";
import { settings } from "../config/settings";
import { tinyWorkerEnvOverlay } from "../tiny/title-client";
import { safeSend } from "../utils/ipc";
import type { SttProgressEvent, SttWorkerInbound, SttWorkerOutbound } from "./asr-protocol";
import type { SttModelKey } from "./models";
@@ -181,13 +182,7 @@ export function createSttSubprocess(): SpawnedSubprocess {
function wrapSubprocess({ proc, inbound, errors, intentionalExit }: SpawnedSubprocess): WorkerHandle {
return {
send(message) {
try {
proc.send(message);
} catch (error) {
logger.debug("stt: send to subprocess failed", {
error: error instanceof Error ? error.message : String(error),
});
}
safeSend(proc, message, "stt");
},
onMessage(handler) {
inbound.add(handler);
@@ -2,6 +2,7 @@ import * as path from "node:path";
import { $env, isBunTestRuntime, isCompiledBinary, logger, workerHostEntry } from "@oh-my-pi/pi-utils";
import type { Subprocess } from "bun";
import { settings } from "../config/settings";
import { safeSend } from "../utils/ipc";
import { tinyModelDeviceSettingToEnv } from "./device";
import { tinyModelDtypeSettingToEnv } from "./dtype";
import {
@@ -216,13 +217,7 @@ export function createTinyTitleSubprocess(): SpawnedSubprocess {
function wrapSubprocess({ proc, inbound, errors, intentionalExit }: SpawnedSubprocess): WorkerHandle {
return {
send(message) {
try {
proc.send(message);
} catch (error) {
logger.debug("tiny-title: send to subprocess failed", {
error: error instanceof Error ? error.message : String(error),
});
}
safeSend(proc, message, "tiny-title");
},
onMessage(handler) {
inbound.add(handler);
+4 -8
View File
@@ -1572,19 +1572,15 @@ export const imageGenTool: CustomTool<typeof imageGenSchema, ImageGenToolDetails
};
export async function getImageGenTools(
modelRegistry?: ModelRegistry,
activeModel?: Model,
_modelRegistry?: ModelRegistry,
_activeModel?: Model,
): Promise<Array<CustomTool<typeof imageGenSchema, ImageGenToolDetails>>> {
const apiKey = await findImageApiKey(modelRegistry, activeModel);
if (!apiKey) return [];
return [imageGenTool];
}
export async function getImageGenToolsWithRegistry(
modelRegistry: ModelRegistry,
activeModel?: Model,
_modelRegistry: ModelRegistry,
_activeModel?: Model,
): Promise<Array<CustomTool<typeof imageGenSchema, ImageGenToolDetails>>> {
const apiKey = await findImageApiKey(modelRegistry, activeModel);
if (!apiKey) return [];
return [imageGenTool];
}
@@ -657,7 +657,10 @@ export function truncateDiffByHunk(
export function shortenPath(filePath: string, homeDir?: string): string {
const home = homeDir ?? os.homedir();
if (home && filePath.startsWith(home)) {
return `~${filePath.slice(home.length)}`;
const suffix = filePath.slice(home.length);
if (suffix === "" || suffix.startsWith(path.posix.sep) || suffix.startsWith(path.win32.sep)) {
return `~${suffix.replaceAll(path.win32.sep, path.posix.sep)}`;
}
}
return filePath;
}
+5 -5
View File
@@ -70,7 +70,7 @@ const searchSchema = type({
.describe(
'file, directory, glob, internal URL, or array of those to search; append `:<lines>` to scope a file to specific line ranges. Omitted or empty -> searches the workspace root (".")',
),
"i?": type("boolean").describe("case-insensitive search"),
"case?": type("boolean").describe("case-sensitive search"),
"gitignore?": type("boolean").describe("respect gitignore"),
"skip?": type("number")
.or("null")
@@ -680,7 +680,7 @@ export class SearchTool implements AgentTool<typeof searchSchema, SearchToolDeta
_onUpdate?: AgentToolUpdateCallback<SearchToolDetails>,
_toolContext?: AgentToolContext,
): Promise<AgentToolResult<SearchToolDetails>> {
const { pattern, paths: rawPaths, i, gitignore, skip } = params;
const { pattern, paths: rawPaths, case: caseSensitive, gitignore, skip } = params;
return untilAborted(signal, async () => {
// Preserve the pattern verbatim — leading/trailing whitespace is
@@ -763,7 +763,7 @@ export class SearchTool implements AgentTool<typeof searchSchema, SearchToolDeta
}
const normalizedContextBefore = this.session.settings.get("search.contextBefore");
const normalizedContextAfter = this.session.settings.get("search.contextAfter");
const ignoreCase = i ?? false;
const ignoreCase = !(caseSensitive ?? true);
const useGitignore = gitignore ?? true;
const patternHasNewline = normalizedPattern.includes("\n") || normalizedPattern.includes("\\n");
const effectiveMultiline = patternHasNewline;
@@ -1272,7 +1272,7 @@ export class SearchTool implements AgentTool<typeof searchSchema, SearchToolDeta
interface SearchRenderArgs {
pattern: string;
paths?: string | string[];
i?: boolean;
case?: boolean;
gitignore?: boolean;
skip?: number;
}
@@ -1443,7 +1443,7 @@ export const searchToolRenderer = {
const paths = toPathList(args.paths);
const meta: string[] = [];
if (paths.length) meta.push(`in ${paths.join(", ")}`);
if (args.i) meta.push("case:insensitive");
if (args.case === false) meta.push("case:insensitive");
if (args.gitignore === false) meta.push("gitignore:false");
if (args.skip !== undefined && args.skip > 0) meta.push(`skip:${args.skip}`);
+2 -7
View File
@@ -3,6 +3,7 @@ import { $env, isBunTestRuntime, isCompiledBinary, logger, workerHostEntry } fro
import type { Subprocess } from "bun";
import { settings } from "../config/settings";
import { tinyWorkerEnvOverlay } from "../tiny/title-client";
import { safeSend } from "../utils/ipc";
import { isTtsLocalModelKey, type TtsLocalModelKey } from "./models";
import type { TtsProgressEvent, TtsWorkerInbound, TtsWorkerOutbound } from "./tts-protocol";
@@ -245,13 +246,7 @@ export function createTtsSubprocess(): SpawnedSubprocess {
function wrapSubprocess({ proc, inbound, errors, intentionalExit }: SpawnedSubprocess): WorkerHandle {
return {
send(message) {
try {
proc.send(message);
} catch (error) {
logger.debug("tts: send to subprocess failed", {
error: error instanceof Error ? error.message : String(error),
});
}
safeSend(proc, message, "tts");
},
onMessage(handler) {
inbound.add(handler);
@@ -13,9 +13,19 @@ export const SUPPORTED_INPUT_IMAGE_MIME_TYPES = SUPPORTED_IMAGE_MIME_TYPES;
* with an opaque HTTP 400. Detect those models so the resize pipeline encodes
* to PNG/JPEG instead — the automatic equivalent of `OMP_NO_WEBP=1`.
*/
export function modelLacksWebpSupport(model: Pick<Model, "provider" | "api"> | undefined): boolean {
export function modelLacksWebpSupport(
model: Pick<Model, "provider" | "api" | "imageInputDecoder"> | undefined,
): boolean {
if (!model) return false;
return model.provider === "ollama" || model.provider === "ollama-cloud" || model.api === "ollama-chat";
return (
model.imageInputDecoder === "stb" ||
model.provider === "ollama" ||
model.provider === "ollama-cloud" ||
model.provider === "llama.cpp" ||
model.provider === "lm-studio" ||
model.provider === "local-server" ||
model.api === "ollama-chat"
);
}
/**
+38
View File
@@ -0,0 +1,38 @@
import { logger } from "@oh-my-pi/pi-utils";
/**
* Narrow a value to a thenable so a rejection handler can be attached.
*
* Mirrors the local helper in `mcp/transports/stdio.ts` (kept separate because
* that copy serves the FileSink stdin-write path and is battle-tested there).
* This shared copy is the home for the IPC `send()` sites.
*/
export function isThenable(value: unknown): value is PromiseLike<unknown> {
return (
value != null &&
(typeof value === "object" || typeof value === "function") &&
typeof (value as { then?: unknown }).then === "function"
);
}
/**
* Send a message to a Bun subprocess over IPC, neutralizing both the
* synchronous throw ("cannot be used after the process has exited") and any
* asynchronous rejection (EPIPE from a pipe that broke between exit being
* observed and the next `send()`). The dead worker is detected separately via
* `onExit`/`onError` and respawned or disabled by the owning client; an
* un-awaited EPIPE rejection must not escape as a fatal unhandled rejection
* that takes down the whole session. See issue #2997.
*
* `label` prefixes the debug log on synchronous failure (e.g. "tts").
*/
export function safeSend(proc: { send(message: unknown): unknown }, message: unknown, label: string): void {
try {
const result = proc.send(message);
if (isThenable(result)) result.then(undefined, () => {});
} catch (error) {
logger.debug(`${label}: send to subprocess failed`, {
error: error instanceof Error ? error.message : String(error),
});
}
}
@@ -0,0 +1,133 @@
import type { AuthStorage, OAuthAccess } from "@oh-my-pi/pi-ai";
import { $env } from "@oh-my-pi/pi-utils";
export const PERPLEXITY_CHAT_BASE_URL = "https://api.perplexity.ai";
export const PERPLEXITY_RESPONSES_BASE_URL = "https://api.perplexity.ai/v1";
export const OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
export const OAUTH_EXPIRY_BUFFER_MS = 5 * 60 * 1000;
export interface ApiConfig {
type: "api_key";
apiKey: string;
provider: "perplexity" | "openrouter";
chatBaseUrl: string;
responsesBaseUrl: string;
modelPrefix: string;
useResponses: boolean;
}
export type PerplexityAuth =
| ApiConfig
| {
type: "oauth";
access: OAuthAccess;
}
| {
type: "cookies";
cookies: string;
}
| {
type: "anonymous";
};
export interface PerplexityAuthOptions {
signal?: AbortSignal;
forceRefresh?: boolean;
}
/** Detect API-key endpoints to try in priority order (Perplexity direct, then OpenRouter). */
export async function getApiConfigs(
authStorage: AuthStorage,
sessionId: string | undefined,
options?: PerplexityAuthOptions,
): Promise<ApiConfig[]> {
const useResponses = $env.PI_PERPLEXITY_RESPONSES === "1";
const configs: ApiConfig[] = [];
const perplexityKey = await authStorage.getApiKey("perplexity", sessionId, options);
if (perplexityKey) {
configs.push({
type: "api_key",
apiKey: perplexityKey,
provider: "perplexity",
chatBaseUrl: PERPLEXITY_CHAT_BASE_URL,
responsesBaseUrl: PERPLEXITY_RESPONSES_BASE_URL,
modelPrefix: "",
useResponses,
});
}
const openrouterKey = await authStorage.getApiKey("openrouter", sessionId, options);
if (openrouterKey) {
configs.push({
type: "api_key",
apiKey: openrouterKey,
provider: "openrouter",
chatBaseUrl: OPENROUTER_BASE_URL,
responsesBaseUrl: OPENROUTER_BASE_URL,
modelPrefix: "perplexity/",
useResponses,
});
}
return configs;
}
/**
* Decode a Perplexity JWT's `exp` claim, in ms. Returns `undefined` when the
* token has no `exp` (which is the common case — Perplexity sessions are
* server-side and effectively non-expiring from the client's POV).
*/
export function jwtExpiryMs(token: string): number | undefined {
const parts = token.split(".");
if (parts.length !== 3) return undefined;
const payload = parts[1];
if (!payload) return undefined;
try {
const decoded = JSON.parse(Buffer.from(payload, "base64url").toString("utf8")) as { exp?: unknown };
if (typeof decoded.exp !== "number" || !Number.isFinite(decoded.exp)) return undefined;
return decoded.exp * 1000;
} catch {
return undefined;
}
}
/** Collect all available auth methods to try in priority order */
export async function getAvailableAuthMethods(
authStorage: AuthStorage,
sessionId: string | undefined,
options?: PerplexityAuthOptions,
): Promise<PerplexityAuth[]> {
const methods: PerplexityAuth[] = [];
// 1. Cookies take precedence over OAuth as noted in comments/docs
const cookies = $env.PERPLEXITY_COOKIES?.trim();
if (cookies) {
methods.push({ type: "cookies", cookies });
}
// 2. Perplexity OAuth (session bearer)
try {
const access = await authStorage.getOAuthAccess("perplexity", sessionId, options);
const token = access?.accessToken;
if (access && token) {
const jwtExpiry = jwtExpiryMs(token);
if (jwtExpiry === undefined || jwtExpiry > Date.now() + OAUTH_EXPIRY_BUFFER_MS) {
methods.push({ type: "oauth", access });
}
}
} catch {
// ignored
}
// 3. API key configs (direct, then openrouter)
const apiConfigs = await getApiConfigs(authStorage, sessionId, options);
methods.push(...apiConfigs);
// 4. Fallback to Perplexity free (anonymous)
if (methods.length === 0) {
methods.push({ type: "anonymous" });
}
return methods;
}
@@ -14,7 +14,6 @@ import {
type AuthStorage,
type Context,
type FetchImpl,
type OAuthAccess,
type Usage,
withOAuthAccess,
} from "@oh-my-pi/pi-ai";
@@ -34,36 +33,19 @@ import { SearchProviderError } from "../../../web/search/types";
import { dateToAgeSeconds } from "../utils";
import type { SearchParams } from "./base";
import { SearchProvider } from "./base";
import { type ApiConfig, getAvailableAuthMethods } from "./perplexity-auth";
import { classifyProviderHttpError, withHardTimeout } from "./utils";
const PERPLEXITY_CHAT_BASE_URL = "https://api.perplexity.ai";
const PERPLEXITY_RESPONSES_BASE_URL = "https://api.perplexity.ai/v1";
const OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1";
const PERPLEXITY_OAUTH_ASK_URL = "https://www.perplexity.ai/rest/sse/perplexity_ask";
const DEFAULT_MAX_TOKENS = 8192;
const DEFAULT_TEMPERATURE = 0.2;
const DEFAULT_NUM_SEARCH_RESULTS = 20;
const OAUTH_EXPIRY_BUFFER_MS = 5 * 60 * 1000;
const OAUTH_API_VERSION = "2.18";
const OAUTH_USER_AGENT = "Perplexity/641 CFNetwork/1568 Darwin/25.2.0";
const ANONYMOUS_USER_AGENT =
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36";
type PerplexityAuth =
| ApiConfig
| {
type: "oauth";
access: OAuthAccess;
}
| {
type: "cookies";
cookies: string;
}
| {
type: "anonymous";
};
interface PerplexityOAuthStreamMarkdownBlock {
answer?: string;
chunks?: string[];
@@ -289,111 +271,6 @@ export interface PerplexitySearchParams {
fetch?: FetchImpl;
}
interface ApiConfig {
type: "api_key";
apiKey: string;
provider: "perplexity" | "openrouter";
chatBaseUrl: string;
responsesBaseUrl: string;
modelPrefix: string;
useResponses: boolean;
}
/** Detect API-key endpoints to try in priority order (Perplexity direct, then OpenRouter). */
async function getApiConfigs(
authStorage: AuthStorage,
sessionId: string | undefined,
signal: AbortSignal | undefined,
): Promise<ApiConfig[]> {
const useResponses = $env.PI_PERPLEXITY_RESPONSES === "1";
const configs: ApiConfig[] = [];
const perplexityKey = await authStorage.getApiKey("perplexity", sessionId, { signal });
if (perplexityKey) {
configs.push({
type: "api_key",
apiKey: perplexityKey,
provider: "perplexity",
chatBaseUrl: PERPLEXITY_CHAT_BASE_URL,
responsesBaseUrl: PERPLEXITY_RESPONSES_BASE_URL,
modelPrefix: "",
useResponses,
});
}
const openrouterKey = await authStorage.getApiKey("openrouter", sessionId, { signal });
if (openrouterKey) {
configs.push({
type: "api_key",
apiKey: openrouterKey,
provider: "openrouter",
chatBaseUrl: OPENROUTER_BASE_URL,
responsesBaseUrl: OPENROUTER_BASE_URL,
modelPrefix: "perplexity/",
useResponses,
});
}
return configs;
}
/**
* Decode a Perplexity JWT's `exp` claim, in ms. Returns `undefined` when the
* token has no `exp` (which is the common case — Perplexity sessions are
* server-side and effectively non-expiring from the client's POV).
*/
function jwtExpiryMs(token: string): number | undefined {
const parts = token.split(".");
if (parts.length !== 3) return undefined;
const payload = parts[1];
if (!payload) return undefined;
try {
const decoded = JSON.parse(Buffer.from(payload, "base64url").toString("utf8")) as { exp?: unknown };
if (typeof decoded.exp !== "number" || !Number.isFinite(decoded.exp)) return undefined;
return decoded.exp * 1000;
} catch {
return undefined;
}
}
/** Collect all available auth methods to try in priority order */
async function getAvailableAuthMethods(
authStorage: AuthStorage,
sessionId: string | undefined,
signal: AbortSignal | undefined,
): Promise<PerplexityAuth[]> {
const methods: PerplexityAuth[] = [];
// 1. Perplexity OAuth & Cookies (same priority - highest)
try {
const access = await authStorage.getOAuthAccess("perplexity", sessionId, { signal });
const token = access?.accessToken;
if (access && token) {
const jwtExpiry = jwtExpiryMs(token);
if (jwtExpiry === undefined || jwtExpiry > Date.now() + OAUTH_EXPIRY_BUFFER_MS) {
methods.push({ type: "oauth", access });
}
}
} catch {
// ignored
}
const cookies = $env.PERPLEXITY_COOKIES?.trim();
if (cookies) {
methods.push({ type: "cookies", cookies });
}
const apiConfigs = await getApiConfigs(authStorage, sessionId, signal);
methods.push(...apiConfigs);
// 5. Fallback to Perplexity free (anonymous)
if (methods.length === 0) {
methods.push({ type: "anonymous" });
}
return methods;
}
interface PerplexityApiStreamMetadata {
id?: string;
model?: string;
@@ -904,7 +781,7 @@ export async function searchPerplexity(params: PerplexitySearchParams): Promise<
request.search_recency_filter = params.search_recency_filter;
}
const authMethods = await getAvailableAuthMethods(params.authStorage, params.sessionId, params.signal);
const authMethods = await getAvailableAuthMethods(params.authStorage, params.sessionId, { signal: params.signal });
let lastError: unknown;
for (const auth of authMethods) {
+16 -8
View File
@@ -20,6 +20,7 @@ import {
import type { Model } from "@oh-my-pi/pi-ai";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { resolveLocalUrlToPath } from "@oh-my-pi/pi-coding-agent/internal-urls";
import {
ACP_BOOTSTRAP_RACE_GUARD_MS,
AcpAgent,
@@ -652,11 +653,14 @@ describe("ACP agent", () => {
const session = harness.findSession(created.sessionId)!;
await harness.agent.setSessionMode({ sessionId: created.sessionId, modeId: "plan" });
const artifactsDir = session.sessionManager.getArtifactsDir();
expect(artifactsDir).not.toBeNull();
// The agent writes to its chosen `local://<slug>-plan.md` and resolves with
// the matching slug — the file is never renamed.
const planPath = path.join(artifactsDir!, "local", "words-counter-plan.md");
const localOptions = {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
};
cleanupRoots.push(resolveLocalUrlToPath("local://", localOptions));
// On Windows, long artifact roots are shortened by the local:// resolver to
// avoid MAX_PATH. Write through the same resolver the ACP handler reads from.
const planPath = resolveLocalUrlToPath("local://words-counter-plan.md", localOptions);
await Bun.write(planPath, "# Words Counter\n\nFile contents.");
const updatesBefore = harness.updates.length;
@@ -722,8 +726,12 @@ describe("ACP agent", () => {
const session = harness.findSession(created.sessionId)!;
await harness.agent.setSessionMode({ sessionId: created.sessionId, modeId: "plan" });
const artifactsDir = session.sessionManager.getArtifactsDir();
const planPath = path.join(artifactsDir!, "local", "PLAN.md");
const localOptions = {
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
getSessionId: () => session.sessionManager.getSessionId(),
};
cleanupRoots.push(resolveLocalUrlToPath("local://", localOptions));
const planPath = resolveLocalUrlToPath("local://PLAN.md", localOptions);
await Bun.write(planPath, "# Words Counter\n\nFile contents.");
const updatesBefore = harness.updates.length;
@@ -737,7 +745,7 @@ describe("ACP agent", () => {
expect(result.content[0]?.text).toMatch(/refinement requested/i);
// Plan file stays put; no rename, no write-access grant.
expect(await Bun.file(planPath).exists()).toBe(true);
expect(await Bun.file(path.join(artifactsDir!, "local", "words-counter.md")).exists()).toBe(false);
expect(await Bun.file(resolveLocalUrlToPath("local://words-counter.md", localOptions)).exists()).toBe(false);
// Plan mode + standing handler stay active so the agent can iterate.
expect(session.planModeState?.enabled).toBe(true);
expect(typeof session.standingResolveHandler).toBe("function");
@@ -13,23 +13,20 @@
*/
import { describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { createAcpSessionFactory } from "@oh-my-pi/pi-coding-agent/main";
import type { CreateAgentSessionOptions, CreateAgentSessionResult } from "@oh-my-pi/pi-coding-agent/sdk";
import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { TempDir } from "@oh-my-pi/pi-utils";
describe("createAcpSessionFactory MCP isolation (issue #1234)", () => {
it("forces enableMCP=false even when baseOptions opts in", async () => {
const tempDir = path.join(os.tmpdir(), `pi-acp-mcp-isolation-${Snowflake.next()}`);
fs.mkdirSync(tempDir, { recursive: true });
const authStorage = await AuthStorage.create(path.join(tempDir, "auth.db"));
const tempDir = TempDir.createSync("@pi-acp-mcp-isolation-");
let authStorage: AuthStorage | undefined;
try {
authStorage = await AuthStorage.create(tempDir.join("auth.db"));
const modelRegistry = new ModelRegistry(authStorage);
const settings = Settings.isolated({});
const fakeSession = {} as AgentSession;
@@ -56,7 +53,7 @@ describe("createAcpSessionFactory MCP isolation (issue #1234)", () => {
const factory = createAcpSessionFactory({
baseOptions: { enableMCP: true } as CreateAgentSessionOptions,
settings,
sessionDir: path.join(tempDir, "sessions"),
sessionDir: tempDir.join("sessions"),
authStorage,
modelRegistry,
parsedArgs: {},
@@ -64,13 +61,17 @@ describe("createAcpSessionFactory MCP isolation (issue #1234)", () => {
createSession,
});
const result = await factory(tempDir);
const result = await factory(tempDir.path());
expect(result).toBe(fakeSession);
expect(captured).toHaveLength(1);
expect(captured[0].enableMCP).toBe(false);
} finally {
authStorage.close();
fs.rmSync(tempDir, { recursive: true, force: true });
try {
authStorage?.close();
} finally {
await Bun.sleep(0);
await tempDir.remove();
}
}
});
});
@@ -1,64 +1,66 @@
import { afterEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { createAgentSession } from "@oh-my-pi/pi-coding-agent/sdk";
import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { TempDir } from "@oh-my-pi/pi-utils";
describe("advisor watchdog prompt discovery", () => {
const tempDirs: string[] = [];
const tempDirs: TempDir[] = [];
afterEach(() => {
afterEach(async () => {
await Bun.sleep(0);
for (const tempDir of tempDirs.splice(0)) {
fs.rmSync(tempDir, { recursive: true, force: true });
await tempDir.remove();
}
});
it("discovers and appends WATCHDOG.md to the advisor prompt", async () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-advisor-watchdog-${Snowflake.next()}-`));
const tempDir = TempDir.createSync("@pi-advisor-watchdog-");
tempDirs.push(tempDir);
const cwd = path.join(tempDir, "project-root");
const cwd = tempDir.join("project-root");
fs.mkdirSync(cwd, { recursive: true });
// Write a WATCHDOG.md file
const watchdogContent = "Watchdog rule: Watch out for cheating on edits.";
fs.writeFileSync(path.join(cwd, "WATCHDOG.md"), watchdogContent, "utf8");
const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
authStorage.setRuntimeApiKey("openai", "test-key");
const modelRegistry = new ModelRegistry(authStorage);
const sessionManager = SessionManager.create(cwd, path.join(tempDir, "sessions"));
const { session } = await createAgentSession({
cwd,
agentDir: tempDir,
sessionManager,
authStorage,
modelRegistry,
settings: (() => {
const s = Settings.isolated({
"async.enabled": false,
"advisor.enabled": true,
});
s.setModelRole("advisor", "openai/gpt-4o-mini");
return s;
})(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
const authStorage = await AuthStorage.create(tempDir.join("testauth.db"));
let session: AgentSession | undefined;
try {
authStorage.setRuntimeApiKey("openai", "test-key");
const modelRegistry = new ModelRegistry(authStorage);
const sessionManager = SessionManager.create(cwd, tempDir.join("sessions"));
const result = await createAgentSession({
cwd,
agentDir: tempDir.path(),
sessionManager,
authStorage,
modelRegistry,
settings: (() => {
const s = Settings.isolated({
"async.enabled": false,
"advisor.enabled": true,
});
s.setModelRole("advisor", "openai/gpt-4o-mini");
return s;
})(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
session = result.session;
expect(session.isAdvisorActive()).toBe(true);
const dump = session.formatAdvisorHistoryAsText();
expect(dump).not.toBeNull();
@@ -67,14 +69,18 @@ describe("advisor watchdog prompt discovery", () => {
expect(dump).toContain(watchdogContent);
expect(dump).toContain("</attention>");
} finally {
await session.dispose();
try {
await session?.dispose();
} finally {
authStorage.close();
}
}
});
it("resolves nested folders and sorts by depth", async () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-advisor-watchdog-${Snowflake.next()}-`));
const tempDir = TempDir.createSync("@pi-advisor-watchdog-");
tempDirs.push(tempDir);
const parentCwd = path.join(tempDir, "project-root");
const parentCwd = tempDir.join("project-root");
const childCwd = path.join(parentCwd, "subfolder");
fs.mkdirSync(childCwd, { recursive: true });
@@ -84,36 +90,37 @@ describe("advisor watchdog prompt discovery", () => {
fs.writeFileSync(path.join(parentCwd, "WATCHDOG.md"), parentWatchdogContent, "utf8");
fs.writeFileSync(path.join(childCwd, "WATCHDOG.md"), childWatchdogContent, "utf8");
const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
authStorage.setRuntimeApiKey("openai", "test-key");
const modelRegistry = new ModelRegistry(authStorage);
const sessionManager = SessionManager.create(childCwd, path.join(tempDir, "sessions"));
const { session } = await createAgentSession({
cwd: childCwd,
agentDir: tempDir,
sessionManager,
authStorage,
modelRegistry,
settings: (() => {
const s = Settings.isolated({
"async.enabled": false,
"advisor.enabled": true,
});
s.setModelRole("advisor", "openai/gpt-4o-mini");
return s;
})(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
const authStorage = await AuthStorage.create(tempDir.join("testauth.db"));
let session: AgentSession | undefined;
try {
authStorage.setRuntimeApiKey("openai", "test-key");
const modelRegistry = new ModelRegistry(authStorage);
const sessionManager = SessionManager.create(childCwd, tempDir.join("sessions"));
const result = await createAgentSession({
cwd: childCwd,
agentDir: tempDir.path(),
sessionManager,
authStorage,
modelRegistry,
settings: (() => {
const s = Settings.isolated({
"async.enabled": false,
"advisor.enabled": true,
});
s.setModelRole("advisor", "openai/gpt-4o-mini");
return s;
})(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
session = result.session;
expect(session.isAdvisorActive()).toBe(true);
const dump = session.formatAdvisorHistoryAsText();
expect(dump).not.toBeNull();
@@ -130,16 +137,20 @@ describe("advisor watchdog prompt discovery", () => {
expect(childIndex).toBeGreaterThan(-1);
expect(parentIndex).toBeLessThan(childIndex);
} finally {
await session.dispose();
try {
await session?.dispose();
} finally {
authStorage.close();
}
}
});
it("discovers user-level and native project-level watchdog files", async () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-advisor-watchdog-${Snowflake.next()}-`));
const tempDir = TempDir.createSync("@pi-advisor-watchdog-");
tempDirs.push(tempDir);
const cwd = path.join(tempDir, "project-root");
const cwd = tempDir.join("project-root");
const ompDir = path.join(cwd, ".omp");
const userAgentDir = path.join(tempDir, "user-agent");
const userAgentDir = tempDir.join("user-agent");
fs.mkdirSync(cwd, { recursive: true });
fs.mkdirSync(ompDir, { recursive: true });
fs.mkdirSync(userAgentDir, { recursive: true });
@@ -152,36 +163,37 @@ describe("advisor watchdog prompt discovery", () => {
fs.writeFileSync(path.join(ompDir, "WATCHDOG.md"), nativeWatchdogContent, "utf8");
fs.writeFileSync(path.join(cwd, "WATCHDOG.md"), standaloneWatchdogContent, "utf8");
const authStorage = await AuthStorage.create(path.join(tempDir, "testauth.db"));
authStorage.setRuntimeApiKey("openai", "test-key");
const modelRegistry = new ModelRegistry(authStorage);
const sessionManager = SessionManager.create(cwd, path.join(tempDir, "sessions"));
const { session } = await createAgentSession({
cwd,
agentDir: userAgentDir,
sessionManager,
authStorage,
modelRegistry,
settings: (() => {
const s = Settings.isolated({
"async.enabled": false,
"advisor.enabled": true,
});
s.setModelRole("advisor", "openai/gpt-4o-mini");
return s;
})(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
const authStorage = await AuthStorage.create(tempDir.join("testauth.db"));
let session: AgentSession | undefined;
try {
authStorage.setRuntimeApiKey("openai", "test-key");
const modelRegistry = new ModelRegistry(authStorage);
const sessionManager = SessionManager.create(cwd, tempDir.join("sessions"));
const result = await createAgentSession({
cwd,
agentDir: userAgentDir,
sessionManager,
authStorage,
modelRegistry,
settings: (() => {
const s = Settings.isolated({
"async.enabled": false,
"advisor.enabled": true,
});
s.setModelRole("advisor", "openai/gpt-4o-mini");
return s;
})(),
model: getBundledModel("openai", "gpt-4o-mini"),
disableExtensionDiscovery: true,
skills: [],
contextFiles: [],
promptTemplates: [],
slashCommands: [],
enableMCP: false,
enableLsp: false,
});
session = result.session;
expect(session.isAdvisorActive()).toBe(true);
const dump = session.formatAdvisorHistoryAsText();
expect(dump).not.toBeNull();
@@ -204,7 +216,11 @@ describe("advisor watchdog prompt discovery", () => {
expect(userIndex).toBeLessThan(nativeIndex);
expect(userIndex).toBeLessThan(standaloneIndex);
} finally {
await session.dispose();
try {
await session?.dispose();
} finally {
authStorage.close();
}
}
});
});
@@ -4,6 +4,7 @@
* focus failure keeps the hub open and surfaces the error as a notice.
*/
import { afterEach, beforeAll, beforeEach, describe, expect, it } from "bun:test";
import * as path from "node:path";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { IrcBus } from "@oh-my-pi/pi-coding-agent/irc/bus";
import { AgentHubOverlayComponent } from "@oh-my-pi/pi-coding-agent/modes/components/agent-hub";
@@ -16,6 +17,7 @@ import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-sessi
import { TempDir } from "@oh-my-pi/pi-utils";
const AGENT_ID = "Worker";
const TEST_CWD = path.resolve("agent-hub-cwd");
function makeHub(focusAgent: (id: string) => Promise<void>) {
const agents = new AgentRegistry();
@@ -89,9 +91,10 @@ describe("Agent hub Enter activation", () => {
it("lists persisted subagent session files after restart", async () => {
using tempDir = TempDir.createSync("@omp-agent-hub-persisted-");
const sessionFile = `${tempDir.path()}/main.jsonl`;
const sessionFile = path.join(tempDir.path(), "main.jsonl");
const workerSessionFile = path.join(tempDir.path(), "main", "Worker.jsonl");
await Bun.write(sessionFile, "");
await Bun.write(`${tempDir.path()}/main/Worker.jsonl`, "");
await Bun.write(workerSessionFile, "");
const agents = new AgentRegistry();
const hub = new AgentHubOverlayComponent({
observers: new SessionObserverRegistry(),
@@ -107,7 +110,7 @@ describe("Agent hub Enter activation", () => {
const rendered = Bun.stripANSI(hub.render(120).join("\n"));
expect(rendered).toContain("Worker");
expect(rendered).toContain("parked");
expect(agents.get("Worker")?.sessionFile).toBe(`${tempDir.path()}/main/Worker.jsonl`);
expect(agents.get("Worker")?.sessionFile).toBe(workerSessionFile);
hub.dispose();
});
@@ -154,7 +157,7 @@ describe("Agent hub Enter activation", () => {
focusResolved.resolve();
},
session: { getToolByName: () => undefined, extensionRunner: undefined },
sessionManager: { getCwd: () => "/tmp", getSessionFile: () => null },
sessionManager: { getCwd: () => TEST_CWD, getSessionFile: () => null },
hideThinkingBlock: false,
};
const controller = new SelectorController(ctx as unknown as InteractiveModeContext);
@@ -203,7 +206,7 @@ describe("Agent hub double-← gating", () => {
collabGuest: { agentRegistry: agents, hubRemote: undefined },
focusAgentSession: async () => {},
session: { getToolByName: () => undefined, extensionRunner: undefined },
sessionManager: { getCwd: () => "/tmp", getSessionFile: () => null },
sessionManager: { getCwd: () => TEST_CWD, getSessionFile: () => null },
hideThinkingBlock: false,
};
const controller = new SelectorController(ctx as unknown as InteractiveModeContext);
@@ -18,9 +18,6 @@
* follow-up stays queued for the next explicit resume rather than auto-running.
*/
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { Agent, type AgentMessage } from "@oh-my-pi/pi-agent-core";
import { createMockModel, type MockModel, type MockResponse } from "@oh-my-pi/pi-ai/providers/mock";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
@@ -31,7 +28,7 @@ import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { USER_INTERRUPT_LABEL } from "@oh-my-pi/pi-coding-agent/session/messages";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { Snowflake, TempDir } from "@oh-my-pi/pi-utils";
const ADVISOR_TYPE = "advisor";
@@ -45,20 +42,23 @@ interface ParkedHarness {
}
describe("AgentSession advisor auto-resume suppression", () => {
let tempDir: string;
let tempDir: TempDir;
let session: AgentSession;
const authStorages: AuthStorage[] = [];
beforeEach(() => {
tempDir = path.join(os.tmpdir(), `pi-advisor-suppress-${Snowflake.next()}`);
fs.mkdirSync(tempDir, { recursive: true });
tempDir = TempDir.createSync("@pi-advisor-suppress-");
});
afterEach(async () => {
// dispose() aborts the agent, cancelling the parked first-turn stream.
await session?.dispose();
for (const authStorage of authStorages.splice(0)) authStorage.close();
fs.rmSync(tempDir, { recursive: true, force: true });
try {
await session?.dispose();
} finally {
for (const authStorage of authStorages.splice(0)) authStorage.close();
await Bun.sleep(0);
await tempDir?.remove();
}
});
/**
@@ -86,10 +86,10 @@ describe("AgentSession advisor auto-resume suppression", () => {
});
const sessionManager = SessionManager.inMemory();
const settings = Settings.isolated({ "compaction.enabled": false });
const authStorage = await AuthStorage.create(path.join(tempDir, `auth-${Snowflake.next()}.db`));
const authStorage = await AuthStorage.create(tempDir.join(`auth-${Snowflake.next()}.db`));
authStorages.push(authStorage);
authStorage.setRuntimeApiKey("anthropic", "test-key");
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir, "models.yml"));
const modelRegistry = new ModelRegistry(authStorage, tempDir.join("models.yml"));
session = new AgentSession({ agent, sessionManager, settings, modelRegistry });
return { session, sessionManager, mock, streamStarted: started.promise };
}
@@ -125,12 +125,19 @@ describe("AgentSession auto-compaction queue resume", () => {
});
afterEach(async () => {
await session.dispose();
authStorage.close();
tempDir.removeSync();
vi.useRealTimers();
getRuntimeSignals().length = 0;
vi.restoreAllMocks();
try {
await session?.dispose();
} finally {
try {
authStorage?.close();
vi.useRealTimers();
await Bun.sleep(0);
await tempDir?.remove();
} finally {
getRuntimeSignals().length = 0;
vi.restoreAllMocks();
}
}
});
it("resumes after threshold compaction when only agent-level queued messages exist", async () => {
@@ -1,7 +1,5 @@
import { afterEach, beforeEach, describe, expect, it, type Mock, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import * as pythonExecutor from "@oh-my-pi/pi-coding-agent/eval/py/executor";
@@ -9,8 +7,9 @@ import type { PythonKernel as PythonKernelInstance } from "@oh-my-pi/pi-coding-a
import * as pythonKernel from "@oh-my-pi/pi-coding-agent/eval/py/kernel";
import { AgentRegistry } from "@oh-my-pi/pi-coding-agent/registry/agent-registry";
import { createAgentSession, type ExtensionFactory, type WorkspaceTree } from "@oh-my-pi/pi-coding-agent/sdk";
import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { Snowflake, TempDir } from "@oh-my-pi/pi-utils";
const OK_EXECUTION = { status: "ok", cancelled: false, timedOut: false, stdinRequested: false } as const;
@@ -69,12 +68,21 @@ const getModel = () => {
};
const createTempProject = () => {
const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), `pi-agent-session-python-cleanup-${Snowflake.next()}-`));
const cwd = path.join(tempDir, "project");
const tempDir = TempDir.createSync(`@pi-agent-session-python-cleanup-${Snowflake.next()}-`);
const cwd = tempDir.join("project");
fs.mkdirSync(cwd, { recursive: true });
return { tempDir, cwd };
};
// createAgentSession opens an AuthStorage at <agentDir>/auth.db that is not
// closed when construction fails. Point agentDir at a separate dir so the
// auth.db handle doesn't keep the per-test project temp dir locked on Windows.
const agentDirPool: TempDir[] = [];
const createAgentDir = (): string => {
const dir = TempDir.createSync("@pi-python-cleanup-agentdir-");
agentDirPool.push(dir);
return dir.path();
};
const emptyWorkspaceTree = (cwd: string): WorkspaceTree => ({
rootPath: cwd,
rendered: ".",
@@ -102,14 +110,14 @@ const expectSleepNear = (sleepSpy: Mock<typeof Bun.sleep>, targetMs: number) =>
).toBe(true);
};
const createSession = async (
tempDir: string,
_tempDir: TempDir,
cwd: string,
options: { extensions?: ExtensionFactory[]; sessionManager?: SessionManager } = {},
) =>
(
await createAgentSession({
cwd,
agentDir: tempDir,
agentDir: createAgentDir(),
sessionManager: options.sessionManager ?? SessionManager.inMemory(cwd),
settings: Settings.isolated({ "python.kernelMode": "session" }),
model: getModel(),
@@ -126,7 +134,6 @@ const createSession = async (
toolNames: ["eval"],
})
).session;
const createMockKernel = () => {
let alive = true;
return {
@@ -144,7 +151,7 @@ const createMockKernel = () => {
};
describe("AgentSession python cleanup", () => {
const tempDirs: string[] = [];
const tempDirs: TempDir[] = [];
let originalNullPrompt: string | undefined;
beforeEach(() => {
@@ -160,9 +167,15 @@ describe("AgentSession python cleanup", () => {
}
originalNullPrompt = undefined;
vi.restoreAllMocks();
AgentStorage.resetInstance();
await pythonExecutor.disposeAllKernelSessions();
for (const tempDir of tempDirs.splice(0)) {
fs.rmSync(tempDir, { recursive: true, force: true });
await Bun.sleep(0);
// Best-effort cleanup: createAgentSession opens AuthStorage/AgentStorage
// inside agentDir that may outlive the test (dispose() doesn't close them).
// On Windows the leaked SQLite handles keep the dir locked; swallow EBUSY
// rather than failing the test — the OS temp dir reaper will clean up.
for (const tempDir of [...tempDirs.splice(0), ...agentDirPool.splice(0)]) {
await tempDir.remove().catch(() => {});
}
});
@@ -170,7 +183,7 @@ describe("AgentSession python cleanup", () => {
const { tempDir, cwd } = createTempProject();
tempDirs.push(tempDir);
const unrelatedKernel = createMockKernel();
const unrelatedCwd = path.join(tempDir, "unrelated-before");
const unrelatedCwd = tempDir.join("unrelated-before");
const throwingExtension: ExtensionFactory = () => {
throw new Error("Extension init failed");
};
@@ -189,7 +202,7 @@ describe("AgentSession python cleanup", () => {
await expect(
createAgentSession({
cwd,
agentDir: tempDir,
agentDir: createAgentDir(),
sessionManager: SessionManager.inMemory(cwd),
settings: Settings.isolated({ "python.kernelMode": "session" }),
model: getModel(),
@@ -237,7 +250,7 @@ describe("AgentSession python cleanup", () => {
const { tempDir, cwd } = createTempProject();
tempDirs.push(tempDir);
const unrelatedKernel = createMockKernel();
const unrelatedCwd = path.join(tempDir, "unrelated-after");
const unrelatedCwd = tempDir.join("unrelated-after");
vi.spyOn(pythonKernel, "checkPythonKernelAvailability").mockResolvedValue({ ok: true });
const startSpy = vi
.spyOn(pythonKernel.PythonKernel, "start")
@@ -257,7 +270,7 @@ describe("AgentSession python cleanup", () => {
await expect(
createAgentSession({
cwd,
agentDir: tempDir,
agentDir: createAgentDir(),
sessionManager: SessionManager.inMemory(cwd),
settings: Settings.isolated({ "python.kernelMode": "session", "memory.backend": "local" }),
model: getModel(),
@@ -0,0 +1,247 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as path from "node:path";
import { scheduler } from "node:timers/promises";
import { Agent } from "@oh-my-pi/pi-agent-core";
import type {
Api,
AssistantMessage,
Context,
Model,
SimpleStreamOptions,
TextContent,
ThinkingContent,
} from "@oh-my-pi/pi-ai";
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
import { THINKING_LOOP_ERROR_MARKER, withGeminiThinkingLoopGuard } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
import { TempDir } from "@oh-my-pi/pi-utils";
const LOOP_PARAGRAPHS = [
"I am now verifying the test module to guarantee there are no compile errors and the code is completely safe.",
"I am now verifying the test module once more to ensure there are no compile errors and the code stays completely safe.",
"I am now re-verifying the test module to confirm there are no compile errors and the code remains completely safe.",
];
function emptyUsage(): AssistantMessage["usage"] {
return {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
totalTokens: 0,
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 },
};
}
function chunkedThinkingLoopStream(model: Model<Api>, options?: SimpleStreamOptions): AssistantMessageEventStream {
const inner = new AssistantMessageEventStream();
queueMicrotask(() => {
const thinking: ThinkingContent = { type: "thinking", thinking: "" };
const partial: AssistantMessage = {
role: "assistant",
content: [thinking],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "stop",
timestamp: Date.now(),
};
inner.push({ type: "start", partial });
inner.push({ type: "thinking_start", contentIndex: 0, partial });
for (let index = 0; index < 12; index++) {
if (options?.signal?.aborted) return;
const delta = `**Confirming Safety ${index}**\n\n${LOOP_PARAGRAPHS[index % LOOP_PARAGRAPHS.length]}\n\n\n`;
thinking.thinking += delta;
inner.push({ type: "thinking_delta", contentIndex: 0, delta, partial });
}
inner.push({ type: "thinking_end", contentIndex: 0, content: thinking.thinking, partial });
inner.push({ type: "done", reason: "stop", message: partial });
});
return withGeminiThinkingLoopGuard(model, options, () => inner);
}
function successStream(model: Model<Api>): AssistantMessageEventStream {
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
const text: TextContent = { type: "text", text: "Recovered after retry." };
const partial: AssistantMessage = {
role: "assistant",
content: [text],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "stop",
timestamp: Date.now(),
};
stream.push({ type: "start", partial });
stream.push({ type: "text_start", contentIndex: 0, partial });
stream.push({ type: "text_delta", contentIndex: 0, delta: text.text, partial });
stream.push({ type: "text_end", contentIndex: 0, content: text.text, partial });
stream.push({ type: "done", reason: "stop", message: partial });
});
return stream;
}
function legacyContentfulLoopErrorStream(model: Model<Api>): AssistantMessageEventStream {
const stream = new AssistantMessageEventStream();
queueMicrotask(() => {
const text: TextContent = { type: "text", text: "Looping visible reasoning garbage." };
const partial: AssistantMessage = {
role: "assistant",
content: [text],
api: model.api,
provider: model.provider,
model: model.id,
usage: emptyUsage(),
stopReason: "error",
errorMessage: `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content. Non-retryable because output was already streamed.`,
timestamp: Date.now(),
};
stream.push({ type: "start", partial });
stream.push({ type: "text_start", contentIndex: 0, partial });
stream.push({ type: "text_delta", contentIndex: 0, delta: text.text, partial });
stream.push({ type: "text_end", contentIndex: 0, content: text.text, partial });
stream.push({ type: "error", reason: "error", error: partial });
});
return stream;
}
describe("AgentSession thinking-loop retry", () => {
let tempDir: TempDir;
let authStorage: AuthStorage;
let session: AgentSession | undefined;
beforeEach(async () => {
tempDir = TempDir.createSync("@pi-thinking-loop-retry-");
authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db"));
authStorage.setRuntimeApiKey("openrouter", "openrouter-test-key");
});
afterEach(async () => {
if (session) {
await session.dispose();
session = undefined;
}
authStorage.close();
tempDir.removeSync();
vi.restoreAllMocks();
});
it("drops a chunked thinking-loop error and retries the turn", async () => {
const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
const modelRegistry = new ModelRegistry(authStorage);
const calls: string[] = [];
const agent = new Agent({
getApiKey: requestedModel => `${requestedModel.provider}-test-key`,
initialState: {
model,
systemPrompt: ["Test"],
tools: [],
messages: [],
},
streamFn: (requestedModel, _context: Context, options?: SimpleStreamOptions) => {
calls.push(`${requestedModel.provider}/${requestedModel.id}`);
return calls.length === 1
? chunkedThinkingLoopStream(requestedModel, options)
: successStream(requestedModel);
},
});
const settings = Settings.isolated({
"compaction.enabled": false,
"retry.enabled": true,
"retry.baseDelayMs": 0,
"retry.maxDelayMs": 5_000,
"retry.maxRetries": 1,
"retry.modelFallback": false,
"todo.enabled": false,
});
settings.setModelRole("default", `${model.provider}/${model.id}`);
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings,
modelRegistry,
});
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
const retryStartEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_start" }>> = [];
const retryEndEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_end" }>> = [];
session.subscribe(event => {
if (event.type === "auto_retry_start") retryStartEvents.push(event);
if (event.type === "auto_retry_end") retryEndEvents.push(event);
});
await session.prompt("Trigger thinking loop once");
await session.waitForIdle();
expect(calls).toEqual(["openrouter/google/gemini-3.5-flash", "openrouter/google/gemini-3.5-flash"]);
expect(retryStartEvents).toHaveLength(1);
expect(retryStartEvents[0].errorMessage).toContain(THINKING_LOOP_ERROR_MARKER);
expect(retryEndEvents).toEqual([{ type: "auto_retry_end", success: true, attempt: 1 }]);
const assistants = session.agent.state.messages.filter(
(message): message is AssistantMessage => message.role === "assistant",
);
expect(assistants).toHaveLength(1);
expect(assistants[0].stopReason).toBe("stop");
expect(assistants[0].content).toEqual([{ type: "text", text: "Recovered after retry." }]);
expect(assistants[0].errorMessage).toBeUndefined();
});
it("starts retry for loop-marker errors even without transient wording", async () => {
const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
const modelRegistry = new ModelRegistry(authStorage);
const calls: string[] = [];
const agent = new Agent({
getApiKey: requestedModel => `${requestedModel.provider}-test-key`,
initialState: {
model,
systemPrompt: ["Test"],
tools: [],
messages: [],
},
streamFn: requestedModel => {
calls.push(`${requestedModel.provider}/${requestedModel.id}`);
return calls.length === 1 ? legacyContentfulLoopErrorStream(requestedModel) : successStream(requestedModel);
},
});
const settings = Settings.isolated({
"compaction.enabled": false,
"retry.enabled": true,
"retry.baseDelayMs": 0,
"retry.maxDelayMs": 5_000,
"retry.maxRetries": 1,
"retry.modelFallback": false,
"todo.enabled": false,
});
settings.setModelRole("default", `${model.provider}/${model.id}`);
session = new AgentSession({
agent,
sessionManager: SessionManager.inMemory(),
settings,
modelRegistry,
});
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
const retryStartEvents: Array<Extract<AgentSessionEvent, { type: "auto_retry_start" }>> = [];
session.subscribe(event => {
if (event.type === "auto_retry_start") retryStartEvents.push(event);
});
await session.prompt("Trigger legacy loop marker once");
await session.waitForIdle();
expect(calls).toEqual(["openrouter/google/gemini-3.5-flash", "openrouter/google/gemini-3.5-flash"]);
expect(retryStartEvents).toHaveLength(1);
expect(retryStartEvents[0].errorMessage).toContain("Non-retryable because output was already streamed");
const assistants = session.agent.state.messages.filter(
(message): message is AssistantMessage => message.role === "assistant",
);
expect(assistants).toHaveLength(1);
expect(assistants[0].content).toEqual([{ type: "text", text: "Recovered after retry." }]);
});
});
@@ -1,9 +1,8 @@
import { Database } from "bun:sqlite";
import { afterEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage";
import { TempDir } from "@oh-my-pi/pi-utils";
import { readTableSql } from "./helpers/sqlite-inspect";
const LEGACY_TIMESTAMP = 1_700_000_000;
@@ -34,18 +33,21 @@ function readSettingsRows(dbPath: string): Array<{ key: string; value: string; u
}
describe("AgentStorage SQLite compatibility", () => {
let tempDir = "";
let tempDir: TempDir;
afterEach(async () => {
AgentStorage.resetInstance();
if (tempDir) {
await fs.rm(tempDir, { recursive: true, force: true });
tempDir = "";
try {
await tempDir.remove();
} catch {}
tempDir = undefined as unknown as TempDir;
}
});
it("creates fresh storage without unixepoch defaults", async () => {
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-agent-storage-fresh-"));
const dbPath = path.join(tempDir, "agent.db");
tempDir = TempDir.createSync("omp-agent-storage-fresh-");
const dbPath = path.join(tempDir.path(), "agent.db");
const storage = await AgentStorage.open(dbPath);
storage.recordModelUsage("openai/gpt-5");
@@ -59,8 +61,8 @@ describe("AgentStorage SQLite compatibility", () => {
});
it("migrates legacy settings and model usage schemas away from unixepoch defaults", async () => {
tempDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-agent-storage-legacy-"));
const dbPath = path.join(tempDir, "agent.db");
tempDir = TempDir.createSync("omp-agent-storage-legacy-");
const dbPath = path.join(tempDir.path(), "agent.db");
const legacyDb = new Database(dbPath);
legacyDb.exec(`
CREATE TABLE schema_version (version INTEGER PRIMARY KEY);
@@ -41,4 +41,54 @@ describe("shouldEnableAppendOnlyContext", () => {
test("auto remains off for unknown providers without prefix-cache signals", () => {
expect(shouldEnableAppendOnlyContext("auto", GENERIC_PROXY)).toBe(false);
});
test("auto enables for local inference providers", () => {
// Ollama serves both `ollama-chat` (cloud-managed) and the openai-responses
// path used by locally pulled models — issue #3033 (llama.cpp KV-cache prefix
// resets every turn without append-only mode).
expect(shouldEnableAppendOnlyContext("auto", { provider: "ollama", baseUrl: "http://127.0.0.1:11434" })).toBe(
true,
);
expect(shouldEnableAppendOnlyContext("auto", { provider: "ollama-cloud", baseUrl: "https://ollama.com" })).toBe(
true,
);
expect(
shouldEnableAppendOnlyContext("auto", { provider: "lm-studio", baseUrl: "http://127.0.0.1:1234/v1" }),
).toBe(true);
// `llama.cpp` is a built-in provider id (ModelRegistry registers it for keyless local discovery);
// the allowlist must catch it even when the user reverse-proxies the server through a public host.
expect(
shouldEnableAppendOnlyContext("auto", { provider: "llama.cpp", baseUrl: "https://llamacpp.example.com/v1" }),
).toBe(true);
expect(shouldEnableAppendOnlyContext("auto", { provider: "llama.cpp", baseUrl: "http://127.0.0.1:8080" })).toBe(
true,
);
});
test("auto enables for loopback and private baseUrls (user-defined llama.cpp/vLLM)", () => {
const cases: Array<{ provider: string; baseUrl: string }> = [
{ provider: "my-llamacpp", baseUrl: "http://localhost:8080/v1" },
{ provider: "my-vllm", baseUrl: "http://127.0.0.1:8000/v1" },
{ provider: "my-sglang", baseUrl: "http://[::1]:30000/v1" },
{ provider: "lan-host", baseUrl: "http://192.168.1.42:11434" },
{ provider: "lan-host", baseUrl: "http://10.0.0.5:11434" },
{ provider: "lan-host", baseUrl: "http://172.17.0.3:11434" },
{ provider: "mdns-host", baseUrl: "http://gpu-box.local:11434" },
];
for (const model of cases) {
expect(shouldEnableAppendOnlyContext("auto", model)).toBe(true);
}
});
test("auto stays off for public hosts that merely share an IP prefix", () => {
// 172.15.x.x sits just outside the RFC1918 16-31 band.
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "http://172.15.0.1/v1" })).toBe(false);
// 172.32.x.x sits just outside the RFC1918 16-31 band on the other side.
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "http://172.32.0.1/v1" })).toBe(false);
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "https://example.com/v1" })).toBe(false);
});
test("malformed baseUrl never crashes the resolver", () => {
expect(shouldEnableAppendOnlyContext("auto", { provider: "x", baseUrl: "not a url" })).toBe(false);
});
});
@@ -1,35 +1,38 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { SelectorController } from "@oh-my-pi/pi-coding-agent/modes/controllers/selector-controller";
import { getProjectAgentDir, Snowflake } from "@oh-my-pi/pi-utils";
import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage";
import { getProjectAgentDir, TempDir } from "@oh-my-pi/pi-utils";
import { YAML } from "bun";
import { beginSettingsTest, restoreSettingsTestState, type SettingsTestState } from "./helpers/settings-test-state";
describe("autocompleteMaxVisible setting", () => {
let settingsState: SettingsTestState | undefined;
let testDir = "";
let tempDir: TempDir;
let agentDir: string;
let projectDir: string;
beforeEach(() => {
settingsState = beginSettingsTest();
testDir = path.join(os.tmpdir(), "test-autocomplete-settings", Snowflake.next());
agentDir = path.join(testDir, "agent");
projectDir = path.join(testDir, "project");
tempDir = TempDir.createSync("test-autocomplete-settings-");
agentDir = path.join(tempDir.path(), "agent");
projectDir = path.join(tempDir.path(), "project");
fs.mkdirSync(agentDir, { recursive: true });
fs.mkdirSync(getProjectAgentDir(projectDir), { recursive: true });
});
afterEach(() => {
afterEach(async () => {
AgentStorage.resetInstance();
restoreSettingsTestState(settingsState);
settingsState = undefined;
if (testDir && fs.existsSync(testDir)) {
fs.rmSync(testDir, { recursive: true, force: true });
if (tempDir) {
try {
await tempDir.remove();
} catch {}
tempDir = undefined as unknown as TempDir;
}
testDir = "";
});
it("should persist and read back a configured value", async () => {
@@ -1,7 +1,4 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { createAutoresearchExtension } from "@oh-my-pi/pi-coding-agent/autoresearch";
import {
buildExperimentState,
@@ -11,7 +8,7 @@ import {
findBestKeptMetric,
reconstructControlState,
} from "@oh-my-pi/pi-coding-agent/autoresearch/state";
import { AutoresearchStorage } from "@oh-my-pi/pi-coding-agent/autoresearch/storage";
import { AutoresearchStorage, closeAllAutoresearchStorages } from "@oh-my-pi/pi-coding-agent/autoresearch/storage";
import type { ExperimentResult } from "@oh-my-pi/pi-coding-agent/autoresearch/types";
import type {
ExtensionAPI,
@@ -19,16 +16,14 @@ import type {
RegisteredCommand,
} from "@oh-my-pi/pi-coding-agent/extensibility/extensions";
import * as git from "@oh-my-pi/pi-coding-agent/utils/git";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { TempDir } from "@oh-my-pi/pi-utils";
afterEach(() => {
vi.restoreAllMocks();
});
function makeTempDir(): string {
const dir = path.join(os.tmpdir(), `pi-autoresearch-test-${Snowflake.next()}`);
fs.mkdirSync(dir, { recursive: true });
return dir;
function makeTempDir(): TempDir {
return TempDir.createSync("@pi-autoresearch-test-");
}
function makeResult(partial: Partial<ExperimentResult>): ExperimentResult {
@@ -113,18 +108,19 @@ describe("autoresearch state math", () => {
});
describe("AutoresearchStorage round-trip", () => {
let dbDir: string;
let dbDir: TempDir;
beforeEach(() => {
dbDir = makeTempDir();
});
afterEach(() => {
fs.rmSync(dbDir, { recursive: true, force: true });
afterEach(async () => {
await Bun.sleep(0);
await dbDir.remove().catch(() => {});
});
function openStorage(): AutoresearchStorage {
return new AutoresearchStorage(path.join(dbDir, "test.db"), dbDir);
return new AutoresearchStorage(dbDir.join("test.db"), dbDir.path());
}
it("persists sessions and exposes the active session", () => {
@@ -516,25 +512,25 @@ function createCommandHarness(
}
describe("autoresearch slash command", () => {
const cleanups: string[] = [];
let dbOverride: string | undefined;
const cleanups: TempDir[] = [];
let dbOverride: TempDir | undefined;
beforeEach(() => {
dbOverride = path.join(os.tmpdir(), `pi-autoresearch-cmd-${Snowflake.next()}`);
fs.mkdirSync(dbOverride, { recursive: true });
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride;
dbOverride = TempDir.createSync("@pi-autoresearch-cmd-");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride.path();
cleanups.push(dbOverride);
});
afterEach(() => {
delete process.env.OMP_AUTORESEARCH_DB_DIR;
closeAllAutoresearchStorages();
for (const dir of cleanups.splice(0)) {
fs.rmSync(dir, { recursive: true, force: true });
dir.removeSync();
}
});
it("enables autoresearch with a notify when invoked bare in a clean repo", async () => {
const dir = makeTempDir();
const dir = makeTempDir().path();
const harness = createCommandHarness(dir, async (_command, args) => {
if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` };
if (args[0] === "branch" && args[1] === "--show-current") return { code: 0, stderr: "", stdout: "main\n" };
@@ -549,7 +545,7 @@ describe("autoresearch slash command", () => {
});
it("forwards a slash argument as the user message and creates a slug branch", async () => {
const dir = makeTempDir();
const dir = makeTempDir().path();
const harness = createCommandHarness(dir, async (_command, args) => {
if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` };
if (args[0] === "branch" && args[1] === "--show-current") return { code: 0, stderr: "", stdout: "main\n" };
@@ -565,7 +561,7 @@ describe("autoresearch slash command", () => {
});
it("aborts with an error when the worktree is dirty", async () => {
const dir = makeTempDir();
const dir = makeTempDir().path();
const harness = createCommandHarness(dir, async (_command, args) => {
if (args[0] === "rev-parse") return { code: 0, stderr: "", stdout: `${dir}\n` };
if (args[0] === "branch" && args[1] === "--show-current") return { code: 0, stderr: "", stdout: "main\n" };
@@ -583,20 +579,20 @@ describe("autoresearch slash command", () => {
});
describe("autoresearch tool-call hook", () => {
const cleanups: string[] = [];
let dbOverride: string;
const cleanups: TempDir[] = [];
let dbOverride: TempDir;
beforeEach(() => {
dbOverride = path.join(os.tmpdir(), `pi-autoresearch-hook-${Snowflake.next()}`);
fs.mkdirSync(dbOverride, { recursive: true });
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride;
dbOverride = TempDir.createSync("@pi-autoresearch-hook-");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride.path();
cleanups.push(dbOverride);
});
afterEach(() => {
delete process.env.OMP_AUTORESEARCH_DB_DIR;
closeAllAutoresearchStorages();
for (const dir of cleanups.splice(0)) {
fs.rmSync(dir, { recursive: true, force: true });
dir.removeSync();
}
});
@@ -1,11 +1,11 @@
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
import { createSessionRuntime } from "@oh-my-pi/pi-coding-agent/autoresearch/state";
import {
type AutoresearchStorage,
closeAllAutoresearchStorages,
openAutoresearchStorage,
type SessionRow,
} from "@oh-my-pi/pi-coding-agent/autoresearch/storage";
@@ -16,7 +16,7 @@ import { createUpdateNotesTool } from "@oh-my-pi/pi-coding-agent/autoresearch/to
import type { ASIData, LogDetails, NumericMetricMap, RunDetails } from "@oh-my-pi/pi-coding-agent/autoresearch/types";
import type { ExtensionAPI, ExtensionContext } from "@oh-my-pi/pi-coding-agent/extensibility/extensions";
import * as git from "@oh-my-pi/pi-coding-agent/utils/git";
import { Snowflake } from "@oh-my-pi/pi-utils";
import { TempDir } from "@oh-my-pi/pi-utils";
import { $ } from "bun";
afterEach(() => {
@@ -29,10 +29,8 @@ function firstTextBlockText(content: Array<TextContent | ImageContent>): string
return block.text;
}
function makeTempDir(prefix = "pi-autoresearch-tools"): string {
const dir = path.join(os.tmpdir(), `${prefix}-${Snowflake.next()}`);
fs.mkdirSync(dir, { recursive: true });
return dir;
function makeTempDir(prefix = "@pi-autoresearch-tools-"): TempDir {
return TempDir.createSync(prefix);
}
function dashboardStub() {
@@ -78,43 +76,47 @@ function createPiHarness(initialTools: string[] = []): PiHarness {
// inside a real repo, so the production tools resolve HEAD/branch from `.git` on
// disk (sub-millisecond) instead of spawning fallback git subprocesses for every
// `repo.root` / `branch.current` / `head.sha` lookup against a bare temp dir.
let templateRepo: string;
let templateBranchRepo: string;
let templateRepo: TempDir;
let templateBranchRepo: TempDir;
let templateBaselineCommit: string;
beforeAll(async () => {
templateRepo = makeTempDir("pi-autoresearch-template");
await Bun.write(path.join(templateRepo, "README.md"), "# baseline\n");
await $`git init --initial-branch=main && git config user.email tester@example.com && git config user.name Tester && git add -A && git commit -m baseline`
.cwd(templateRepo)
templateRepo = makeTempDir("@pi-autoresearch-template-");
await Bun.write(path.join(templateRepo.path(), "README.md"), "# baseline\n");
await $`git init --initial-branch=main && git config core.autocrlf false && git config user.email tester@example.com && git config user.name Tester && git add -A && git commit -m baseline`
.cwd(templateRepo.path())
.quiet();
templateBaselineCommit = (await $`git rev-parse HEAD`.cwd(templateRepo).text()).trim();
templateBaselineCommit = (await $`git rev-parse HEAD`.cwd(templateRepo.path()).text()).trim();
// Second fixture: harness committed and already on an `autoresearch/*` branch,
// the baseline for log_experiment's on-branch keep/discard scenarios.
templateBranchRepo = makeTempDir("pi-autoresearch-template-branch");
fs.cpSync(templateRepo, templateBranchRepo, { recursive: true });
await Bun.write(path.join(templateBranchRepo, "autoresearch.sh"), "#!/usr/bin/env bash\necho METRIC m=1\n");
await $`git add -A && git commit -m harness && git checkout -b autoresearch/base`.cwd(templateBranchRepo).quiet();
templateBranchRepo = makeTempDir("@pi-autoresearch-template-branch-");
fs.cpSync(templateRepo.path(), templateBranchRepo.path(), { recursive: true });
await Bun.write(path.join(templateBranchRepo.path(), "autoresearch.sh"), "#!/usr/bin/env bash\necho METRIC m=1\n");
await $`git add -A && git commit -m harness && git checkout -b autoresearch/base`
.cwd(templateBranchRepo.path())
.quiet();
});
afterAll(() => {
fs.rmSync(templateRepo, { recursive: true, force: true });
fs.rmSync(templateBranchRepo, { recursive: true, force: true });
afterAll(async () => {
closeAllAutoresearchStorages();
await Bun.sleep(0);
await templateRepo.remove();
await templateBranchRepo.remove();
});
// Independent working copy of the template repo: baseline commit on `main`,
// committer identity configured, ready for per-test branch/commit scenarios.
function freshRepo(): { dir: string; baselineCommit: string } {
const dir = makeTempDir();
fs.cpSync(templateRepo, dir, { recursive: true });
const dir = makeTempDir().path();
fs.cpSync(templateRepo.path(), dir, { recursive: true });
return { dir, baselineCommit: templateBaselineCommit };
}
// Like freshRepo, but already on an `autoresearch/*` branch with the harness
// committed — the baseline for log_experiment's on-branch keep/discard paths.
function freshBranchRepo(): { dir: string } {
const dir = makeTempDir();
fs.cpSync(templateBranchRepo, dir, { recursive: true });
const dir = makeTempDir().path();
fs.cpSync(templateBranchRepo.path(), dir, { recursive: true });
return { dir };
}
@@ -162,16 +164,18 @@ function seedCompletedRun(
}
describe("init_experiment", () => {
let dbOverride: string;
let dbOverride: TempDir;
beforeEach(() => {
dbOverride = makeTempDir("pi-autoresearch-init-db");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride;
dbOverride = makeTempDir("@pi-autoresearch-init-db-");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride.path();
});
afterEach(() => {
afterEach(async () => {
delete process.env.OMP_AUTORESEARCH_DB_DIR;
fs.rmSync(dbOverride, { recursive: true, force: true });
closeAllAutoresearchStorages();
await Bun.sleep(0);
await dbOverride.remove();
});
it("opens a new session and persists scope and metric metadata", async () => {
@@ -340,16 +344,18 @@ describe("init_experiment", () => {
});
describe("run_experiment", () => {
let dbOverride: string;
let dbOverride: TempDir;
beforeEach(() => {
dbOverride = makeTempDir("pi-autoresearch-run-db");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride;
dbOverride = makeTempDir("@pi-autoresearch-run-db-");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride.path();
});
afterEach(() => {
afterEach(async () => {
delete process.env.OMP_AUTORESEARCH_DB_DIR;
fs.rmSync(dbOverride, { recursive: true, force: true });
closeAllAutoresearchStorages();
await Bun.sleep(0);
await dbOverride.remove();
});
it("rejects when no session is active", async () => {
@@ -429,16 +435,18 @@ describe("run_experiment", () => {
});
describe("log_experiment", () => {
let dbOverride: string;
let dbOverride: TempDir;
beforeEach(() => {
dbOverride = makeTempDir("pi-autoresearch-log-db");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride;
dbOverride = makeTempDir("@pi-autoresearch-log-db-");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride.path();
});
afterEach(() => {
afterEach(async () => {
delete process.env.OMP_AUTORESEARCH_DB_DIR;
fs.rmSync(dbOverride, { recursive: true, force: true });
closeAllAutoresearchStorages();
await Bun.sleep(0);
await dbOverride.remove();
});
async function setupRun(dir: string, runtime = createSessionRuntime()) {
@@ -563,7 +571,7 @@ describe("log_experiment", () => {
it("flags previously logged runs via flag_runs", async () => {
// Bare temp dir (no repo): the session is created with `branch: null`, so the
// tool's branch lookup must also resolve to null to match it.
const dir = makeTempDir();
const dir = makeTempDir().path();
const storage = await openAutoresearchStorage(dir);
const session = storage.openSession({
name: "speed",
@@ -831,16 +839,18 @@ describe("log_experiment", () => {
});
describe("update_notes", () => {
let dbOverride: string;
let dbOverride: TempDir;
beforeEach(() => {
dbOverride = makeTempDir("pi-autoresearch-notes-db");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride;
dbOverride = makeTempDir("@pi-autoresearch-notes-db-");
process.env.OMP_AUTORESEARCH_DB_DIR = dbOverride.path();
});
afterEach(() => {
afterEach(async () => {
delete process.env.OMP_AUTORESEARCH_DB_DIR;
fs.rmSync(dbOverride, { recursive: true, force: true });
closeAllAutoresearchStorages();
await Bun.sleep(0);
await dbOverride.remove().catch(() => {});
});
it("replaces session notes and refreshes runtime state", async () => {
@@ -0,0 +1,137 @@
import { describe, expect, it } from "bun:test";
import type {
Api,
ApiKeyResolver,
AssistantMessage,
AssistantMessageEvent,
AssistantMessageEventStream,
Model,
} from "@oh-my-pi/pi-ai";
import { type BenchModelRegistry, runBenchCommand } from "@oh-my-pi/pi-coding-agent/cli/bench-cli";
function fakeModel(provider: string, id: string): Model<Api> {
return {
provider,
id,
name: id,
api: "openai-completions",
maxTokens: 4096,
contextWindow: 128_000,
} as unknown as Model<Api>;
}
function fakeStream(): AssistantMessageEventStream {
const message = {
role: "assistant",
content: [],
stopReason: "stop",
usage: { input: 5, output: 20 },
duration: 120,
ttft: 30,
} as unknown as AssistantMessage;
const events = [
{ type: "text_delta", delta: "hi" },
{ type: "done", message },
] as unknown as AssistantMessageEvent[];
const iterator = (async function* () {
for (const event of events) yield event;
})();
return Object.assign(iterator, { result: async () => message }) as unknown as AssistantMessageEventStream;
}
interface FakeRegistryOptions {
models: Model<Api>[];
authedProviders: string[];
canonicalId?: (model: Model<Api>) => string | undefined;
canonicalVariants?: Record<string, Model<Api>[]>;
}
function fakeRegistry(opts: FakeRegistryOptions): BenchModelRegistry {
const authed = new Set(opts.authedProviders);
return {
getAll: () => opts.models,
hasConfiguredAuth: model => authed.has(model.provider),
getApiKey: async model => (authed.has(model.provider) ? "sk-test" : undefined),
resolver: () => (() => Promise.resolve("sk-test")) as unknown as ApiKeyResolver,
getCanonicalId: opts.canonicalId,
getCanonicalVariants: opts.canonicalVariants
? canonicalId =>
(opts.canonicalVariants?.[canonicalId] ?? []).map(model => ({
canonicalId,
selector: `${model.provider}/${model.id}`,
model,
source: "bundled" as const,
}))
: undefined,
};
}
async function runBench(selector: string, registry: BenchModelRegistry) {
const stderr: string[] = [];
const summary = await runBenchCommand(
{ models: [selector], flags: { runs: 1, maxTokens: 64, json: false } },
{
createRuntime: async () => ({ modelRegistry: registry, settings: undefined, close: () => {} }),
randomSessionId: () => "sess-1",
writeStdout: () => {},
writeStderr: text => stderr.push(text),
setExitCode: () => {},
streamSimple: () => fakeStream(),
now: () => 0,
stdoutIsTTY: false,
},
);
return { summary, stderr: stderr.join("") };
}
describe("bench credential-aware provider selection", () => {
it("redirects an ambiguous shared-id selector to an authenticated provider", async () => {
// Catalog order makes the unauthenticated `groq` win the default resolution.
const registry = fakeRegistry({
models: [fakeModel("groq", "openai/gpt-oss-20b"), fakeModel("openrouter", "openai/gpt-oss-20b")],
authedProviders: ["openrouter"],
});
const { summary, stderr } = await runBench("openai/gpt-oss-20b", registry);
expect(summary.models[0].model).toBe("openrouter/openai/gpt-oss-20b");
expect(summary.failures).toBe(0);
expect(stderr).toContain('no credentials for "groq"');
expect(stderr).toContain("openrouter/openai/gpt-oss-20b");
});
it("redirects across providers whose local ids differ, via canonical variants", async () => {
// Bare `gpt-oss-20b` resolves to fireworks (unauthed) by flat-id match; the
// only authenticated equivalent is openrouter under a *different* local id,
// so the swap must travel through the canonical variant index.
const fireworks = fakeModel("fireworks", "gpt-oss-20b");
const openrouter = fakeModel("openrouter", "openai/gpt-oss-20b");
const registry = fakeRegistry({
models: [fireworks, openrouter],
authedProviders: ["openrouter"],
canonicalId: model => (model === fireworks || model === openrouter ? "gpt-oss-20b" : undefined),
canonicalVariants: { "gpt-oss-20b": [fireworks, openrouter] },
});
const { summary, stderr } = await runBench("gpt-oss-20b", registry);
expect(summary.models[0].model).toBe("openrouter/openai/gpt-oss-20b");
expect(summary.failures).toBe(0);
expect(stderr).toContain('no credentials for "fireworks"');
});
it("honors an explicitly pinned provider even without credentials", async () => {
const registry = fakeRegistry({
models: [fakeModel("groq", "openai/gpt-oss-20b"), fakeModel("openrouter", "openai/gpt-oss-20b")],
authedProviders: ["openrouter"],
});
const { summary, stderr } = await runBench("groq/openai/gpt-oss-20b", registry);
// Pinned selector is authoritative: no redirect, surfaces the no-credentials failure.
expect(summary.models[0].model).toBe("groq/openai/gpt-oss-20b");
expect(summary.failures).toBe(1);
expect(summary.models[0].results[0]).toMatchObject({ ok: false });
expect(stderr).not.toContain("benchmarking");
});
});
+13 -8
View File
@@ -1,23 +1,23 @@
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { runConfigCommand } from "@oh-my-pi/pi-coding-agent/cli/config-cli";
import { resetSettingsForTest } from "@oh-my-pi/pi-coding-agent/config/settings";
import { getConfigRootDir, setAgentDir } from "@oh-my-pi/pi-utils";
import { AgentStorage } from "@oh-my-pi/pi-coding-agent/session/agent-storage";
import { getConfigRootDir, setAgentDir, TempDir } from "@oh-my-pi/pi-utils";
let testAgentDir = "";
let testAgentDir: TempDir | undefined;
const originalAgentDir = process.env.PI_CODING_AGENT_DIR;
const fallbackAgentDir = path.join(getConfigRootDir(), "agent");
beforeEach(async () => {
beforeEach(() => {
resetSettingsForTest();
testAgentDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-config-cli-"));
setAgentDir(testAgentDir);
testAgentDir = TempDir.createSync("omp-config-cli-");
setAgentDir(testAgentDir.path());
});
afterEach(async () => {
vi.restoreAllMocks();
AgentStorage.resetInstance();
resetSettingsForTest();
if (originalAgentDir) {
setAgentDir(originalAgentDir);
@@ -25,7 +25,12 @@ afterEach(async () => {
setAgentDir(fallbackAgentDir);
delete process.env.PI_CODING_AGENT_DIR;
}
await fs.rm(testAgentDir, { recursive: true, force: true });
if (testAgentDir) {
try {
await testAgentDir.remove();
} catch {}
testAgentDir = undefined;
}
});
describe("config CLI schema coverage", () => {
@@ -71,7 +71,7 @@ describe("Python tool bridge HTTP server", () => {
expect(res.status).toBe(200);
expect(body).toEqual({ ok: true, value: "file body" });
expect(calls).toHaveLength(1);
// `_i` survives the bridge round trip so transcript renderers have a label.
// `i` survives the bridge round trip so transcript renderers have a label.
expect((calls[0]!.args as Record<string, unknown>)[INTENT_FIELD]).toBe("py prelude");
} finally {
unregister();
@@ -0,0 +1,66 @@
import { afterEach, beforeEach, describe, expect, it } from "bun:test";
import * as fs from "node:fs/promises";
import * as os from "node:os";
import * as path from "node:path";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { CursorExecHandlers } from "@oh-my-pi/pi-coding-agent/cursor";
import { SearchTool, type ToolSession } from "@oh-my-pi/pi-coding-agent/tools";
function createTestSession(cwd: string, overrides: Partial<ToolSession> = {}): ToolSession {
return {
cwd,
hasUI: false,
getSessionFile: () => null,
getSessionSpawns: () => "*",
settings: Settings.isolated(),
...overrides,
};
}
describe("CursorExecHandlers.grep bridge", () => {
let cwd: string;
let searchTool: SearchTool;
let handlers: CursorExecHandlers;
beforeEach(async () => {
cwd = await fs.mkdtemp(path.join(os.tmpdir(), "cursor-exec-test-"));
await Bun.write(path.join(cwd, "sample.txt"), "Hello World\nhello world\n");
searchTool = new SearchTool(createTestSession(cwd));
handlers = new CursorExecHandlers({
cwd,
tools: new Map([["search", searchTool as any]]),
});
});
afterEach(async () => {
await fs.rm(cwd, { recursive: true, force: true });
});
it("maps caseInsensitive parameter correctly through the grep bridge", async () => {
// 1. By default/omitted caseInsensitive, should be case-sensitive (match count 1 for "hello")
const defaultResult = await handlers.grep({
toolCallId: "call-1",
path: cwd,
pattern: "hello",
} as any);
expect(defaultResult.details?.matchCount).toBe(1);
// 2. If caseInsensitive: true, should be case-insensitive (match count 2 for "hello")
const insensitiveResult = await handlers.grep({
toolCallId: "call-2",
path: cwd,
pattern: "hello",
caseInsensitive: true,
} as any);
expect(insensitiveResult.details?.matchCount).toBe(2);
// 3. If caseInsensitive: false, should be case-sensitive (match count 1 for "hello")
const sensitiveResult = await handlers.grep({
toolCallId: "call-3",
path: cwd,
pattern: "hello",
caseInsensitive: false,
} as any);
expect(sensitiveResult.details?.matchCount).toBe(1);
});
});
@@ -9,6 +9,7 @@
*/
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
import type { AssistantMessage } from "@oh-my-pi/pi-ai";
import { THINKING_LOOP_ERROR_MARKER } from "@oh-my-pi/pi-ai/utils/thinking-loop";
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
import { AssistantMessageComponent } from "@oh-my-pi/pi-coding-agent/modes/components/assistant-message";
import { ErrorBannerComponent } from "@oh-my-pi/pi-coding-agent/modes/components/error-banner";
@@ -60,15 +61,22 @@ function createFixture(streamingMessage?: AssistantMessage) {
};
const showPinnedError = vi.fn();
const clearPinnedError = vi.fn();
const statusContainer = {
clear: vi.fn(),
addChild: vi.fn(),
};
const session = { isTtsrAbortPending: false, retryAttempt: 0 };
const ctx = {
isInitialized: true,
init: vi.fn(async () => {}),
ui: { requestRender: vi.fn() },
ui: { requestRender: vi.fn(), requestComponentRender: vi.fn() },
statusLine: { invalidate: vi.fn() },
updateEditorTopBorder: vi.fn(),
ensureLoadingAnimation: vi.fn(),
statusContainer,
loadingAnimation: undefined,
retryLoader: undefined,
editor: {},
streamingComponent: streamingMessage ? streamingComponent : undefined,
streamingMessage,
@@ -121,6 +129,35 @@ describe("EventController error banner", () => {
expect(streamingComponent.setErrorPinned).toHaveBeenCalledWith(false);
});
it("clears retryable thinking-loop banners without restoring the dropped inline error", async () => {
const errorMessage = `${THINKING_LOOP_ERROR_MARKER}: the model repeated near-identical content. Treating as a stream stall and retrying.`;
const message = makeAssistantMessage({ stopReason: "error", errorMessage });
const { controller, clearPinnedError, streamingComponent } = createFixture(message);
await controller.handleEvent({ type: "message_end", message } as Extract<
AgentSessionEvent,
{ type: "message_end" }
>);
clearPinnedError.mockClear();
streamingComponent.setErrorPinned.mockClear();
await controller.handleEvent({
type: "auto_retry_start",
attempt: 1,
maxAttempts: 2,
delayMs: 0,
errorMessage,
} as Extract<AgentSessionEvent, { type: "auto_retry_start" }>);
expect(clearPinnedError).toHaveBeenCalledTimes(1);
expect(streamingComponent.setErrorPinned).not.toHaveBeenCalledWith(false);
await controller.handleEvent({
type: "auto_retry_end",
success: true,
attempt: 1,
} as Extract<AgentSessionEvent, { type: "auto_retry_end" }>);
});
it("does not pin a banner for a normal assistant stop", async () => {
const message = makeAssistantMessage({ stopReason: "stop" });
const { controller, showPinnedError } = createFixture(message);

Some files were not shown because too many files have changed in this diff Show More