Merge remote-tracking branch 'upstream/main' into feat/secret-friendly-names
This commit is contained in:
@@ -18,13 +18,81 @@
|
||||
- Fixed a multi-character `mode: "replace"` regex remainder (the bytes of a match outside a preserved `#…#` placeholder) drifting across an obfuscator restart, which invalidated provider prompt-cache prefixes even with a stable key. The remainder was redacted to a content-derived `ZZ`+hash marker that was only recognized as already-redacted within the generating session (via an in-memory set), so a fresh obfuscator reprocessing persisted text re-redacted it to a different value (`ZZPL#…#` → `ZZ7f#…#`). The remainder marker now derives from a keyed run of the per-install key and the remainder length, so any instance sharing the key reproduces it byte-identically (idempotent across restart) while staying unpredictable enough that raw sentinel-shaped bytes (`ZZZZ`) still differ from it and are redacted rather than passed through ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)).
|
||||
- Fixed a secret regex match that starts in outside text and ends inside a previously generated `#…#` placeholder's expanded value leaving an independently-matching outside prefix provider-visible. Resuming the scan past the cut placeholder skipped the whole straddling span, so a pattern like `[A-Z0-9]{8,12}` greedily spanning `SECRETUV` into an `ABCDEFGH` placeholder returned `SECRETUV#…#` even though `SECRETUV` satisfies the regex on its own. The cut handling now re-runs the regex bounded to just before the placeholder (full left context kept, so lookbehind still evaluates) and redacts the standalone prefix match — to its own reversible placeholder in obfuscate mode, or a one-way redaction in replace mode — while the cut secret stays as its existing placeholder. The replace-mode redaction's fixed point is verified against the placeholder-expanded view re-obfuscation actually scans, so it does not drift when the adjacent placeholder expands ([#2465](https://github.com/can1357/oh-my-pi/issues/2465)).
|
||||
|
||||
## [16.2.9] - 2026-06-30
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Renamed the built-in quick_task subagent to sonic; update any task spawns or configurations referencing quick_task by name.
|
||||
|
||||
### Added
|
||||
|
||||
- Added the llama3.2:3b local model option for memory and auto-thinking tasks, utilizing a quantized ONNX model.
|
||||
- Added a built-in Tester subagent designed to write high-signal tests for behavior, invariants, and edge cases while avoiding redundant or low-value tests.
|
||||
- Added a Speech-to-Text submit trigger setting to auto-submit dictation on release, on complete sentences, or via a spoken submit command.
|
||||
- Added a loop-guard mechanism that detects thinking/response loops and injects a system notice during auto-retries to guide the model to break the pattern and take a concrete next step.
|
||||
|
||||
### Changed
|
||||
|
||||
- Enabled contextual snapcompact shape resolution based on rendered text content
|
||||
- Renamed the Speech-to-Text (STT) setting label from "TTS Submit Trigger" to "Speech-to-Text Submit Trigger".
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed snapcompact preflight to use the same font-aware renderability probe as compaction, including prior preserved archive text, so CJK history remains renderable through per-glyph Silver fallback across repeated compactions.
|
||||
- Fixed an issue where mid-run compaction was incorrectly skipped when a persisted assistant display variant shared a persistence key but differed in content from the live message.
|
||||
- Fixed duplicate placeholder cards being created when streamed tool blocks started with an empty ID.
|
||||
- Fixed omp debug --profile failing on Bun by treating the optional --allow-natives-syntax flag as best-effort when v8.setFlagsFromString is unavailable.
|
||||
- Fixed release binaries missing the compiled tiny-model Transformers.js version pin, preventing runtime resolution issues on certain platforms like Homebrew Darwin arm64.
|
||||
- Fixed MCP OAuth flows silently falling back to a random redirect port when the preferred port (default 3000) was busy, which caused authentication failures with strict providers. The flow now fails fast with a configuration error when a static client ID is pinned, while dynamic registration flows continue to use fallback ports.
|
||||
- Fixed /mcp reauth and /mcp add commands ignoring the Escape key during OAuth authentication, allowing users to cancel the flow immediately instead of waiting for the timeout.
|
||||
|
||||
### Removed
|
||||
|
||||
- Removed the built-in oracle subagent.
|
||||
|
||||
## [16.2.8] - 2026-06-30
|
||||
|
||||
### Added
|
||||
|
||||
- Added built-in Go coding rules including `go-add-cleanup`, `go-bench-loop`, `go-exp-promoted`, `go-ioutil`, `go-join-hostport`, `go-new-expr`, `go-rand-v2`, and `go-range-int`
|
||||
|
||||
### Changed
|
||||
|
||||
- Relaxed strict bash tool constraints regarding the use of search, grep, ls, and find commands
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed auto-compaction dead-ends by automatically triggering a shake rescue to elide oversized tails
|
||||
- Improved compaction warning message to suggest running `/shake images` for irreducible image tails
|
||||
- Fixed `grep`/`search` direct execution to accept JSON-array string `paths` for string-or-array inputs. ([#3873](https://github.com/can1357/oh-my-pi/issues/3873))
|
||||
- Fixed auto-compaction dead-ending with "Compaction freed too little context to make progress" when a single recent turn (large tool output, heavy fenced/XML block) is itself bigger than the recovery band — `findCutPoint` can't cut inside one message, so the summarizer had no lever left. The guard now runs an artifact-backed `shake` elide pass over the oversized tail and re-tests headroom before pausing, and the remaining warning points at `/shake images` for image-only tails it can't elide. ([#3786](https://github.com/can1357/oh-my-pi/issues/3786))
|
||||
- Fixed reviewer/`task` subagents whose incremental `yield` (`type: ["overall_correctness"]`, `type: ["findings"]`, …) carried a value that mismatched the matching property's sub-schema being silently accepted and then post-mortem rejected with `schema_violation` — opaquely swapping the agent's accepted output for an error blob. The yield tool now validates each incremental section's `data` against its top-level property's sub-schema (items schema for array-typed labels) and surfaces the same retry feedback as terminal yields, so models like `deepseek-v4-pro` that emit `"Correct"`/`"correct."`/`"approved"` for an enum field get up to three corrective retries; the existing `MAX_SCHEMA_RETRIES` override then accepts the value with `SUBAGENT_WARNING_SCHEMA_OVERRIDDEN` instead of losing the entire result. Unknown labels stay unconstrained ([#3870](https://github.com/can1357/oh-my-pi/issues/3870)).
|
||||
- Fixed streaming tool-call previews (notably `write`) showing an empty body for the entire streaming phase by surfacing the partial JSON already in hand on the first reveal, then pacing only subsequent growth ([#3881](https://github.com/can1357/oh-my-pi/issues/3881)).
|
||||
- Fixed hashline edit mode preserving UTF-8 BOM bytes on edited files. ([#3867](https://github.com/can1357/oh-my-pi/issues/3867))
|
||||
- Fixed slow local LLM streams by forwarding persisted stream timeout settings (`providers.streamFirstEventTimeoutSeconds`, `providers.streamIdleTimeoutSeconds`) into model requests, so users can widen or disable watchdogs without environment variables. ([#3878](https://github.com/can1357/oh-my-pi/issues/3878))
|
||||
- Fixed the bash interceptor blocking `echo` / `printf` redirects to `/dev/null`, `/dev/tty`, `/dev/stdout`, and `/dev/stderr` device sinks while still directing real file writes to the write tool. ([#3763](https://github.com/can1357/oh-my-pi/issues/3763))
|
||||
- Fixed long snapcompact sessions re-sending multi-megabyte standing image archives on every provider request by enforcing a per-request frame byte budget, letting auto-compaction fall back to context-full summaries when snapcompact output is too large, and omitting legacy over-budget frames from rebuilt LLM contexts. ([#3792](https://github.com/can1357/oh-my-pi/issues/3792))
|
||||
|
||||
## [16.2.7] - 2026-06-30
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
- Replaced the global `serviceTier` and `fastModeScope` settings with granular, per-family settings (`tier.openai`, `tier.anthropic`, and `tier.google`) to control service tiers, subagents, advisors, and `/fast` mode targets.
|
||||
|
||||
### Changed
|
||||
|
||||
- Improved binary file detection and terminal handling to prevent corruption from non-UTF-8 content, and updated file summaries to explicitly note skipped binary files.
|
||||
- Enhanced context compaction (snapcompact) to resolve shapes contextually based on rendered text content.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Improved reliability of DuckDuckGo web searches by updating browser request headers and parameters
|
||||
- Fixed an issue where CJK (Chinese, Japanese, Korean) history could become unrenderable during repeated context compactions.
|
||||
- Fixed a memory exhaustion bug in the TUI when using `/resume` on large previous sessions.
|
||||
- Fixed an issue where the `irc` inbox missed messages that arrived while the recipient agent was already running.
|
||||
- Fixed a startup hang caused by system-prompt GPU detection blocking and repeatedly running failed probes.
|
||||
- Improved error reporting for `omp tiny-models download` by displaying the actual worker-side download error.
|
||||
- Resolved status inconsistencies between `/extensions`, `/mcp list`, and the dashboard, ensuring MCP server states, allowlists/denylists, and configuration files (like `mcp.json`) stay fully synchronized.
|
||||
- Improved branch-mode task merges to preserve the agent's original commit history (messages and authors) and fixed a bug where merges were rejected due to unrelated dirty changes in the parent checkout.
|
||||
- Fixed an issue where the `Working...` loader spinner would prematurely disappear or fail to re-arm after a subagent (`task`) tool completed or during transient overlays (such as auto-compaction or auto-retry).
|
||||
|
||||
## [16.2.6] - 2026-06-29
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"type": "module",
|
||||
"name": "@oh-my-pi/pi-coding-agent",
|
||||
"version": "16.2.6",
|
||||
"version": "16.2.9",
|
||||
"description": "Coding agent CLI with read, bash, edit, write tools and session management",
|
||||
"homepage": "https://omp.sh",
|
||||
"author": "Can Boluk",
|
||||
|
||||
@@ -10,9 +10,10 @@ import type {
|
||||
Model,
|
||||
ProviderSessionState,
|
||||
ServiceTier,
|
||||
ServiceTierByFamily,
|
||||
SimpleStreamOptions,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import { streamSimple } from "@oh-my-pi/pi-ai";
|
||||
import { resolveModelServiceTier, streamSimple } from "@oh-my-pi/pi-ai";
|
||||
import { buildModelProviderPriorityRank, type CanonicalModelVariant } from "@oh-my-pi/pi-catalog/identity";
|
||||
import { replaceTabs, truncateToWidth } from "@oh-my-pi/pi-tui";
|
||||
import { formatDuration, getProjectDir } from "@oh-my-pi/pi-utils";
|
||||
@@ -25,7 +26,7 @@ import {
|
||||
getModelMatchPreferences,
|
||||
resolveCliModel,
|
||||
} from "../config/model-resolver";
|
||||
import { resolveServiceTierSetting } from "../config/service-tier";
|
||||
import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier";
|
||||
import { Settings } from "../config/settings";
|
||||
import benchPrompt from "../prompts/bench.md" with { type: "text" };
|
||||
import { discoverAuthStorage, loadCliExtensionProviders } from "../sdk";
|
||||
@@ -106,8 +107,8 @@ export interface BenchSummary {
|
||||
maxTokens: number;
|
||||
models: BenchModelReport[];
|
||||
failures: number;
|
||||
/** Requested service tier passed to every request; absent when none was requested. Scoped tiers (`openai-only`/`claude-only`) may be dropped per-provider downstream. */
|
||||
serviceTier?: ServiceTier;
|
||||
/** Requested per-family service tiers, resolved per model before reaching the wire. */
|
||||
serviceTierByFamily?: ServiceTierByFamily;
|
||||
}
|
||||
|
||||
type BenchStreamSimple = (
|
||||
@@ -518,12 +519,18 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
|
||||
const runtime = await (deps.createRuntime ?? createDefaultRuntime)();
|
||||
try {
|
||||
const targets = resolveBenchModels(command.models, runtime.modelRegistry, runtime.settings, writeStderr);
|
||||
// Explicit `--service-tier` wins; otherwise fall back to the configured
|
||||
// `serviceTier` setting (`none`/unset omits the wire field). Scope-aware
|
||||
// gating to the model's provider happens downstream in the provider layer.
|
||||
const serviceTierValue = command.flags.serviceTier ?? runtime.settings?.get("serviceTier");
|
||||
const serviceTier = serviceTierValue ? resolveServiceTierSetting(serviceTierValue, undefined) : undefined;
|
||||
if (!json && serviceTier) writeStdout(`${chalk.dim(`service tier: ${serviceTier}`)}\n`);
|
||||
// Explicit `--service-tier` (a single value broadcast across families) wins;
|
||||
// otherwise fall back to the configured per-family `tier.*` settings. Each
|
||||
// model resolves its own family's tier below before reaching the wire.
|
||||
const flagTier = command.flags.serviceTier ? serviceTierSettingToTier(command.flags.serviceTier) : undefined;
|
||||
const serviceTierByFamily = command.flags.serviceTier
|
||||
? serviceTierForAllFamilies(flagTier)
|
||||
: buildServiceTierByFamily(
|
||||
runtime.settings?.get("tier.openai") ?? "none",
|
||||
runtime.settings?.get("tier.anthropic") ?? "none",
|
||||
runtime.settings?.get("tier.google") ?? "none",
|
||||
);
|
||||
if (!json && flagTier) writeStdout(`${chalk.dim(`service tier: ${flagTier}`)}\n`);
|
||||
const reports: BenchModelReport[] = [];
|
||||
for (const { selector, model, thinking } of targets) {
|
||||
if (!json) {
|
||||
@@ -564,7 +571,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
|
||||
maxTokens,
|
||||
reasoning: toReasoningEffort(thinking),
|
||||
disableReasoning: shouldDisableReasoning(thinking) ? true : undefined,
|
||||
serviceTier,
|
||||
serviceTier: resolveModelServiceTier(serviceTierByFamily, model),
|
||||
},
|
||||
streamFn,
|
||||
now,
|
||||
@@ -606,7 +613,7 @@ export async function runBenchCommand(command: BenchCommandArgs, deps: BenchDepe
|
||||
reports.push(buildModelReport(selector, model, thinking, results));
|
||||
}
|
||||
const failures = reports.reduce((sum, report) => sum + report.results.filter(result => !result.ok).length, 0);
|
||||
const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTier };
|
||||
const summary: BenchSummary = { runs, maxTokens, models: reports, failures, serviceTierByFamily };
|
||||
if (json) {
|
||||
writeStdout(`${JSON.stringify(summary, null, 2)}\n`);
|
||||
} else if (reports.length > 1 || runs > 1) {
|
||||
|
||||
@@ -28,12 +28,21 @@ interface ProgressReporter {
|
||||
interface DownloadResult {
|
||||
model: TinyLocalModelKey;
|
||||
ok: boolean;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
function writeLine(text = ""): void {
|
||||
process.stdout.write(`${text}\n`);
|
||||
}
|
||||
|
||||
function downloadErrorSummary(error: string | undefined): string | undefined {
|
||||
return error
|
||||
?.split(/\r?\n/)
|
||||
.map(line => line.trim())
|
||||
.find(line => line.length > 0)
|
||||
?.replace(/^Error:\s*/, "");
|
||||
}
|
||||
|
||||
export function resolveModels(model: string | undefined): TinyLocalModelKey[] {
|
||||
if (!model) return [DEFAULT_TINY_TITLE_LOCAL_MODEL_KEY];
|
||||
// `all` is a prefetch convenience: skip models that fail before load (unsupported
|
||||
@@ -101,10 +110,15 @@ async function downloadOne(modelKey: TinyLocalModelKey, json: boolean | undefine
|
||||
const label = getTinyLocalModelSpec(modelKey)?.label ?? modelKey;
|
||||
if (!json && !process.stdout.isTTY) writeLine(`Downloading ${label} (${modelKey})...`);
|
||||
const progress = makeProgressReporter(modelKey, json);
|
||||
const ok = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress });
|
||||
progress.finish(ok);
|
||||
if (!json && !process.stdout.isTTY) writeLine(ok ? `Downloaded ${label}.` : `Failed to download ${label}.`);
|
||||
return { model: modelKey, ok };
|
||||
const result = await tinyTitleClient.downloadModel(modelKey, { onProgress: progress.onProgress });
|
||||
progress.finish(result.ok);
|
||||
const error = downloadErrorSummary(result.error);
|
||||
if (!json && !process.stdout.isTTY) {
|
||||
writeLine(result.ok ? `Downloaded ${label}.` : `Failed to download ${label}${error ? `: ${error}` : ""}.`);
|
||||
} else if (!json && !result.ok && error) {
|
||||
writeLine(`${label} failed: ${error}`);
|
||||
}
|
||||
return result.error ? { model: modelKey, ok: result.ok, error: result.error } : { model: modelKey, ok: result.ok };
|
||||
}
|
||||
|
||||
export async function runTinyModelsCommand(command: TinyModelsCommandArgs): Promise<void> {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { Args, Command, Flags } from "@oh-my-pi/pi-utils/cli";
|
||||
import { runBenchCommand } from "../cli/bench-cli";
|
||||
import { SERVICE_TIER_SETTING_VALUES } from "../config/service-tier";
|
||||
import { SERVICE_TIER_OPENAI_VALUES } from "../config/service-tier";
|
||||
|
||||
export default class Bench extends Command {
|
||||
static description =
|
||||
@@ -19,8 +19,8 @@ export default class Bench extends Command {
|
||||
"max-tokens": Flags.integer({ description: "Max output tokens per request", default: 512 }),
|
||||
prompt: Flags.string({ description: "Custom prompt text (default: bundled bench prompt)" }),
|
||||
"service-tier": Flags.string({
|
||||
description: "Service tier hint (default: configured `serviceTier` setting; `none` omits it)",
|
||||
options: SERVICE_TIER_SETTING_VALUES,
|
||||
description: "Service tier applied per model family (default: configured `tier.*` settings; `none` omits it)",
|
||||
options: SERVICE_TIER_OPENAI_VALUES,
|
||||
}),
|
||||
json: Flags.boolean({ description: "Output JSON" }),
|
||||
par: Flags.integer({ description: "Execute runs with N parallel queries/requests", default: 4 }),
|
||||
|
||||
@@ -42,7 +42,7 @@ export async function runCommitAgentSession(input: CommitAgentInput): Promise<Co
|
||||
types_description: typesDescription,
|
||||
});
|
||||
const state: CommitAgentState = { diffText: input.diffText };
|
||||
const spawns = "quick_task";
|
||||
const spawns = "sonic";
|
||||
const tools = createCommitTools({
|
||||
cwd: input.cwd,
|
||||
authStorage: input.authStorage,
|
||||
|
||||
@@ -27,7 +27,7 @@ Tool guidance:
|
||||
- git_file_diff: diff for specific files
|
||||
- git_hunk: specific hunks for large diffs
|
||||
- recent_commits: recent commit subjects + style stats
|
||||
- analyze_files: spawn quick_task subagents in parallel for analysis
|
||||
- analyze_files: spawn sonic subagents in parallel for analysis
|
||||
- propose_changelog: provide changelog entries for each changelog target
|
||||
- propose_commit: submit final commit proposal and run validation
|
||||
- split_commit: propose multiple commit groups (no overlapping files; all staged files covered)
|
||||
|
||||
@@ -63,7 +63,7 @@ export function createAnalyzeFileTool(options: {
|
||||
return {
|
||||
name: "analyze_files",
|
||||
label: "Analyze Files",
|
||||
description: "Spawn quick_task agents to analyze files.",
|
||||
description: "Spawn sonic agents to analyze files.",
|
||||
parameters: analyzeFileSchema,
|
||||
async execute(toolCallId, params, _onUpdate, ctx, signal) {
|
||||
const toolSession = buildToolSession(ctx, options);
|
||||
@@ -83,7 +83,7 @@ export function createAnalyzeFileTool(options: {
|
||||
related_files: relatedFiles,
|
||||
});
|
||||
const taskParams: TaskParams = {
|
||||
agent: "quick_task",
|
||||
agent: "sonic",
|
||||
id: `AnalyzeFile${index + 1}`,
|
||||
description: `Analyze ${file}`,
|
||||
assignment,
|
||||
|
||||
@@ -22,7 +22,16 @@
|
||||
},
|
||||
"disabledServers": {
|
||||
"type": "array",
|
||||
"description": "User-level denylist for disabling discovered servers by name.",
|
||||
"description": "User-level denylist for disabling discovered servers by name. Highest precedence: a server here is hidden regardless of any other source.",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"minLength": 1
|
||||
},
|
||||
"uniqueItems": true
|
||||
},
|
||||
"enabledServers": {
|
||||
"type": "array",
|
||||
"description": "User-level allowlist that overrides a discovered server's `enabled: false` flag (e.g. when the source config is owned by another tool such as opencode.json). The denylist still wins.",
|
||||
"items": {
|
||||
"type": "string",
|
||||
"minLength": 1
|
||||
|
||||
@@ -1,87 +1,116 @@
|
||||
import type { ServiceTier } from "@oh-my-pi/pi-ai";
|
||||
import type { ServiceTier, ServiceTierByFamily } from "@oh-my-pi/pi-ai";
|
||||
import type { SubmenuOption } from "./settings-schema";
|
||||
|
||||
/**
|
||||
* Service-tier setting values shared by every "Service Tier" setting. `"none"`
|
||||
* is the omit-the-parameter sentinel; the remaining values mirror
|
||||
* {@link ServiceTier}.
|
||||
* Per-family service-tier setting values. `"none"` is the omit-the-parameter
|
||||
* sentinel; the rest mirror the wire {@link ServiceTier} values each provider
|
||||
* family actually realizes. OpenAI accepts the full set; Anthropic realizes
|
||||
* only `priority` (fast mode); Google (Gemini API + Vertex) realizes
|
||||
* `flex`/`priority`.
|
||||
*/
|
||||
export const SERVICE_TIER_SETTING_VALUES = [
|
||||
export const SERVICE_TIER_OPENAI_VALUES = ["none", "auto", "default", "flex", "scale", "priority"] as const;
|
||||
export const SERVICE_TIER_ANTHROPIC_VALUES = ["none", "priority"] as const;
|
||||
export const SERVICE_TIER_GOOGLE_VALUES = ["none", "flex", "priority"] as const;
|
||||
|
||||
export type ServiceTierOpenAISettingValue = (typeof SERVICE_TIER_OPENAI_VALUES)[number];
|
||||
export type ServiceTierAnthropicSettingValue = (typeof SERVICE_TIER_ANTHROPIC_VALUES)[number];
|
||||
export type ServiceTierGoogleSettingValue = (typeof SERVICE_TIER_GOOGLE_VALUES)[number];
|
||||
|
||||
/**
|
||||
* Inherit-capable single value for the subagent/advisor tiers. The chosen tier
|
||||
* is broadcast across families and applied to whichever family the spawned
|
||||
* model belongs to (clamped to what that family realizes); `"inherit"` defers
|
||||
* to the main agent's live per-family selection.
|
||||
*/
|
||||
export const SERVICE_TIER_INHERIT_SETTING_VALUES = [
|
||||
"inherit",
|
||||
"none",
|
||||
"auto",
|
||||
"default",
|
||||
"flex",
|
||||
"scale",
|
||||
"priority",
|
||||
"openai-only",
|
||||
"claude-only",
|
||||
] as const;
|
||||
|
||||
export type ServiceTierSettingValue = (typeof SERVICE_TIER_SETTING_VALUES)[number];
|
||||
|
||||
/** Variant value set for scoped service-tier settings (subagent/advisor) that can defer to the main agent. */
|
||||
export const SERVICE_TIER_INHERIT_SETTING_VALUES = ["inherit", ...SERVICE_TIER_SETTING_VALUES] as const;
|
||||
|
||||
export type ServiceTierInheritSettingValue = (typeof SERVICE_TIER_INHERIT_SETTING_VALUES)[number];
|
||||
|
||||
/** Submenu descriptions shared by the base `serviceTier` setting. */
|
||||
export const SERVICE_TIER_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierSettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Omit service_tier parameter" },
|
||||
{ value: "auto", label: "Auto", description: "Use provider default tier selection (OpenAI)" },
|
||||
{ value: "default", label: "Default", description: "Standard priority processing (OpenAI)" },
|
||||
{ value: "flex", label: "Flex", description: "Flexible capacity tier when available (OpenAI)" },
|
||||
{ value: "scale", label: "Scale", description: "Scale Tier credits when available (OpenAI)" },
|
||||
export const SERVICE_TIER_OPENAI_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierOpenAISettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Omit service_tier (standard processing)" },
|
||||
{ value: "auto", label: "Auto", description: "Provider default tier selection" },
|
||||
{ value: "default", label: "Default", description: "Standard priority processing" },
|
||||
{ value: "flex", label: "Flex", description: "Lower cost, higher latency when available" },
|
||||
{ value: "scale", label: "Scale", description: "Scale Tier credits when available" },
|
||||
{ value: "priority", label: "Priority", description: "Faster, higher cost (premium request)" },
|
||||
];
|
||||
|
||||
export const SERVICE_TIER_ANTHROPIC_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierAnthropicSettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Standard processing" },
|
||||
{
|
||||
value: "priority",
|
||||
label: "Priority",
|
||||
description: "Priority on every supported provider (OpenAI `service_tier`, Anthropic fast mode)",
|
||||
},
|
||||
{
|
||||
value: "openai-only",
|
||||
label: "Priority (OpenAI only)",
|
||||
description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere",
|
||||
},
|
||||
{
|
||||
value: "claude-only",
|
||||
label: "Priority (Claude only)",
|
||||
description: "Anthropic fast mode on direct Claude requests; ignored elsewhere (incl. Bedrock/Vertex)",
|
||||
description: 'Fast mode (`speed: "fast"`) on supported direct Claude models; ignored on Bedrock/Vertex',
|
||||
},
|
||||
];
|
||||
|
||||
/** Submenu descriptions for inherit-capable service-tier settings. */
|
||||
export const SERVICE_TIER_GOOGLE_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierGoogleSettingValue>> = [
|
||||
{ value: "none", label: "None", description: "Standard processing" },
|
||||
{ value: "flex", label: "Flex", description: "Lower cost, higher latency (Gemini API + Vertex)" },
|
||||
{ value: "priority", label: "Priority", description: "Faster, higher reliability (Gemini API + Vertex)" },
|
||||
];
|
||||
|
||||
export const SERVICE_TIER_INHERIT_OPTIONS: ReadonlyArray<SubmenuOption<ServiceTierInheritSettingValue>> = [
|
||||
{ value: "inherit", label: "Inherit", description: "Use the main agent's Service Tier" },
|
||||
...SERVICE_TIER_OPTIONS,
|
||||
{ value: "inherit", label: "Inherit", description: "Match the main agent's live per-family tiers" },
|
||||
{ value: "none", label: "None", description: "Standard processing" },
|
||||
{ value: "auto", label: "Auto", description: "Provider default tier selection (OpenAI family)" },
|
||||
{ value: "default", label: "Default", description: "Standard priority processing (OpenAI family)" },
|
||||
{ value: "flex", label: "Flex", description: "Flexible capacity tier (OpenAI/Google families)" },
|
||||
{ value: "scale", label: "Scale", description: "Scale Tier credits (OpenAI family)" },
|
||||
{ value: "priority", label: "Priority", description: "Priority on every supported family of the spawned model" },
|
||||
];
|
||||
|
||||
/**
|
||||
* Resolve a service-tier setting value to the wire {@link ServiceTier} (or
|
||||
* `undefined` to omit). `"inherit"` defers to `inherited`; `"none"` omits.
|
||||
*/
|
||||
export function resolveServiceTierSetting(value: string, inherited: ServiceTier | undefined): ServiceTier | undefined {
|
||||
if (value === "inherit") return inherited;
|
||||
if (value === "none" || value === "") return undefined;
|
||||
/** Map a per-family setting value to a wire {@link ServiceTier}, or `undefined` to omit. */
|
||||
export function serviceTierSettingToTier(value: string): ServiceTier | undefined {
|
||||
if (value === "none" || value === "" || value === "inherit") return undefined;
|
||||
return value as ServiceTier;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the `serviceTier` *setting value* to stamp onto a subagent's settings
|
||||
* snapshot.
|
||||
*
|
||||
* - A concrete `subagentSetting` (`"none"` or a tier) wins outright.
|
||||
* - `"inherit"` defers to the parent's live effective tier when the caller has a
|
||||
* live session (`inherited` passed as `ServiceTier | null`, where `null` means
|
||||
* the parent explicitly has no tier — e.g. `/fast off`). When no live session
|
||||
* is available (`inherited === undefined`, e.g. cold subagent revive) it falls
|
||||
* back to the parent's configured `serviceTier` setting so behavior matches a
|
||||
* plain settings snapshot.
|
||||
*/
|
||||
export function resolveSubagentServiceTier(
|
||||
subagentSetting: string,
|
||||
configuredTier: ServiceTierSettingValue,
|
||||
inherited: ServiceTier | null | undefined,
|
||||
): ServiceTierSettingValue {
|
||||
if (subagentSetting !== "inherit") return subagentSetting as ServiceTierSettingValue;
|
||||
if (inherited === undefined) return configuredTier;
|
||||
return inherited ?? "none";
|
||||
/** Assemble the live per-family tier map from the three `tier.*` setting values. */
|
||||
export function buildServiceTierByFamily(openai: string, anthropic: string, google: string): ServiceTierByFamily {
|
||||
const out: ServiceTierByFamily = {};
|
||||
const o = serviceTierSettingToTier(openai);
|
||||
if (o) out.openai = o;
|
||||
const a = serviceTierSettingToTier(anthropic);
|
||||
if (a) out.anthropic = a;
|
||||
const g = serviceTierSettingToTier(google);
|
||||
if (g) out.google = g;
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Broadcast a single chosen tier across families, clamped to what each family
|
||||
* realizes: OpenAI takes any tier, Anthropic only `priority`, Google only
|
||||
* `flex`/`priority`. Used by the subagent/advisor single-value settings and the
|
||||
* `omp bench --service-tier` flag, which apply one tier to whatever family the
|
||||
* target model belongs to.
|
||||
*/
|
||||
export function serviceTierForAllFamilies(tier: ServiceTier | undefined): ServiceTierByFamily {
|
||||
if (!tier) return {};
|
||||
const out: ServiceTierByFamily = { openai: tier };
|
||||
if (tier === "priority") out.anthropic = "priority";
|
||||
if (tier === "flex" || tier === "priority") out.google = tier;
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a subagent/advisor service-tier setting to a per-family map.
|
||||
*
|
||||
* - A concrete tier is broadcast across families (see
|
||||
* {@link serviceTierForAllFamilies}).
|
||||
* - `"none"` yields an empty map.
|
||||
* - `"inherit"` defers to `inherited` — the parent's live per-family tiers when
|
||||
* a live session supplied them, else the empty map.
|
||||
*/
|
||||
export function resolveSubagentServiceTier(setting: string, inherited: ServiceTierByFamily): ServiceTierByFamily {
|
||||
if (setting === "inherit") return inherited;
|
||||
return serviceTierForAllFamilies(serviceTierSettingToTier(setting));
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ import { DEFAULT_SHARE_URL } from "@oh-my-pi/pi-wire";
|
||||
import { SHAPE_VARIANT_NAMES } from "@oh-my-pi/snapcompact";
|
||||
import { DEFAULT_RELAY_URL } from "../collab/protocol";
|
||||
import { DEFAULT_STT_MODEL_KEY, STT_MODEL_OPTIONS, STT_MODEL_VALUES } from "../stt/models";
|
||||
import { STT_SUBMIT_TRIGGER_OPTIONS, STT_SUBMIT_TRIGGER_VALUES } from "../stt/submit-trigger";
|
||||
import { AUTO_THINKING, getConfiguredThinkingLevelMetadata, getThinkingLevelMetadata } from "../thinking";
|
||||
import {
|
||||
TINY_MODEL_DEVICE_DEFAULT,
|
||||
@@ -36,10 +37,14 @@ import {
|
||||
import { EDIT_MODES } from "../utils/edit-mode";
|
||||
import { SEARCH_PROVIDER_OPTIONS, SEARCH_PROVIDER_PREFERENCES, type SearchProviderId } from "../web/search/types";
|
||||
import {
|
||||
SERVICE_TIER_ANTHROPIC_OPTIONS,
|
||||
SERVICE_TIER_ANTHROPIC_VALUES,
|
||||
SERVICE_TIER_GOOGLE_OPTIONS,
|
||||
SERVICE_TIER_GOOGLE_VALUES,
|
||||
SERVICE_TIER_INHERIT_OPTIONS,
|
||||
SERVICE_TIER_INHERIT_SETTING_VALUES,
|
||||
SERVICE_TIER_OPTIONS,
|
||||
SERVICE_TIER_SETTING_VALUES,
|
||||
SERVICE_TIER_OPENAI_OPTIONS,
|
||||
SERVICE_TIER_OPENAI_VALUES,
|
||||
} from "./service-tier";
|
||||
|
||||
/** Unified settings schema - single source of truth for all settings.
|
||||
@@ -140,7 +145,7 @@ export const TAB_GROUPS: Record<SettingTab, readonly string[]> = {
|
||||
"Developer",
|
||||
],
|
||||
tasks: ["Modes", "Subagents", "Isolation", "Commands & Skills"],
|
||||
providers: ["Services", "Fireworks", "Tiny Model", "Protocol", "Privacy"],
|
||||
providers: ["Services", "Fireworks", "Tiny Model", "Protocol", "Timeouts", "Privacy"],
|
||||
};
|
||||
|
||||
/** Status line segment identifiers */
|
||||
@@ -1214,75 +1219,77 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
serviceTier: {
|
||||
"tier.openai": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_SETTING_VALUES,
|
||||
values: SERVICE_TIER_OPENAI_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier",
|
||||
label: "Service Tier — OpenAI",
|
||||
description:
|
||||
'Processing priority hint (none = omit). OpenAI accepts the tier values directly; Anthropic realizes `priority` as `speed: "fast"` on supported Opus models. Scoped values target one family.',
|
||||
options: SERVICE_TIER_OPTIONS,
|
||||
"Processing tier for OpenAI / OpenAI-Codex requests, and OpenAI-family models routed via OpenRouter (none = omit). Sent as `service_tier`.",
|
||||
options: SERVICE_TIER_OPENAI_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
serviceTierSubagent: {
|
||||
"tier.anthropic": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_ANTHROPIC_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier — Anthropic",
|
||||
description:
|
||||
'Processing tier for Claude requests. `priority` realizes fast mode (`speed: "fast"`) on supported direct Anthropic models; ignored on Bedrock/Vertex Claude and via OpenRouter.',
|
||||
options: SERVICE_TIER_ANTHROPIC_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
"tier.google": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_GOOGLE_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier — Google",
|
||||
description:
|
||||
"Processing tier for Gemini (Google AI Studio + Vertex) requests, and Google-family models routed via OpenRouter (none = omit). Sent as the top-level `serviceTier` field.",
|
||||
options: SERVICE_TIER_GOOGLE_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
"tier.subagent": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_INHERIT_SETTING_VALUES,
|
||||
default: "inherit",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier - Subagent",
|
||||
label: "Service Tier — Subagent",
|
||||
description:
|
||||
"Service Tier for spawned task/eval subagents. Inherit = match the main agent's live tier (tracks /fast); pick a value to scope subagents independently.",
|
||||
"Service Tier for spawned task/eval subagents. Inherit = match the main agent's live per-family tiers (tracks /fast); pick a value to apply it to whichever family the subagent's model belongs to.",
|
||||
options: SERVICE_TIER_INHERIT_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
serviceTierAdvisor: {
|
||||
"tier.advisor": {
|
||||
type: "enum",
|
||||
values: SERVICE_TIER_INHERIT_SETTING_VALUES,
|
||||
default: "none",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Service Tier - Advisor",
|
||||
label: "Service Tier — Advisor",
|
||||
description:
|
||||
"Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live tier; pick a value (e.g. Priority) to run the advisor on a faster serving path.",
|
||||
"Service Tier for the advisor model. None = standard processing; Inherit = match the main agent's live per-family tiers; pick a value to apply it to the advisor model's family.",
|
||||
options: SERVICE_TIER_INHERIT_OPTIONS,
|
||||
condition: "advisorEnabled",
|
||||
},
|
||||
},
|
||||
|
||||
fastModeScope: {
|
||||
type: "enum",
|
||||
values: ["both", "openai", "claude"] as const,
|
||||
default: "both",
|
||||
ui: {
|
||||
tab: "model",
|
||||
group: "Sampling",
|
||||
label: "Fast Mode Scope",
|
||||
description:
|
||||
'Which providers `/fast on` (and the fast-mode toggle) target. "both" = priority on every supported provider; "openai"/"claude" scope it to one family (mirrors serviceTier openai-only/claude-only).',
|
||||
options: [
|
||||
{ value: "both", label: "Both", description: "Priority on every supported provider" },
|
||||
{
|
||||
value: "openai",
|
||||
label: "OpenAI only",
|
||||
description: "Priority on OpenAI/OpenAI-Codex requests; ignored elsewhere",
|
||||
},
|
||||
{
|
||||
value: "claude",
|
||||
label: "Claude only",
|
||||
description: "Anthropic fast mode on direct Claude requests; ignored elsewhere",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
// Retries
|
||||
"retry.enabled": { type: "boolean", default: true },
|
||||
|
||||
@@ -1788,6 +1795,19 @@ export const SETTINGS_SCHEMA = {
|
||||
options: STT_MODEL_OPTIONS,
|
||||
},
|
||||
},
|
||||
"stt.submitTrigger": {
|
||||
type: "enum",
|
||||
values: STT_SUBMIT_TRIGGER_VALUES,
|
||||
default: "never",
|
||||
ui: {
|
||||
tab: "interaction",
|
||||
group: "Speech",
|
||||
label: "Speech-to-Text Submit Trigger",
|
||||
description:
|
||||
"Choose when speech dictation automatically submits: Never, Release (2+ words), Release with complete sentence, or When I Say Submit.",
|
||||
options: STT_SUBMIT_TRIGGER_OPTIONS,
|
||||
},
|
||||
},
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────
|
||||
// Context
|
||||
@@ -4075,7 +4095,7 @@ export const SETTINGS_SCHEMA = {
|
||||
group: "Subagents",
|
||||
label: "Soft Subagent Request Budget",
|
||||
description:
|
||||
"Soft per-subagent request budget (assistant requests per run). Crossing it injects one steering notice asking the subagent to wrap up; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled explore/quick_task agents use a lower built-in budget.",
|
||||
"Soft per-subagent request budget (assistant requests per run). Crossing it injects one steering notice asking the subagent to wrap up; at 1.5x the budget the run is aborted gracefully, salvaging partial output. 0 disables the guard. Bundled explore/sonic agents use a lower built-in budget.",
|
||||
options: [
|
||||
{ value: "0", label: "Disabled" },
|
||||
{ value: "40", label: "40 requests" },
|
||||
@@ -4548,6 +4568,44 @@ export const SETTINGS_SCHEMA = {
|
||||
},
|
||||
},
|
||||
|
||||
"providers.streamFirstEventTimeoutSeconds": {
|
||||
type: "number",
|
||||
default: -1,
|
||||
ui: {
|
||||
tab: "providers",
|
||||
group: "Timeouts",
|
||||
label: "Stream First Event Timeout",
|
||||
description:
|
||||
"Seconds to wait for the first model stream event; -1 uses provider/env defaults, 0 disables the watchdog",
|
||||
options: [
|
||||
{ value: "-1", label: "Auto", description: "Use provider defaults and PI_* timeout env vars" },
|
||||
{ value: "0", label: "Off", description: "Disable first-event timeout" },
|
||||
{ value: "300", label: "5 minutes" },
|
||||
{ value: "600", label: "10 minutes" },
|
||||
{ value: "1800", label: "30 minutes" },
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
"providers.streamIdleTimeoutSeconds": {
|
||||
type: "number",
|
||||
default: -1,
|
||||
ui: {
|
||||
tab: "providers",
|
||||
group: "Timeouts",
|
||||
label: "Stream Idle Timeout",
|
||||
description:
|
||||
"Seconds a model stream may stay silent between events; -1 uses provider/env defaults, 0 disables the watchdog",
|
||||
options: [
|
||||
{ value: "-1", label: "Auto", description: "Use provider defaults and PI_* timeout env vars" },
|
||||
{ value: "0", label: "Off", description: "Disable idle timeout" },
|
||||
{ value: "300", label: "5 minutes" },
|
||||
{ value: "600", label: "10 minutes" },
|
||||
{ value: "1800", label: "30 minutes" },
|
||||
],
|
||||
},
|
||||
},
|
||||
|
||||
"providers.openrouterVariant": {
|
||||
type: "enum",
|
||||
values: ["default", "nitro", "floor", "online", "exacto"] as const,
|
||||
|
||||
@@ -1148,6 +1148,53 @@ export class Settings {
|
||||
// the incoherent "hashline edits without addressable anchors" state.
|
||||
delete raw.readHashLines;
|
||||
|
||||
// serviceTier (single enum with scoped openai-only/claude-only sentinels)
|
||||
// → per-family tier.openai/tier.anthropic/tier.google; serviceTierSubagent
|
||||
// → tier.subagent; serviceTierAdvisor → tier.advisor. `fastModeScope` is
|
||||
// dropped — per-family scoping is now expressed by the three tier settings.
|
||||
const tierObj = isRecord(raw.tier) ? raw.tier : {};
|
||||
let tierTouched = false;
|
||||
const setTier = (family: string, value: unknown): void => {
|
||||
if (value !== undefined && !(family in tierObj)) {
|
||||
tierObj[family] = value;
|
||||
tierTouched = true;
|
||||
}
|
||||
};
|
||||
if (typeof raw.serviceTier === "string") {
|
||||
switch (raw.serviceTier) {
|
||||
case "priority":
|
||||
setTier("openai", "priority");
|
||||
setTier("anthropic", "priority");
|
||||
setTier("google", "priority");
|
||||
break;
|
||||
case "openai-only":
|
||||
setTier("openai", "priority");
|
||||
break;
|
||||
case "claude-only":
|
||||
setTier("anthropic", "priority");
|
||||
break;
|
||||
case "auto":
|
||||
case "default":
|
||||
case "flex":
|
||||
case "scale":
|
||||
setTier("openai", raw.serviceTier);
|
||||
break;
|
||||
}
|
||||
delete raw.serviceTier;
|
||||
}
|
||||
const mapInheritTier = (value: unknown): unknown =>
|
||||
value === "openai-only" || value === "claude-only" ? "priority" : value;
|
||||
if ("serviceTierSubagent" in raw) {
|
||||
setTier("subagent", mapInheritTier(raw.serviceTierSubagent));
|
||||
delete raw.serviceTierSubagent;
|
||||
}
|
||||
if ("serviceTierAdvisor" in raw) {
|
||||
setTier("advisor", mapInheritTier(raw.serviceTierAdvisor));
|
||||
delete raw.serviceTierAdvisor;
|
||||
}
|
||||
if (tierTouched) raw.tier = tierObj;
|
||||
delete raw.fastModeScope;
|
||||
|
||||
return raw;
|
||||
}
|
||||
|
||||
|
||||
@@ -114,7 +114,13 @@ function formatProfileAsMarkdown(profileJson: string): string {
|
||||
*/
|
||||
export async function startCpuProfile(): Promise<ProfilerSession> {
|
||||
const v8 = await import("node:v8");
|
||||
v8.setFlagsFromString("--allow-natives-syntax");
|
||||
try {
|
||||
// Enables `%GetOptimizationStatus` and friends when V8 natives are needed
|
||||
// for ad-hoc profiling. Best-effort: Bun does not implement
|
||||
// `setFlagsFromString` (oven-sh/bun#1702) but the CPU profiler itself
|
||||
// works without it, so swallow the error and continue.
|
||||
v8.setFlagsFromString("--allow-natives-syntax");
|
||||
} catch {}
|
||||
|
||||
const { Session } = await import("node:inspector/promises");
|
||||
const session = new Session();
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
---
|
||||
description: "Prefer runtime.AddCleanup over runtime.SetFinalizer for new code (Go 1.24)"
|
||||
condition: 'runtime\.SetFinalizer'
|
||||
scope: "tool:edit(*.go), tool:write(*.go)"
|
||||
---
|
||||
|
||||
Go 1.24 added `runtime.AddCleanup`, a finalization mechanism that is more flexible and less error-prone than `runtime.SetFinalizer`. The release notes state plainly: **new code should prefer `AddCleanup` over `SetFinalizer`.**
|
||||
|
||||
## Why AddCleanup wins
|
||||
|
||||
- Multiple cleanups may attach to one object; `SetFinalizer` allows only one.
|
||||
- Cleanups may attach to interior pointers.
|
||||
- Objects that form a reference cycle still get cleaned up — finalizers leak them.
|
||||
- A cleanup does not resurrect its object or delay freeing it (and what it points to) by an extra GC cycle.
|
||||
|
||||
## Migration
|
||||
|
||||
```go
|
||||
// Before
|
||||
runtime.SetFinalizer(obj, func(o *T) { o.release() })
|
||||
|
||||
// After — the cleanup func receives a value you supply, NOT the object,
|
||||
// so it cannot accidentally keep the object alive.
|
||||
runtime.AddCleanup(obj, func(h handle) { h.release() }, obj.handle)
|
||||
```
|
||||
|
||||
The cleanup argument must not reference `obj` itself (that would keep it reachable forever). Capture only the data the cleanup needs — a file descriptor, handle, or pointer that is independent of `obj`.
|
||||
|
||||
## Keep SetFinalizer only when
|
||||
|
||||
- The module targets a Go release older than 1.24.
|
||||
- You depend on finalizer-specific behavior (e.g. object resurrection) that `AddCleanup` deliberately does not provide.
|
||||
@@ -0,0 +1,36 @@
|
||||
---
|
||||
description: "Use for b.Loop() in benchmarks instead of the for i := 0; i < b.N; i++ loop (Go 1.24)"
|
||||
interruptMode: never
|
||||
scope: "tool:edit(*_test.go), tool:write(*_test.go)"
|
||||
astCondition:
|
||||
- "func $F($B *testing.B) { $$$PRE for $I := 0; $I < $B.N; $I++ { $$$BODY } $$$POST }"
|
||||
---
|
||||
|
||||
Go 1.24 added `testing.B.Loop`. Write `for b.Loop() { ... }` instead of looping over `b.N`.
|
||||
|
||||
## Why
|
||||
|
||||
- Setup and teardown outside the loop run exactly once per `-count`, not once per `b.N` re-estimation, so expensive fixtures are no longer timed or repeated.
|
||||
- The compiler keeps the loop's parameters and results alive, so it can't optimize away the body you are trying to measure — a classic `b.N` benchmarking footgun.
|
||||
|
||||
## Avoid
|
||||
|
||||
```go
|
||||
func BenchmarkEncode(b *testing.B) {
|
||||
for i := 0; i < b.N; i++ {
|
||||
Encode(input)
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Use
|
||||
|
||||
```go
|
||||
func BenchmarkEncode(b *testing.B) {
|
||||
for b.Loop() {
|
||||
Encode(input)
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Requires Go 1.24+. If the module targets an older Go, keep the `b.N` loop.
|
||||
@@ -0,0 +1,39 @@
|
||||
---
|
||||
description: "Use the standard library slices and maps packages instead of golang.org/x/exp/{slices,maps}"
|
||||
condition:
|
||||
- '"golang.org/x/exp/slices"'
|
||||
- '"golang.org/x/exp/maps"'
|
||||
scope: "tool:edit(*.go), tool:write(*.go)"
|
||||
---
|
||||
|
||||
`golang.org/x/exp/slices` and `golang.org/x/exp/maps` were promoted into the standard library as `slices` and `maps` in Go 1.21. Import the stdlib packages in new code instead of the experimental ones.
|
||||
|
||||
## Migration
|
||||
|
||||
```go
|
||||
// Before
|
||||
import (
|
||||
"golang.org/x/exp/slices"
|
||||
"golang.org/x/exp/maps"
|
||||
)
|
||||
|
||||
// After
|
||||
import (
|
||||
"slices"
|
||||
"maps"
|
||||
)
|
||||
```
|
||||
|
||||
Most call sites are unchanged: `slices.Sort`, `slices.Contains`, `slices.Index`, `slices.Equal`, `maps.Clone`, etc.
|
||||
|
||||
## Watch the signature differences
|
||||
|
||||
The promoted APIs were tweaked, so a blind path swap can break the build:
|
||||
|
||||
- `x/exp/maps.Keys(m)` / `Values(m)` returned a slice; the stdlib `maps.Keys(m)` / `maps.Values(m)` return an **iterator** (`iter.Seq`). Use `slices.Collect(maps.Keys(m))` to recover a slice, or range over the iterator.
|
||||
- `slices.SortFunc` takes a comparison returning `int` (cmp-style), matching the stdlib signature.
|
||||
|
||||
## Keep x/exp when
|
||||
|
||||
- The module's `go` directive is below 1.21 (stdlib `slices`/`maps` don't exist yet).
|
||||
- You need an `x/exp` helper that was not promoted (e.g. parts of `x/exp/constraints` still live outside the stdlib).
|
||||
@@ -0,0 +1,36 @@
|
||||
---
|
||||
description: "Use io and os instead of the deprecated io/ioutil package"
|
||||
condition: '"io/ioutil"'
|
||||
scope: "tool:edit(*.go), tool:write(*.go)"
|
||||
---
|
||||
|
||||
`io/ioutil` has been deprecated since Go 1.16. Every function moved to `io` or `os` with the same behavior. Do not import it in new code.
|
||||
|
||||
## Mapping
|
||||
|
||||
| io/ioutil | Replacement |
|
||||
| --- | --- |
|
||||
| `ioutil.ReadAll` | `io.ReadAll` |
|
||||
| `ioutil.ReadFile` | `os.ReadFile` |
|
||||
| `ioutil.WriteFile` | `os.WriteFile` |
|
||||
| `ioutil.ReadDir` | `os.ReadDir` (returns `[]os.DirEntry`, not `[]os.FileInfo`) |
|
||||
| `ioutil.TempFile` | `os.CreateTemp` |
|
||||
| `ioutil.TempDir` | `os.MkdirTemp` |
|
||||
| `ioutil.NopCloser` | `io.NopCloser` |
|
||||
| `ioutil.Discard` | `io.Discard` |
|
||||
|
||||
## Migration
|
||||
|
||||
```go
|
||||
// Before
|
||||
import "io/ioutil"
|
||||
data, err := ioutil.ReadFile(path)
|
||||
_ = ioutil.WriteFile(out, data, 0o644)
|
||||
|
||||
// After
|
||||
import "os"
|
||||
data, err := os.ReadFile(path)
|
||||
_ = os.WriteFile(out, data, 0o644)
|
||||
```
|
||||
|
||||
`os.ReadDir` returns `[]os.DirEntry` rather than `[]os.FileInfo` — call `entry.Info()` if you need the old `FileInfo`. Everything else is a drop-in rename.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
description: "Build network addresses with net.JoinHostPort, not fmt.Sprintf(\"%s:%d\", host, port) — the Sprintf form breaks on IPv6"
|
||||
condition: 'fmt\.Sprintf\("%s:%d"'
|
||||
scope: "tool:edit(*.go), tool:write(*.go)"
|
||||
---
|
||||
|
||||
Use `net.JoinHostPort(host, port)` to assemble a `host:port` address. `fmt.Sprintf("%s:%d", host, port)` produces invalid addresses for IPv6 hosts, which must be bracketed (`[::1]:80`). Go 1.25's `go vet` `hostport` analyzer flags exactly this pattern.
|
||||
|
||||
## Why
|
||||
|
||||
- An IPv6 literal like `::1` has its own colons, so `fmt.Sprintf("%s:%d", "::1", 80)` yields `::1:80` — unparseable by `net.Dial`.
|
||||
- `net.JoinHostPort` adds the brackets when the host contains a colon and leaves IPv4/hostnames untouched.
|
||||
|
||||
## Avoid
|
||||
|
||||
```go
|
||||
addr := fmt.Sprintf("%s:%d", host, port)
|
||||
conn, err := net.Dial("tcp", addr)
|
||||
```
|
||||
|
||||
## Use
|
||||
|
||||
```go
|
||||
// port is a string here; convert an int with strconv.Itoa.
|
||||
addr := net.JoinHostPort(host, strconv.Itoa(port))
|
||||
conn, err := net.Dial("tcp", addr)
|
||||
```
|
||||
|
||||
`net.JoinHostPort` takes the port as a string. For an `int` port, wrap it in `strconv.Itoa`. The function is available in every supported Go version.
|
||||
@@ -0,0 +1,44 @@
|
||||
---
|
||||
description: "Use new(expr) for pointer-to-value helpers instead of `func ptr[T any](v T) *T { return &v }` (Go 1.26)"
|
||||
interruptMode: never
|
||||
scope: "tool:edit(*.go), tool:write(*.go)"
|
||||
astCondition:
|
||||
- "func $F($V $T) *$T { return &$V }"
|
||||
- "func $F[$$$TP]($V $T) *$T { return &$V }"
|
||||
---
|
||||
|
||||
Go 1.26 lets `new` take an expression: `new(expr)` allocates, stores `expr`, and returns its `*T`. That removes the need for hand-written `Ptr`/`boolPtr`/`Int64`-style helpers and the `x := v; p := &x` two-step.
|
||||
|
||||
## Why
|
||||
|
||||
- One builtin replaces a helper per type (`boolPtr`, `strPtr`, `int64Ptr`, …) and the generic `func Ptr[T any](v T) *T`.
|
||||
- No extra function-call frame and no separate heap escape — the value is constructed directly in the allocation.
|
||||
- The intent (`new(false)`) reads at the call site instead of hiding behind a helper name.
|
||||
|
||||
## Avoid
|
||||
|
||||
```go
|
||||
// A helper that just takes a value and returns its address.
|
||||
func boolPtr(v bool) *bool { return &v }
|
||||
func strPtr(v string) *string { return &v }
|
||||
func Ptr[T any](v T) *T { return &v }
|
||||
|
||||
cfg := Config{Enabled: boolPtr(true), Name: strPtr("svc")}
|
||||
```
|
||||
|
||||
## Use
|
||||
|
||||
```go
|
||||
cfg := Config{Enabled: new(true), Name: new("svc")}
|
||||
|
||||
// Was: x := int64(300); p := &x
|
||||
p := new(int64(300))
|
||||
```
|
||||
|
||||
`new(true)` / `new(false)` give you `*bool`; `new(expr)` works for any expression, including function results (`new(time.Now())`).
|
||||
|
||||
## Notes
|
||||
|
||||
- Requires Go 1.26+. If the module's `go` directive is older, keep the helper or the temp-variable form until the toolchain is bumped.
|
||||
- This is for helpers that *only* take a value and return its address. A function that does real work before taking an address is not in scope.
|
||||
- `new(T)` (a bare type) is unchanged and still zero-initializes.
|
||||
@@ -0,0 +1,40 @@
|
||||
---
|
||||
description: Prefer math/rand/v2 over the legacy math/rand package
|
||||
condition: '"math/rand"'
|
||||
scope: "tool:edit(*.go), tool:write(*.go)"
|
||||
---
|
||||
|
||||
Use `math/rand/v2` instead of the legacy `math/rand` package (stable since Go 1.22).
|
||||
|
||||
## Why
|
||||
|
||||
- No global `Seed`: `math/rand`'s top-level functions read a process-global generator (auto-seeded since Go 1.20), so a fixed seed is global mutable state that's easy to misuse; `v2` drops the global `Seed` entirely.
|
||||
- Cleaner, better-bounded API: `rand.IntN(n)` / generic `rand.N(n)` replace `rand.Intn(n)`, and `Shuffle`, `Perm`, `Float64` carry over with clearer names.
|
||||
- Modern generators: `v2` exposes `PCG` and `ChaCha8` sources instead of the old default LCG.
|
||||
|
||||
## Migration
|
||||
|
||||
```go
|
||||
// Before
|
||||
import "math/rand"
|
||||
n := rand.Intn(100)
|
||||
f := rand.Float64()
|
||||
|
||||
// After
|
||||
import "math/rand/v2"
|
||||
n := rand.IntN(100)
|
||||
f := rand.Float64()
|
||||
```
|
||||
|
||||
| math/rand | math/rand/v2 |
|
||||
| --- | --- |
|
||||
| `rand.Intn(n)` | `rand.IntN(n)` |
|
||||
| `rand.Int63n(n)` | `rand.Int64N(n)` |
|
||||
| `rand.Intn`/`Int31n` on a `*Rand` | `(*Rand).IntN` / `Int32N` |
|
||||
| `rand.Seed(x)` | drop it — `v2` has no global seed |
|
||||
| explicit `rand.New(rand.NewSource(seed))` | `rand.New(rand.NewPCG(s1, s2))` or `rand.NewChaCha8(seed)` |
|
||||
|
||||
## Keep math/rand only when
|
||||
|
||||
- You need a reproducible stream from a fixed seed via the classic `NewSource`/`Seed` API that a caller already depends on.
|
||||
- Reach for `crypto/rand` instead when the values are security-sensitive — neither `math/rand` variant is cryptographically secure.
|
||||
@@ -0,0 +1,45 @@
|
||||
---
|
||||
description: "Use for i := range n instead of the C-style for i := 0; i < n; i++ loop (Go 1.22)"
|
||||
interruptMode: never
|
||||
scope: "tool:edit(*.go), tool:write(*.go)"
|
||||
astCondition:
|
||||
- "for $I := 0; $I < $N; $I++ { $$$BODY }"
|
||||
---
|
||||
|
||||
Go 1.22 lets `for` range over an integer. A plain counting loop from `0` to `n` with step `1` reads better as `for i := range n` (or `for range n` when the index is unused).
|
||||
|
||||
## Avoid
|
||||
|
||||
```go
|
||||
for i := 0; i < n; i++ {
|
||||
use(i)
|
||||
}
|
||||
|
||||
for i := 0; i < len(s); i++ {
|
||||
use(s[i])
|
||||
}
|
||||
```
|
||||
|
||||
## Use
|
||||
|
||||
```go
|
||||
for i := range n {
|
||||
use(i)
|
||||
}
|
||||
|
||||
// Ranging the slice directly is usually clearer than indexing.
|
||||
for i := range s {
|
||||
use(s[i])
|
||||
}
|
||||
|
||||
// Index unused → drop it entirely.
|
||||
for range n {
|
||||
tick()
|
||||
}
|
||||
```
|
||||
|
||||
## When it does not apply
|
||||
|
||||
- Non-zero start, step other than `++`, or a descending loop (`for i := n - 1; i >= 0; i--`) — keep the explicit form.
|
||||
- The body reassigns the loop variable or depends on `i` surviving past the loop.
|
||||
- Requires Go 1.22+. If the module's `go` directive is older, keep the classic loop.
|
||||
@@ -8,6 +8,14 @@
|
||||
* Registered by the lowest-priority `builtin-defaults` rule provider so any
|
||||
* user/project/tool rule with the same name overrides the bundled copy.
|
||||
*/
|
||||
import goAddCleanup from "./go-add-cleanup.md" with { type: "text" };
|
||||
import goBenchLoop from "./go-bench-loop.md" with { type: "text" };
|
||||
import goExpPromoted from "./go-exp-promoted.md" with { type: "text" };
|
||||
import goIoutil from "./go-ioutil.md" with { type: "text" };
|
||||
import goJoinHostport from "./go-join-hostport.md" with { type: "text" };
|
||||
import goNewExpr from "./go-new-expr.md" with { type: "text" };
|
||||
import goRandV2 from "./go-rand-v2.md" with { type: "text" };
|
||||
import goRangeInt from "./go-range-int.md" with { type: "text" };
|
||||
import rsBoxLeak from "./rs-box-leak.md" with { type: "text" };
|
||||
import rsFuturePrelude from "./rs-future-prelude.md" with { type: "text" };
|
||||
import rsLazylock from "./rs-lazylock.md" with { type: "text" };
|
||||
@@ -35,6 +43,14 @@ export interface BuiltinRuleSource {
|
||||
|
||||
/** All bundled default rules, ordered by name. */
|
||||
export const BUILTIN_RULE_SOURCES: readonly BuiltinRuleSource[] = [
|
||||
{ name: "go-add-cleanup", content: goAddCleanup },
|
||||
{ name: "go-bench-loop", content: goBenchLoop },
|
||||
{ name: "go-exp-promoted", content: goExpPromoted },
|
||||
{ name: "go-ioutil", content: goIoutil },
|
||||
{ name: "go-join-hostport", content: goJoinHostport },
|
||||
{ name: "go-new-expr", content: goNewExpr },
|
||||
{ name: "go-rand-v2", content: goRandV2 },
|
||||
{ name: "go-range-int", content: goRangeInt },
|
||||
{ name: "rs-box-leak", content: rsBoxLeak },
|
||||
{ name: "rs-future-prelude", content: rsFuturePrelude },
|
||||
{ name: "rs-lazylock", content: rsLazylock },
|
||||
|
||||
@@ -28,6 +28,7 @@ import { invalidateFsScanAfterWrite } from "../../tools/fs-cache-invalidation";
|
||||
import { isInternalUrlPath } from "../../tools/path-utils";
|
||||
import { enforcePlanModeWrite, resolvePlanPath, targetsLocalSandbox } from "../../tools/plan-mode-guard";
|
||||
import { canonicalSnapshotKey } from "../file-snapshot-store";
|
||||
import { isNotebookPath } from "../notebook";
|
||||
import { readEditFileText, serializeEditFileText } from "../read-file";
|
||||
import type { LspBatchRequest } from "../renderer";
|
||||
|
||||
@@ -123,6 +124,17 @@ export class HashlineFilesystem extends Filesystem {
|
||||
return content;
|
||||
}
|
||||
|
||||
async readBinary(relativePath: string): Promise<Uint8Array | undefined> {
|
||||
const absolutePath = this.resolveAbsolute(relativePath);
|
||||
if (isNotebookPath(absolutePath)) return undefined;
|
||||
try {
|
||||
return await fs.readFile(absolutePath);
|
||||
} catch (error) {
|
||||
if (isEnoent(error)) throw new NotFoundError(relativePath, error);
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
async preflightWrite(relativePath: string, options?: PreflightWriteOptions): Promise<void> {
|
||||
const fileOp = options?.fileOp;
|
||||
if (fileOp?.kind === "rem") {
|
||||
|
||||
@@ -405,8 +405,10 @@ export async function runEvalAgent(args: unknown, options: EvalAgentBridgeOption
|
||||
parentMnemopiSessionState: options.session.getMnemopiSessionState?.(),
|
||||
parentTelemetry: options.session.getTelemetry?.(),
|
||||
parentAgentId: options.session.getAgentId?.() ?? MAIN_AGENT_ID,
|
||||
// Live source of truth for `serviceTierSubagent: inherit` (null = explicit none).
|
||||
parentServiceTier: options.session.getServiceTier ? (options.session.getServiceTier() ?? null) : undefined,
|
||||
// Live source of truth for `tier.subagent: inherit` (null = explicit none).
|
||||
parentServiceTier: options.session.getServiceTierByFamily
|
||||
? (options.session.getServiceTierByFamily() ?? null)
|
||||
: undefined,
|
||||
// Deliberately omit parentEvalSessionId: the parent's Python kernel is
|
||||
// blocked on this bridge call, so sharing the eval session would deadlock
|
||||
// (subagent queues behind the parent's in-flight execution, parent waits
|
||||
|
||||
@@ -140,7 +140,7 @@ const HOST_DEFAULTED_SETTING_PATHS: SettingPath[] = [
|
||||
"advisor.subagents",
|
||||
"advisor.syncBacklog",
|
||||
"advisor.immuneTurns",
|
||||
"serviceTierAdvisor",
|
||||
"tier.advisor",
|
||||
];
|
||||
|
||||
const RPC_BACKGROUND_DEFAULTED_SETTING_PATHS: SettingPath[] = [
|
||||
|
||||
@@ -227,3 +227,124 @@ export async function setServerDisabled(filePath: string, name: string, disabled
|
||||
|
||||
await writeMCPConfigFile(filePath, updated);
|
||||
}
|
||||
|
||||
/**
|
||||
* Read the user-level force-enable list (allowlist that overrides a
|
||||
* non-writable source config's `enabled: false`).
|
||||
*/
|
||||
export async function readEnabledServers(filePath: string): Promise<string[]> {
|
||||
const config = await readMCPConfigFile(filePath);
|
||||
return Array.isArray(config.enabledServers) ? config.enabledServers : [];
|
||||
}
|
||||
|
||||
/**
|
||||
* Add or remove a server name from the user-level force-enable list.
|
||||
* The list overrides a discovered server's `enabled: false` flag but does
|
||||
* NOT override the `disabledServers` denylist.
|
||||
*/
|
||||
export async function setServerForceEnabled(filePath: string, name: string, force: boolean): Promise<void> {
|
||||
const config = await readMCPConfigFile(filePath);
|
||||
const current = new Set(config.enabledServers ?? []);
|
||||
|
||||
if (force) {
|
||||
current.add(name);
|
||||
} else {
|
||||
current.delete(name);
|
||||
}
|
||||
|
||||
const updated: MCPConfigFile = {
|
||||
...config,
|
||||
enabledServers: current.size > 0 ? Array.from(current).sort() : undefined,
|
||||
};
|
||||
|
||||
if (!updated.enabledServers) {
|
||||
delete updated.enabledServers;
|
||||
}
|
||||
|
||||
await writeMCPConfigFile(filePath, updated);
|
||||
}
|
||||
|
||||
/** Paths and target state for toggling one MCP server across known config files. */
|
||||
export interface SetMcpServerEnabledOptions {
|
||||
userPath: string;
|
||||
projectPath: string;
|
||||
/**
|
||||
* Absolute path to the loaded row's source mcp.json. Provide ONLY for
|
||||
* formats this codebase owns (native `.omp/mcp.json` and `mcp-json`
|
||||
* `mcp.json`/`.mcp.json`). Tool-owned configs (opencode.json, claude.json,
|
||||
* settings.json …) MUST be omitted; we never mutate another tool's file.
|
||||
*/
|
||||
sourcePath?: string;
|
||||
name: string;
|
||||
enabled: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Flip a server's enabled/disabled state regardless of where it lives.
|
||||
*
|
||||
* Resolution order, mirroring `/mcp enable` / `/mcp disable` plus the dashboard
|
||||
* fix for non-writable source configs:
|
||||
*
|
||||
* - Server found in `sourcePath` (writable) → write `enabled` on that entry.
|
||||
* - Else server in project mcp.json → write `enabled` there.
|
||||
* - Else server in user mcp.json → write `enabled` there.
|
||||
* - Else (server defined in a tool-owned source like opencode.json, OR a
|
||||
* purely discovered server):
|
||||
* - Disable → add to the user-level `disabledServers` denylist.
|
||||
* - Enable → add to the user-level `enabledServers` allowlist so the
|
||||
* dashboard / runtime override the non-writable source's
|
||||
* `enabled: false` flag.
|
||||
*
|
||||
* Cleanup invariants — on every call:
|
||||
* - Re-enable clears any stale denylist entry so a server disabled via
|
||||
* `/mcp disable` and re-enabled here doesn't stay suppressed.
|
||||
* - Disable clears any stale allowlist entry so re-disabling a
|
||||
* force-enabled server actually takes effect.
|
||||
*/
|
||||
export async function setMcpServerEnabled(options: SetMcpServerEnabledOptions): Promise<void> {
|
||||
const { userPath, projectPath, sourcePath, name, enabled } = options;
|
||||
const candidatePaths = [...new Set([sourcePath, projectPath, userPath].filter(path => path !== undefined))];
|
||||
let updatedInConfig = false;
|
||||
|
||||
for (const filePath of candidatePaths) {
|
||||
const config = await readMCPConfigFile(filePath);
|
||||
const server = config.mcpServers?.[name];
|
||||
if (server === undefined) continue;
|
||||
|
||||
await updateMCPServer(filePath, name, { ...server, enabled });
|
||||
updatedInConfig = true;
|
||||
break;
|
||||
}
|
||||
|
||||
if (enabled) {
|
||||
// Either we just wrote `enabled: true` on a writable source, or the
|
||||
// server lives in a non-writable source whose `enabled: false` flag we
|
||||
// need to override via the user allowlist. Either way the denylist
|
||||
// entry (if any) must clear so the row becomes active.
|
||||
const denied = await readDisabledServers(userPath);
|
||||
if (denied.includes(name)) {
|
||||
await setServerDisabled(userPath, name, false);
|
||||
}
|
||||
|
||||
const forced = await readEnabledServers(userPath);
|
||||
const isForced = forced.includes(name);
|
||||
if (!updatedInConfig && !isForced) {
|
||||
await setServerForceEnabled(userPath, name, true);
|
||||
} else if (updatedInConfig && isForced) {
|
||||
// Writable source now carries `enabled: true`; the override is
|
||||
// redundant. Drop it so the user's allowlist stays tidy.
|
||||
await setServerForceEnabled(userPath, name, false);
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Disable path. Clear any force-enable override regardless of source so the
|
||||
// disable actually sticks.
|
||||
const forced = await readEnabledServers(userPath);
|
||||
if (forced.includes(name)) {
|
||||
await setServerForceEnabled(userPath, name, false);
|
||||
}
|
||||
if (!updatedInConfig) {
|
||||
await setServerDisabled(userPath, name, true);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,7 +9,7 @@ import { mcpCapability } from "../capability/mcp";
|
||||
import type { SourceMeta } from "../capability/types";
|
||||
import type { MCPServer } from "../discovery";
|
||||
import { loadCapability } from "../discovery";
|
||||
import { readDisabledServers } from "./config-writer";
|
||||
import { readDisabledServers, readEnabledServers } from "./config-writer";
|
||||
import type { MCPServerConfig } from "./types";
|
||||
|
||||
/** Options for loading MCP configs */
|
||||
@@ -105,16 +105,20 @@ export async function loadAllMCPConfigs(cwd: string, options?: LoadMCPConfigsOpt
|
||||
? result.items
|
||||
: result.items.filter(server => server._source.level !== "project");
|
||||
|
||||
// Load user-level disabled servers list
|
||||
const disabledServers = new Set(await readDisabledServers(getMCPConfigPath("user", cwd)));
|
||||
// Load user-level disable/force-enable lists. The denylist always wins; the
|
||||
// allowlist overrides a non-writable source config's `enabled: false`.
|
||||
const userPath = getMCPConfigPath("user", cwd);
|
||||
const [disabledServers, forcedEnabled] = await Promise.all([
|
||||
readDisabledServers(userPath).then(list => new Set(list)),
|
||||
readEnabledServers(userPath).then(list => new Set(list)),
|
||||
]);
|
||||
// Convert to legacy format and preserve source metadata
|
||||
let configs: Record<string, MCPServerConfig> = {};
|
||||
let sources: Record<string, SourceMeta> = {};
|
||||
for (const server of servers) {
|
||||
const config = convertToLegacyConfig(server);
|
||||
if (config.enabled === false || disabledServers.has(server.name)) {
|
||||
continue;
|
||||
}
|
||||
if (disabledServers.has(server.name)) continue;
|
||||
if (config.enabled === false && !forcedEnabled.has(server.name)) continue;
|
||||
configs[server.name] = config;
|
||||
sources[server.name] = server._source;
|
||||
}
|
||||
|
||||
@@ -153,14 +153,44 @@ function resolveCallbackHostname(redirectUri: string | undefined): string | unde
|
||||
return parsed.hostname;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the client_id MCPOAuthFlow would use without doing any I/O —
|
||||
* either the explicitly configured value or one embedded as a query parameter
|
||||
* in the authorization URL. Returns `undefined` when no client_id is known
|
||||
* statically, which is the trigger for dynamic client registration in
|
||||
* {@link MCPOAuthFlow.#tryRegisterClient}.
|
||||
*/
|
||||
function staticClientIdFromConfig(config: MCPOAuthConfig): string | undefined {
|
||||
const fromConfig = config.clientId?.trim();
|
||||
if (fromConfig) return fromConfig;
|
||||
try {
|
||||
return new URL(config.authorizationUrl).searchParams.get("client_id") ?? undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
function resolveCallbackOptions(config: MCPOAuthConfig): OAuthCallbackFlowOptions {
|
||||
const redirectUri = resolveRedirectUri(config.redirectUri);
|
||||
validateRedirectConfig(config, redirectUri);
|
||||
// When a client_id is already pinned (config-supplied or embedded in the
|
||||
// authorization URL), it was registered against a specific redirect URI.
|
||||
// Silently advertising a different port at the authorize endpoint would
|
||||
// be rejected by providers like Atlassian (HTTP 500 in the browser, local
|
||||
// flow hangs until the 5-minute timeout), so fail fast instead.
|
||||
//
|
||||
// When no client_id is pinned, MCPOAuthFlow will attempt dynamic client
|
||||
// registration on demand with whichever loopback URI we actually bound —
|
||||
// the provider issues a client_id tied to *that* URI, so the random-port
|
||||
// fallback remains safe for first-install DCR flows whose preferred port
|
||||
// happens to be occupied.
|
||||
const allowPortFallback = staticClientIdFromConfig(config) === undefined;
|
||||
return {
|
||||
preferredPort: resolveCallbackPort(config.callbackPort, redirectUri),
|
||||
callbackPath: resolveCallbackPath(config.callbackPath, redirectUri),
|
||||
callbackHostname: resolveCallbackHostname(redirectUri),
|
||||
redirectUri,
|
||||
allowPortFallback,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -400,6 +430,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
|
||||
"Content-Type": "application/x-www-form-urlencoded",
|
||||
},
|
||||
body: params.toString(),
|
||||
signal: this.ctrl.signal,
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
@@ -453,14 +484,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
|
||||
}
|
||||
|
||||
#resolveClientId(config: MCPOAuthConfig): string | undefined {
|
||||
const fromConfig = config.clientId?.trim();
|
||||
if (fromConfig) return fromConfig;
|
||||
|
||||
try {
|
||||
return new URL(config.authorizationUrl).searchParams.get("client_id") ?? undefined;
|
||||
} catch {
|
||||
return undefined;
|
||||
}
|
||||
return staticClientIdFromConfig(config);
|
||||
}
|
||||
#resourceFromAuthorizationUrl(authorizationUrl: string): string | undefined {
|
||||
try {
|
||||
@@ -495,6 +519,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
|
||||
"Content-Type": "application/json",
|
||||
Accept: "application/json",
|
||||
},
|
||||
signal: this.ctrl.signal,
|
||||
body: JSON.stringify({
|
||||
client_name: "oh-my-pi",
|
||||
redirect_uris: [redirectUri],
|
||||
@@ -560,6 +585,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
|
||||
const response = await this.#fetch(wellKnownUrl, {
|
||||
method: "GET",
|
||||
headers: { Accept: "application/json" },
|
||||
signal: this.ctrl.signal,
|
||||
});
|
||||
if (!response.ok) return null;
|
||||
const metadata = (await response.json()) as { registration_endpoint?: string };
|
||||
@@ -578,6 +604,7 @@ export class MCPOAuthFlow extends OAuthCallbackFlow {
|
||||
method: "GET",
|
||||
redirect: "manual",
|
||||
headers: { Accept: "text/plain,text/html,application/json" },
|
||||
signal: this.ctrl.signal,
|
||||
});
|
||||
if (response.status < 400) return;
|
||||
const body = await response.text();
|
||||
|
||||
@@ -111,7 +111,10 @@ export const MCP_CONFIG_SCHEMA_URL =
|
||||
export interface MCPConfigFile {
|
||||
$schema?: string;
|
||||
mcpServers?: Record<string, MCPServerConfig>;
|
||||
/** Names to hide regardless of any source `enabled` flag. Highest precedence. */
|
||||
disabledServers?: string[];
|
||||
/** Names to force-enable when a non-writable source reports `enabled: false`. */
|
||||
enabledServers?: string[];
|
||||
}
|
||||
|
||||
// =============================================================================
|
||||
|
||||
@@ -24,7 +24,9 @@ import {
|
||||
truncateToWidth,
|
||||
visibleWidth,
|
||||
} from "@oh-my-pi/pi-tui";
|
||||
import { getMCPConfigPath, logger } from "@oh-my-pi/pi-utils";
|
||||
import { Settings } from "../../../config/settings";
|
||||
import { setMcpServerEnabled } from "../../../mcp/config-writer";
|
||||
import { getTabBarTheme } from "../../../modes/shared";
|
||||
import { theme } from "../../../modes/theme/theme";
|
||||
import { matchesAppInterrupt } from "../../../modes/utils/keybinding-matchers";
|
||||
@@ -263,6 +265,14 @@ export class ExtensionDashboard implements Component {
|
||||
const sm = this.settings ?? Settings.instance;
|
||||
if (!sm) return;
|
||||
|
||||
// MCP toggles route through the canonical denylist in
|
||||
// `~/.omp/agent/mcp.json` so `/mcp list`, the MCP runtime, and this
|
||||
// dashboard agree on every server's enabled state (issue #3827).
|
||||
if (extensionId.startsWith("mcp:")) {
|
||||
void this.#toggleMcpExtension(extensionId, enabled, sm);
|
||||
return;
|
||||
}
|
||||
|
||||
const disabled = ((sm.get("disabledExtensions") as string[]) ?? []).slice();
|
||||
if (enabled) {
|
||||
const index = disabled.indexOf(extensionId);
|
||||
@@ -281,6 +291,42 @@ export class ExtensionDashboard implements Component {
|
||||
void this.#refreshFromState();
|
||||
}
|
||||
|
||||
async #toggleMcpExtension(extensionId: string, enabled: boolean, sm: Settings): Promise<void> {
|
||||
const name = extensionId.slice("mcp:".length);
|
||||
try {
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", this.cwd),
|
||||
projectPath: getMCPConfigPath("project", this.cwd),
|
||||
sourcePath: this.#writableMcpSourcePath(extensionId),
|
||||
name,
|
||||
enabled,
|
||||
});
|
||||
} catch (error) {
|
||||
logger.warn("Failed to persist MCP toggle", { name, enabled, error: String(error) });
|
||||
}
|
||||
|
||||
// Reconcile `settings.disabledExtensions` with the canonical mcp.json
|
||||
// state so a legacy `mcp:<name>` flag from before this routing change
|
||||
// doesn't keep the server marked disabled after the user re-enables it
|
||||
// via the UI.
|
||||
const stored = ((sm.get("disabledExtensions") as string[]) ?? []).slice();
|
||||
const had = stored.indexOf(extensionId);
|
||||
if (enabled && had !== -1) {
|
||||
stored.splice(had, 1);
|
||||
sm.set("disabledExtensions", stored);
|
||||
this.#applyDisabledExtensions(stored);
|
||||
}
|
||||
|
||||
await this.#refreshFromState();
|
||||
}
|
||||
|
||||
#writableMcpSourcePath(extensionId: string): string | undefined {
|
||||
const extension = this.#state.extensions.find(ext => ext.id === extensionId);
|
||||
if (!extension) return undefined;
|
||||
if (extension.source.provider !== "native" && extension.source.provider !== "mcp-json") return undefined;
|
||||
return extension.path;
|
||||
}
|
||||
|
||||
async #refreshFromState(): Promise<void> {
|
||||
const refreshToken = ++this.#refreshToken;
|
||||
// Remember the current tab so it survives the re-sort.
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
*/
|
||||
import * as path from "node:path";
|
||||
import { fuzzyMatch } from "@oh-my-pi/pi-tui";
|
||||
import { logger } from "@oh-my-pi/pi-utils";
|
||||
import { getMCPConfigPath, logger } from "@oh-my-pi/pi-utils";
|
||||
import type { ContextFile } from "../../../capability/context-file";
|
||||
import type { ExtensionModule } from "../../../capability/extension-module";
|
||||
import type { Hook } from "../../../capability/hook";
|
||||
@@ -22,6 +22,7 @@ import {
|
||||
isProviderEnabled,
|
||||
loadCapability,
|
||||
} from "../../../discovery";
|
||||
import { readDisabledServers, readEnabledServers } from "../../../mcp/config-writer";
|
||||
import type {
|
||||
DashboardState,
|
||||
Extension,
|
||||
@@ -141,12 +142,32 @@ export async function loadAllExtensions(cwd?: string, disabledIds?: string[]): P
|
||||
logger.warn("Failed to load extension-modules capability", { error: String(error) });
|
||||
}
|
||||
|
||||
// Load MCP servers
|
||||
// Load MCP servers. The dashboard mirrors `/mcp list` (issue #3827) by
|
||||
// honoring the same disable signals: the dashboard-private settings list,
|
||||
// the per-server `enabled: false` flag, and the user-level `disabledServers`
|
||||
// denylist that `/mcp disable` writes through `setServerDisabled`. The
|
||||
// user-level `enabledServers` allowlist overrides a non-writable source's
|
||||
// `enabled: false` (e.g. opencode.json) but never the denylist.
|
||||
try {
|
||||
const userMcpPath = cwd ? getMCPConfigPath("user", cwd) : undefined;
|
||||
const [mcpDisabledNames, mcpForcedEnabled] = await Promise.all([
|
||||
userMcpPath
|
||||
? readDisabledServers(userMcpPath)
|
||||
.then(list => new Set(list))
|
||||
.catch(() => new Set<string>())
|
||||
: Promise.resolve(new Set<string>()),
|
||||
userMcpPath
|
||||
? readEnabledServers(userMcpPath)
|
||||
.then(list => new Set(list))
|
||||
.catch(() => new Set<string>())
|
||||
: Promise.resolve(new Set<string>()),
|
||||
]);
|
||||
const mcps = await loadCapability<MCPServer>("mcps", loadOpts);
|
||||
for (const server of mcps.all) {
|
||||
const id = makeExtensionId("mcp", server.name);
|
||||
const isDisabled = disabledExtensions.has(id);
|
||||
const forced = mcpForcedEnabled.has(server.name);
|
||||
const sourceSaysDisabled = server.enabled === false && !forced;
|
||||
const isDisabled = mcpDisabledNames.has(server.name) || disabledExtensions.has(id) || sourceSaysDisabled;
|
||||
const isShadowed = (server as { _shadowed?: boolean })._shadowed;
|
||||
const providerEnabled = isProviderEnabled(server._source.provider);
|
||||
|
||||
|
||||
@@ -63,6 +63,14 @@ export interface MCPAddWizardOAuthResult {
|
||||
interface MCPAddWizardOAuthOptions {
|
||||
serverUrl?: string;
|
||||
resource?: string;
|
||||
/**
|
||||
* External cancellation source. Aborting it tears down the in-flight OAuth
|
||||
* flow and surfaces a neutral cancellation error. The wizard wires its own
|
||||
* controller here so Esc cancels the OAuth wait instead of stepping back
|
||||
* through the form (the wizard is focused, so the editor's Esc hook does
|
||||
* not fire).
|
||||
*/
|
||||
abortSignal?: AbortSignal;
|
||||
}
|
||||
|
||||
interface WizardState {
|
||||
@@ -135,6 +143,12 @@ export class MCPAddWizard extends Container {
|
||||
| null = null;
|
||||
#onTestConnectionCallback: ((config: MCPServerConfig) => Promise<void>) | null = null;
|
||||
#onRenderCallback: (() => void) | null = null;
|
||||
/**
|
||||
* Set while the OAuth callback is in flight; populated by
|
||||
* {@link #launchOAuthFlow} and consumed by {@link handleInput} so Esc
|
||||
* cancels the OAuth wait instead of stepping back through the form.
|
||||
*/
|
||||
#oauthAbort: AbortController | null = null;
|
||||
|
||||
constructor(
|
||||
onComplete: (name: string, config: MCPServerConfig, scope: Scope) => void,
|
||||
@@ -473,6 +487,15 @@ export class MCPAddWizard extends Container {
|
||||
}
|
||||
|
||||
handleInput(keyData: string): void {
|
||||
// While an OAuth callback is being awaited, Esc/Ctrl+C aborts the flow
|
||||
// rather than stepping back through the form: the wizard advertises
|
||||
// "(Press Esc to cancel)" during the wait, and stepping back would
|
||||
// leave the OAuth login orphaned.
|
||||
if (this.#oauthAbort && (keyData === "\x03" || matchesAppInterrupt(keyData))) {
|
||||
this.#oauthAbort.abort("MCP OAuth flow cancelled by user");
|
||||
return;
|
||||
}
|
||||
|
||||
// Handle Ctrl+C to cancel wizard immediately
|
||||
if (keyData === "\x03") {
|
||||
// Ctrl+C pressed - cancel wizard
|
||||
@@ -1153,6 +1176,7 @@ export class MCPAddWizard extends Container {
|
||||
this.#contentContainer.addChild(new Text(theme.fg("muted", "(Press Esc to cancel)"), 0, 0));
|
||||
this.#requestRender();
|
||||
|
||||
this.#oauthAbort = new AbortController();
|
||||
try {
|
||||
// Call OAuth handler
|
||||
const oauthResource = this.#state.oauthResource || (this.#state.transport === "stdio" ? "" : this.#state.url);
|
||||
@@ -1165,6 +1189,7 @@ export class MCPAddWizard extends Container {
|
||||
{
|
||||
serverUrl: this.#state.url || undefined,
|
||||
resource: oauthResource || undefined,
|
||||
abortSignal: this.#oauthAbort.signal,
|
||||
},
|
||||
);
|
||||
|
||||
@@ -1237,16 +1262,29 @@ export class MCPAddWizard extends Container {
|
||||
healthPassed ? 1000 : 2000,
|
||||
);
|
||||
} catch (error) {
|
||||
// Show error with options to retry or go back
|
||||
// User cancellation has its own neutral heading + tip; everything else
|
||||
// keeps the "OAuth authentication failed" framing so the existing tips
|
||||
// stay meaningful. Name-matching avoids importing controller types.
|
||||
const cancelled = error instanceof Error && error.name === "MCPOAuthCancelledError";
|
||||
const errorMsg = sanitize(error instanceof Error ? error.message : String(error));
|
||||
this.#contentContainer.clear();
|
||||
this.#contentContainer.addChild(new Text(theme.fg("error", "✗ OAuth authentication failed"), 0, 0));
|
||||
this.#contentContainer.addChild(
|
||||
new Text(
|
||||
cancelled ? theme.fg("muted", "○ OAuth cancelled") : theme.fg("error", "✗ OAuth authentication failed"),
|
||||
0,
|
||||
0,
|
||||
),
|
||||
);
|
||||
this.#contentContainer.addChild(new Spacer(1));
|
||||
this.#contentContainer.addChild(new Text(errorMsg, 0, 0));
|
||||
this.#contentContainer.addChild(new Spacer(1));
|
||||
|
||||
// Provide helpful tips based on error type
|
||||
if (errorMsg.includes("timeout") || errorMsg.includes("timed out")) {
|
||||
if (cancelled) {
|
||||
this.#contentContainer.addChild(
|
||||
new Text(theme.fg("muted", "Tip: Choose Retry to launch the browser again."), 0, 0),
|
||||
);
|
||||
} else if (errorMsg.includes("timeout") || errorMsg.includes("timed out")) {
|
||||
this.#contentContainer.addChild(
|
||||
new Text(theme.fg("muted", "Tip: Complete authorization faster next time"), 0, 0),
|
||||
);
|
||||
@@ -1272,6 +1310,8 @@ export class MCPAddWizard extends Container {
|
||||
// Set up as a selector step
|
||||
this.#selectedIndex = 0;
|
||||
this.#currentStep = "oauth-error";
|
||||
} finally {
|
||||
this.#oauthAbort = null;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -24,7 +24,12 @@ import { getKnownRoleIds, getRoleInfo, MODEL_ROLE_IDS, MODEL_ROLES } from "../..
|
||||
import type { Settings } from "../../config/settings";
|
||||
import { type ThemeColor, theme } from "../../modes/theme/theme";
|
||||
import { matchesSelectDown, matchesSelectUp } from "../../modes/utils/keybinding-matchers";
|
||||
import { AUTO_THINKING, type ConfiguredThinkingLevel, getConfiguredThinkingLevelMetadata } from "../../thinking";
|
||||
import {
|
||||
AUTO_THINKING,
|
||||
type ConfiguredThinkingLevel,
|
||||
getConfiguredThinkingLevelMetadata,
|
||||
parseConfiguredThinkingLevel,
|
||||
} from "../../thinking";
|
||||
import { getTabBarTheme } from "../shared";
|
||||
import { DynamicBorder } from "./dynamic-border";
|
||||
|
||||
@@ -342,10 +347,7 @@ export class ModelSelectorComponent extends Container {
|
||||
if (resolved.model) {
|
||||
nextRoles[role] = {
|
||||
model: resolved.model,
|
||||
thinkingLevel:
|
||||
resolved.explicitThinkingLevel && resolved.thinkingLevel !== undefined
|
||||
? resolved.thinkingLevel
|
||||
: ThinkingLevel.Inherit,
|
||||
thinkingLevel: this.#getResolvedRoleThinkingLevel(role, resolved),
|
||||
autoSelected: false,
|
||||
};
|
||||
}
|
||||
@@ -363,10 +365,7 @@ export class ModelSelectorComponent extends Container {
|
||||
if (!resolved.model) continue;
|
||||
nextRoles[role] = {
|
||||
model: resolved.model,
|
||||
thinkingLevel:
|
||||
resolved.explicitThinkingLevel && resolved.thinkingLevel !== undefined
|
||||
? resolved.thinkingLevel
|
||||
: ThinkingLevel.Inherit,
|
||||
thinkingLevel: this.#getResolvedRoleThinkingLevel(role, resolved),
|
||||
autoSelected: true,
|
||||
};
|
||||
}
|
||||
@@ -1059,6 +1058,19 @@ export class ModelSelectorComponent extends Container {
|
||||
);
|
||||
}
|
||||
}
|
||||
#getResolvedRoleThinkingLevel(
|
||||
role: string,
|
||||
resolved: { explicitThinkingLevel: boolean; thinkingLevel?: ThinkingLevel },
|
||||
): ConfiguredThinkingLevel {
|
||||
if (resolved.explicitThinkingLevel && resolved.thinkingLevel !== undefined) {
|
||||
return resolved.thinkingLevel;
|
||||
}
|
||||
if (role === "default") {
|
||||
return parseConfiguredThinkingLevel(this.#settings.get("defaultThinkingLevel")) ?? ThinkingLevel.Inherit;
|
||||
}
|
||||
return ThinkingLevel.Inherit;
|
||||
}
|
||||
|
||||
#getThinkingLevelsForModel(model: Model): ReadonlyArray<ConfiguredThinkingLevel> {
|
||||
return [ThinkingLevel.Inherit, ThinkingLevel.Off, AUTO_THINKING, ...getSupportedEfforts(model)];
|
||||
}
|
||||
|
||||
@@ -609,6 +609,15 @@ export class EventController {
|
||||
}
|
||||
for (const content of this.ctx.streamingMessage.content) {
|
||||
if (content.type !== "toolCall") continue;
|
||||
// Anthropic/OpenAI open a streamed tool block with an empty id (and
|
||||
// `{}` args) before the id/arguments arrive; Gemini assembles the
|
||||
// whole call first, so it never hits this. Keying `pendingTools` by
|
||||
// "" would create a placeholder card, and the later real-id frame —
|
||||
// `pendingTools.has(realId)` false — would create a SECOND card,
|
||||
// orphaning the blank one (no `tool_execution_*` event ever carries
|
||||
// "", so it is never matched, updated, or removed). Defer until the
|
||||
// provider assigns the real id.
|
||||
if (!content.id) continue;
|
||||
if (content.name === "read") {
|
||||
if (!readArgsHaveTarget(content.arguments)) {
|
||||
// Args still streaming — defer until path is parseable so we can route to the
|
||||
@@ -890,6 +899,13 @@ export class EventController {
|
||||
}
|
||||
|
||||
async #handleToolExecutionEnd(event: Extract<AgentSessionEvent, { type: "tool_execution_end" }>): Promise<void> {
|
||||
// A transient overlay (auto-compaction / auto-retry / handoff) that ran
|
||||
// between this tool's start and end could have detached the working
|
||||
// loader. `tool_execution_update` already reconciles this so the spinner
|
||||
// reappears mid-tool; mirror it here so subagent (`task`) completions —
|
||||
// which only fire `tool_execution_end`, never `_update` — do not leave
|
||||
// the UI looking idle while the session keeps streaming (#3857).
|
||||
this.#ensureWorkingLoaderWhileStreaming();
|
||||
if (event.toolName === "read") {
|
||||
if (this.#inlineReadToolImages(event.toolCallId, event.result)) {
|
||||
const component = this.ctx.pendingTools.get(event.toolCallId);
|
||||
|
||||
@@ -62,6 +62,16 @@ function withTimeout<T>(promise: Promise<T>, timeoutMs: number, message: string,
|
||||
}, timeoutMs);
|
||||
return Promise.race([promise, timeoutPromise]).finally(() => clearTimeout(timer));
|
||||
}
|
||||
function raceAbortSignal<T>(promise: Promise<T>, signal: AbortSignal, createError: () => Error): Promise<T> {
|
||||
if (signal.aborted) return Promise.reject(createError());
|
||||
|
||||
const aborted = Promise.withResolvers<never>();
|
||||
const onAbort = (): void => aborted.reject(createError());
|
||||
signal.addEventListener("abort", onAbort, { once: true });
|
||||
return Promise.race([promise, aborted.promise]).finally(() => {
|
||||
signal.removeEventListener("abort", onAbort);
|
||||
});
|
||||
}
|
||||
|
||||
/** Renders the MCP OAuth fallback URL without hard-wrapping the copy target. */
|
||||
export class MCPAuthorizationLinkPrompt implements Component {
|
||||
@@ -135,6 +145,22 @@ interface OAuthFlowResult {
|
||||
resource?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Thrown by {@link MCPCommandController}'s OAuth handler when the user (or a
|
||||
* caller-supplied {@link AbortSignal}) cancels the in-flight flow. Distinct
|
||||
* from network/timeout failures so callers can surface a neutral
|
||||
* "cancelled" status instead of an error banner.
|
||||
*/
|
||||
export class MCPOAuthCancelledError extends Error {
|
||||
constructor(message = "OAuth flow cancelled") {
|
||||
super(message);
|
||||
this.name = "MCPOAuthCancelledError";
|
||||
}
|
||||
}
|
||||
|
||||
/** Reason recorded on the OAuth flow's AbortController when the user hits Esc. */
|
||||
const MCP_OAUTH_USER_CANCEL_REASON = "MCP OAuth flow cancelled by user";
|
||||
|
||||
type MCPAddScope = "user" | "project";
|
||||
type MCPAddTransport = "http" | "sse";
|
||||
|
||||
@@ -521,6 +547,10 @@ export class MCPCommandController {
|
||||
userClientSecret: finalConfig.oauth?.clientSecret,
|
||||
});
|
||||
} catch (oauthError) {
|
||||
if (oauthError instanceof MCPOAuthCancelledError) {
|
||||
this.ctx.showStatus(`Add cancelled for "${parsed.initialName}"`);
|
||||
return;
|
||||
}
|
||||
this.ctx.showError(
|
||||
`OAuth flow failed for "${parsed.initialName}": ${oauthError instanceof Error ? oauthError.message : String(oauthError)}`,
|
||||
);
|
||||
@@ -587,6 +617,14 @@ export class MCPCommandController {
|
||||
serverUrl?: string;
|
||||
resource?: string;
|
||||
stripSameOriginResource?: boolean;
|
||||
/**
|
||||
* External cancellation source: when this signal aborts, the in-flight
|
||||
* OAuth flow is torn down and {@link MCPOAuthCancelledError} is thrown.
|
||||
* Wizards (which own focus and absorb Esc themselves) pass their own
|
||||
* controller here; editor-focused callers rely on the Esc hook
|
||||
* installed below instead.
|
||||
*/
|
||||
abortSignal?: AbortSignal;
|
||||
},
|
||||
): Promise<OAuthFlowResult> {
|
||||
const authStorage = this.ctx.session.modelRegistry.authStorage;
|
||||
@@ -614,6 +652,26 @@ export class MCPCommandController {
|
||||
}
|
||||
let manualInputClaim: { promise: Promise<string>; clear: (reason?: string) => void } | undefined;
|
||||
const oauthTimeout = new AbortController();
|
||||
// User Esc and external aborts route through here; the timeout path sets
|
||||
// its own reason and leaves this flag false so the catch can distinguish
|
||||
// "user cancelled" (status) from "deadline elapsed" (error).
|
||||
let userCancelled = false;
|
||||
const requestUserCancel = (reason: string): void => {
|
||||
userCancelled = true;
|
||||
if (!oauthTimeout.signal.aborted) oauthTimeout.abort(reason);
|
||||
};
|
||||
const originalOnEscape = this.ctx.editor.onEscape;
|
||||
this.ctx.editor.onEscape = () => requestUserCancel(MCP_OAUTH_USER_CANCEL_REASON);
|
||||
const externalSignal = opts?.abortSignal;
|
||||
const onExternalAbort = (): void => {
|
||||
const reason = externalSignal?.reason;
|
||||
requestUserCancel(typeof reason === "string" ? reason : MCP_OAUTH_USER_CANCEL_REASON);
|
||||
};
|
||||
if (externalSignal?.aborted) {
|
||||
onExternalAbort();
|
||||
} else {
|
||||
externalSignal?.addEventListener("abort", onExternalAbort, { once: true });
|
||||
}
|
||||
try {
|
||||
// Create OAuth flow
|
||||
const flow = new MCPOAuthFlow(
|
||||
@@ -641,7 +699,7 @@ export class MCPCommandController {
|
||||
block.addChild(new Spacer(1));
|
||||
block.addChild(
|
||||
new Text(
|
||||
theme.fg("muted", "Waiting for authorization... (Press Ctrl+C to cancel, 5 minute timeout)"),
|
||||
theme.fg("muted", "Waiting for authorization... (Press Esc to cancel, 5 minute timeout)"),
|
||||
1,
|
||||
0,
|
||||
),
|
||||
@@ -687,9 +745,18 @@ export class MCPCommandController {
|
||||
},
|
||||
);
|
||||
|
||||
// Execute OAuth flow with 5 minute timeout
|
||||
const createAbortError = (): Error => {
|
||||
const reason = String(oauthTimeout.signal.reason ?? "MCP OAuth flow aborted");
|
||||
return userCancelled ? new MCPOAuthCancelledError() : new Error(reason);
|
||||
};
|
||||
if (oauthTimeout.signal.aborted) throw createAbortError();
|
||||
|
||||
// Execute OAuth flow with 5 minute timeout. Race the login itself
|
||||
// against the abort signal because Esc/external abort may fire before
|
||||
// MCPOAuthFlow reaches OAuthCallbackFlow.#waitForCallback, where the
|
||||
// underlying callback server normally observes the signal.
|
||||
const credentials = await withTimeout(
|
||||
flow.login(),
|
||||
raceAbortSignal(flow.login(), oauthTimeout.signal, createAbortError),
|
||||
5 * 60 * 1000,
|
||||
"OAuth flow timed out after 5 minutes",
|
||||
() => oauthTimeout.abort("MCP OAuth flow timed out"),
|
||||
@@ -727,6 +794,14 @@ export class MCPCommandController {
|
||||
resource: flow.resource,
|
||||
};
|
||||
} catch (error) {
|
||||
// User-initiated cancel (Esc or external signal) → neutral status, not
|
||||
// a failure. Check the flag we set in `requestUserCancel`, not the
|
||||
// abort reason: the timeout path also aborts but with a different
|
||||
// reason, and we want it to surface as a timeout error below.
|
||||
if (userCancelled) {
|
||||
throw new MCPOAuthCancelledError();
|
||||
}
|
||||
|
||||
const errorMsg = error instanceof Error ? error.message : String(error);
|
||||
|
||||
// Provide helpful error messages based on failure type
|
||||
@@ -742,6 +817,8 @@ export class MCPCommandController {
|
||||
throw new Error(`OAuth authentication failed: ${errorMsg}`);
|
||||
}
|
||||
} finally {
|
||||
this.ctx.editor.onEscape = originalOnEscape;
|
||||
externalSignal?.removeEventListener("abort", onExternalAbort);
|
||||
manualInputClaim?.clear("Manual MCP OAuth input cleared");
|
||||
}
|
||||
}
|
||||
@@ -1629,6 +1706,10 @@ export class MCPCommandController {
|
||||
];
|
||||
this.#showMessage(lines.join("\n"));
|
||||
} catch (error) {
|
||||
if (error instanceof MCPOAuthCancelledError) {
|
||||
this.ctx.showStatus(`Reauthorization cancelled for "${name}"`);
|
||||
return;
|
||||
}
|
||||
this.ctx.showError(`Failed to reauthorize server: ${error instanceof Error ? error.message : String(error)}`);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -144,7 +144,7 @@ export class ToolArgsRevealController {
|
||||
entry = {
|
||||
component: undefined,
|
||||
target: partialJson,
|
||||
revealed: 0,
|
||||
revealed: clampSliceEnd(partialJson, partialJson.length),
|
||||
rawInput,
|
||||
exposeRawPartialJson,
|
||||
parsedArgs: {},
|
||||
|
||||
@@ -100,6 +100,7 @@ import { formatDuration } from "../slash-commands/helpers/format";
|
||||
import { STTController, type SttState } from "../stt";
|
||||
import { discoverTitleSystemPromptFile, resolvePromptInput } from "../system-prompt";
|
||||
import { formatTaskId } from "../task/render";
|
||||
import type { ConfiguredThinkingLevel } from "../thinking";
|
||||
import type { LspStartupServerInfo } from "../tools";
|
||||
import { normalizeLocalScheme } from "../tools/path-utils";
|
||||
import { replaceTabs, TRUNCATE_LENGTHS, truncateToWidth } from "../tools/render-utils";
|
||||
@@ -493,8 +494,8 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
#goalTurnHadToolCalls = false;
|
||||
#goalContinuationTurnInFlight = false;
|
||||
#goalSuppressNextContinuation = false;
|
||||
#planModePreviousModelState: { model: Model; thinkingLevel?: ThinkingLevel } | undefined;
|
||||
#pendingModelSwitch: { model: Model; thinkingLevel?: ThinkingLevel } | undefined;
|
||||
#planModePreviousModelState: { model: Model; thinkingLevel?: ConfiguredThinkingLevel } | undefined;
|
||||
#pendingModelSwitch: { model: Model; thinkingLevel?: ConfiguredThinkingLevel } | undefined;
|
||||
#planModeHasEntered = false;
|
||||
#planReviewOverlay: PlanReviewOverlay | undefined;
|
||||
#planReviewOverlayHandle: OverlayHandle | undefined;
|
||||
@@ -1917,7 +1918,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
const planThinkingLevel = resolved.explicitThinkingLevel ? resolved.thinkingLevel : undefined;
|
||||
|
||||
this.#planModePreviousModelState = currentModel
|
||||
? { model: currentModel, thinkingLevel: this.session.thinkingLevel }
|
||||
? { model: currentModel, thinkingLevel: this.session.configuredThinkingLevel() }
|
||||
: undefined;
|
||||
|
||||
if (!sameModel) {
|
||||
@@ -2125,7 +2126,7 @@ export class InteractiveMode implements InteractiveModeContext {
|
||||
});
|
||||
}
|
||||
|
||||
async #restorePlanPreviousModel(prev: { model: Model; thinkingLevel?: ThinkingLevel }): Promise<void> {
|
||||
async #restorePlanPreviousModel(prev: { model: Model; thinkingLevel?: ConfiguredThinkingLevel }): Promise<void> {
|
||||
if (modelsAreEqual(this.session.model, prev.model)) {
|
||||
// Same model — only thinking level may differ. Avoid setModelTemporary()
|
||||
// which would reset provider-side sessions and break continuity.
|
||||
|
||||
@@ -99,9 +99,9 @@ export function buildFileMentionBlock(files: FileMentionMessage["files"], indent
|
||||
const block = new TranscriptBlock();
|
||||
for (const file of files) {
|
||||
let suffix: string;
|
||||
if (file.skippedReason === "tooLarge") {
|
||||
if (file.skippedReason === "tooLarge" || file.skippedReason === "binary") {
|
||||
const size = typeof file.byteSize === "number" ? formatBytes(file.byteSize) : "unknown size";
|
||||
suffix = `(skipped: ${size})`;
|
||||
suffix = file.skippedReason === "binary" ? `(skipped: binary, ${size})` : `(skipped: ${size})`;
|
||||
} else {
|
||||
suffix = file.image
|
||||
? "(image)"
|
||||
|
||||
@@ -1,54 +0,0 @@
|
||||
---
|
||||
name: oracle
|
||||
description: Wise senior engineer to consult or delegate work to — debugging, architecture, second opinions, and hands-on implementation when asked.
|
||||
spawns: explore
|
||||
model: pi/slow
|
||||
thinking-level: xhigh
|
||||
---
|
||||
|
||||
You are the wise guy on the team — a senior engineer with deep judgment that other agents consult when they are stuck, uncertain, or need a second opinion. You also take direct delegation: if the caller hands you work, you do it, including reads, writes, edits, and running commands.
|
||||
|
||||
You diagnose, decide, and execute. You match the mode to the ask:
|
||||
- **Consult**: explain the root cause, lay out tradeoffs, recommend a path.
|
||||
- **Delegate**: carry the work to completion — modify files, run verification, deliver a finished change.
|
||||
|
||||
<directives>
|
||||
- You MUST reason from first principles. The caller already tried the obvious.
|
||||
- You MUST use tools to verify claims. You NEVER speculate about code behavior — read it.
|
||||
- You MUST identify root causes, not symptoms. If the caller says "X is broken", determine *why* X is broken.
|
||||
- You MUST surface hidden assumptions — in the code, in the caller's framing, in the environment.
|
||||
- You SHOULD consider at least two hypotheses before converging on one.
|
||||
- You SHOULD invoke tools in parallel when investigating multiple hypotheses.
|
||||
- When the problem is architectural, you MUST weigh tradeoffs explicitly: what does each option cost, what does it buy, what does it foreclose.
|
||||
- When delegated implementation work, you MUST finish it: edit the files, run the relevant tests/checks, and report exactly what changed.
|
||||
</directives>
|
||||
|
||||
<decision-framework>
|
||||
Apply pragmatic minimalism:
|
||||
- **Bias toward simplicity**: The right solution is the least complex one that fulfills actual requirements. Resist hypothetical future needs.
|
||||
- **Leverage what exists**: Favor modifications to current code and established patterns over introducing new components. New dependencies or infrastructure require explicit justification.
|
||||
- **One clear path**: Present a single primary recommendation. Mention alternatives only when they offer substantially different tradeoffs worth considering.
|
||||
- **Match depth to complexity**: Quick questions get quick answers. Reserve thorough analysis for genuinely complex problems.
|
||||
- **Signal the investment**: Tag recommendations with estimated effort — Quick (<1h), Short (1-4h), Medium (1-2d), Large (3d+).
|
||||
</decision-framework>
|
||||
|
||||
<procedure>
|
||||
1. Read the problem statement carefully. Identify what was already tried, what failed, and whether the caller wants advice or execution.
|
||||
2. Form 2-3 hypotheses for the root cause (for diagnosis) or 2-3 viable approaches (for design).
|
||||
3. Use tools to gather evidence — read relevant code, trace data flow, check types, search for related patterns. Parallelize independent reads.
|
||||
4. Eliminate hypotheses based on evidence. Narrow to the most likely cause or best approach.
|
||||
5. If consulting: deliver verdict with supporting evidence and a concrete recommendation.
|
||||
6. If implementing: make the changes, verify them, and report the diff and verification result.
|
||||
</procedure>
|
||||
|
||||
<scope-discipline>
|
||||
- Do ONLY what was asked. No unsolicited refactors or improvements.
|
||||
- If you notice other issues, list at most 2 as "Optional future considerations" at the end.
|
||||
- You NEVER expand the problem surface beyond the original request.
|
||||
- Exhaust provided context before reaching for tools. External lookups fill genuine gaps, not curiosity.
|
||||
</scope-discipline>
|
||||
|
||||
<critical>
|
||||
You MUST keep going until the problem is solved or the work is finished. Before finalizing: re-scan for unstated assumptions, verify claims are grounded in code not invented, check for overly strong language not justified by evidence.
|
||||
The caller came to you because they trust your judgment. Get it right.
|
||||
</critical>
|
||||
@@ -0,0 +1,107 @@
|
||||
---
|
||||
name: Tester
|
||||
description: Authoritative test writer. ALWAYS delegate test authoring to this agent — NEVER write tests yourself. Writes high-signal tests defending real contracts (behavior, invariants, edge cases) and refuses worthless tests that assert plumbing or restate the code.
|
||||
tools: read, grep, glob, bash, edit, write, lsp, ast_grep, ast_edit
|
||||
spawns: explore
|
||||
model: pi/task
|
||||
thinking-level: high
|
||||
---
|
||||
|
||||
<system-conventions>
|
||||
RFC 2119 applies to MUST, REQUIRED, SHOULD, RECOMMENDED, MAY, OPTIONAL. `NEVER` and `AVOID` MUST be interpreted as aliases for `MUST NOT` and `SHOULD NOT` respectively.
|
||||
</system-conventions>
|
||||
|
||||
You are a staff test engineer with taste. You write tests that earn their place in the suite and you delete — or refuse to write — tests that don't. You have agency: when asked for coverage that proves nothing, you write the test that would actually catch the bug instead.
|
||||
|
||||
<stakes>
|
||||
A test suite is a liability until it pays for itself. Every worthless test is negative value: it costs CI time, blocks honest refactors, and lulls the team into false confidence while the real bug ships. A test's only job is to FAIL when behavior breaks and PASS otherwise. A test that cannot fail for any real defect is noise wearing a green check. You are here because models flood codebases with exactly that noise. You write the opposite.
|
||||
</stakes>
|
||||
|
||||
<critical>
|
||||
- The litmus for every test: **name the concrete, externally observable contract it defends** — a behavior, output shape, state transition, error mapping, invariant, or a regression-prone parsing boundary. Cannot name it in one sentence? NEVER write the test.
|
||||
- Mutation test in your head: if a plausible bug — a flipped condition, an off-by-one, a wrong return value, a dropped case — would still let the test PASS, the test is worthless. Discard it.
|
||||
- You NEVER write tests that assert plumbing or restate the implementation. The forbidden classes are enumerated in `<worthless-tests>` and are hard prohibitions.
|
||||
- You MUST match the repo's existing test conventions — framework, file layout, naming, assertion style. A second convention beside an existing one is PROHIBITED.
|
||||
</critical>
|
||||
|
||||
<anti-patterns name="worthless-tests">
|
||||
NEVER write any of these. Each is a green check that survives real bugs:
|
||||
- **Config/setter echo.** Setting a value then asserting it reads back (`set(x, 30); expect(get(x)).toBe(30)`) tests the language's assignment, not your code.
|
||||
- **Source-grep.** Reading an implementation/build file and asserting on its TEXT — `expect(src).toContain("newFn()")`, `.toMatch(/import …/)`, `.not.toContain("oldName")`, "comment says X". Tests how code LOOKS, breaks on rename/reflow, passes while behavior is broken. Enforce structural facts with a type test or lint rule; enforce behavior by running the code.
|
||||
- **Tautologies.** `expect(true).toBe(true)`, `expect(x).toBe(x)`, asserting a constant equals its literal.
|
||||
- **Bare no-throw.** `expect(() => f()).not.toThrow()` with no assertion on the result. "It ran" is not a contract.
|
||||
- **Construction smoke.** "Constructs without error", "package boots", "command starts" — unless that wiring genuinely can't be exercised in-process AND a real failure mode hides there.
|
||||
- **Mock round-trips.** Asserting a mock was called with the args you just passed it. You tested the mock, not the system.
|
||||
- **Existence/shape-only.** Non-empty string, length-grew, "field is defined", "returns an object with key Y" — without asserting the VALUE that matters.
|
||||
- **Default snapshots.** Asserting every field of a default config equals its current default. A harmless default change shouldn't redden a test. Assert logical behavior, not the current state.
|
||||
- **Field-wiring.** Asserting an option passed in lands on a property, or that a getter returns the value the constructor stored. Test the downstream BEHAVIOR that depends on it, not the assignment.
|
||||
- **Duplicate-layer coverage.** Re-proving through mocks what an integration test already proves. Drop the narrower restatement.
|
||||
|
||||
When asked for coverage that would only produce the above, you write the test that actually exercises the behavior, and you state in your result why the requested shape was worthless.
|
||||
</anti-patterns>
|
||||
|
||||
<what-to-test>
|
||||
Aim every test at something that can actually break:
|
||||
- **Behavior & outputs** — given input, the observable result (return value, emitted event, written file, error surfaced).
|
||||
- **State transitions** — the legal and illegal moves of a stateful component; one test per invariant or transition, not one per field touched.
|
||||
- **Invariants across fields** — relationships that MUST hold (sorted output stays sorted, sum of parts equals total, encode∘decode is identity).
|
||||
- **Edge & boundary values** — zero, empty, one, max, negative, off-by-one, overflow, unicode, the value just inside and just outside a limit.
|
||||
- **Precedence & resolution** — arg beats env beats default; later override wins; first-match-wins.
|
||||
- **Error paths** — trigger the REAL failure (bad input, missing dep, denied permission) and assert the surfaced contract (error type, message mapping, exit code). NEVER instantiate the error class directly or inspect internal metadata.
|
||||
- **Regression-prone parsing boundaries** — the exact bytes where a parser/serializer historically broke; pin past regressions with a named case.
|
||||
</what-to-test>
|
||||
|
||||
<techniques>
|
||||
Reach for the right shape; do not reinvent what the repo's framework already gives you.
|
||||
- **Table-driven tests.** One body, many `{ name, input, expected }` rows covering boundaries and equivalence classes plus error cases. Name every row so a failure points at the case. The default shape for any function with a clear input→output mapping.
|
||||
- **Subtests.** Group related cases under one parent with isolated setup and independent failure reporting. Prefer over many tiny near-duplicate test functions.
|
||||
- **Property-based tests.** Assert invariants over generated inputs — round-trip identity, idempotence (`f(f(x)) == f(x)`), commutativity, monotonicity, "never panics and output stays well-formed". Catches cases you wouldn't enumerate by hand.
|
||||
- **Deterministic randomness.** Seed every generator and PRINT the seed on failure so a red run reproduces exactly. NEVER use an unseeded clock-derived source — flaky tests are worse than no tests.
|
||||
- **Fuzz tests.** For parsers, decoders, deserializers, anything eating untrusted bytes: feed mutated/random input, assert no crash and that invariants hold. Seed the corpus from known-tricky inputs and every past regression.
|
||||
- **Benchmarks.** ONLY when performance is part of the contract. Measure the operation, not setup; consume the result so it isn't optimized away; compare against a baseline or threshold. A benchmark that asserts nothing is documentation, not a test.
|
||||
- **Golden/snapshot.** Only for genuinely stable, human-reviewed output where exact bytes are the contract (codegen, serialized formats). NEVER snapshot volatile or incidental output — it becomes a rubber stamp nobody reads.
|
||||
</techniques>
|
||||
|
||||
<black-box>
|
||||
- **Test through the public API**, the way a real consumer calls it. Place tests in an EXTERNAL test package/module (separate namespace, no access to internals) so the compiler forbids reaching past the contract. This is the default and it forces you to test what callers depend on.
|
||||
- **Internal (white-box) tests only for private invariants with no observable surface** — e.g. a balancing property of an internal tree, a cache eviction order. Justify each one; if the invariant has an observable effect, test that effect from outside instead.
|
||||
- NEVER reach into private state to assert what you could observe through the public surface. Coupling tests to internals is what makes refactors painful and tempts people to delete the suite.
|
||||
</black-box>
|
||||
|
||||
<fakes>
|
||||
- **Prefer real implementations.** If the dependency is cheap and deterministic, use the real thing.
|
||||
- **Prefer hand-written fakes over mocking frameworks.** A small in-memory implementation of an interface is type-checked, readable, survives refactors, and tests behavior. Mocking frameworks pull you toward asserting call counts and argument sequences — that is plumbing, and it breaks on every harmless internal change.
|
||||
- **Mock only true external boundaries** — network, wall clock, filesystem, system randomness, third-party services — and even there a fake beats a mock. Inject the boundary; never patch globals.
|
||||
- NEVER use module-registry mocking that leaks across test files. Spy on the imported object and restore in teardown.
|
||||
</fakes>
|
||||
|
||||
<isolation>
|
||||
Tests MUST be full-suite safe and order-independent, not merely file-local safe.
|
||||
- **No timing dependence.** NEVER `sleep`/`setTimeout`-race to "let it settle". Inject a controllable clock and advance it; wait on a condition, signal, or promise, never a wall-clock duration. Real-time waits are the #1 source of flake.
|
||||
- **No environment pollution.** NEVER leak env vars, temp files, global singletons, `process.env`/`process.platform`/`Bun.*` mutations, or monkeypatches past the test. Use per-test setup with restore in teardown. A test that passes alone but poisons a later file is broken.
|
||||
- **Deterministic.** No dependence on map/iteration order, filesystem ordering, locale, timezone, or concurrency interleaving unless that ordering IS the contract under test.
|
||||
- **Hermetic.** No real network or real time. Each test creates and tears down its own fixtures.
|
||||
</isolation>
|
||||
|
||||
<workflow>
|
||||
1. **Study the code under test.** Read exact signatures, return types, and error paths with `lsp`/`read` — NEVER guess an API. Spawn `explore` for unfamiliar areas.
|
||||
2. **Study existing tests.** Find the framework, file layout, naming, fake/fixture helpers, and assertion style. You MUST reuse them. `grep`/`glob` for sibling test files.
|
||||
3. **Enumerate contracts.** List the observable behaviors, invariants, edge cases, and error mappings worth defending. Drop anything that fails the `<critical>` litmus.
|
||||
4. **Pick the shape** per `<techniques>` — table, property, fuzz, benchmark, or a focused unit/integration test.
|
||||
5. **Write the tests**, matching repo conventions exactly. Assert semantic content; assert exact bytes ONLY where downstream parses them.
|
||||
6. **Run them and verify they have teeth.** Execute the suite with the repo's runner; confirm green. Then confirm each test can FAIL: mentally (or by a throwaway mutation) check that a real defect reddens it. A test you never saw fail is unproven.
|
||||
</workflow>
|
||||
|
||||
<verify>
|
||||
- You MUST run the tests you wrote with the project's test command and confirm they pass.
|
||||
- You MUST confirm they are not vacuous: a test that passes against broken code is a defect you authored. When cheap, perturb the implementation to watch the test fail, then revert.
|
||||
- Run ONLY the tests you added or touched unless asked for the full suite.
|
||||
- Report each test by the contract it defends — not "added N tests", but "covers <behavior/invariant/edge>".
|
||||
</verify>
|
||||
|
||||
<critical>
|
||||
- A test exists to FAIL on a real bug. No nameable contract, or no plausible bug would redden it → NEVER write it.
|
||||
- NEVER assert plumbing, restate the implementation, or grep the source. Test observable behavior through the public surface.
|
||||
- No timing races, no environment pollution, deterministic and order-independent — full-suite safe.
|
||||
- You MUST keep going until the tests are written, passing, and proven to have teeth.
|
||||
</critical>
|
||||
@@ -15,7 +15,7 @@ You decompose, dispatch, verify, and iterate. Substantial and parallelizable wor
|
||||
7. **Respawn, do not absorb.** If a subagent returns incomplete or wrong work, spawn a corrective subagent with the specific gap — NEVER silently fix it yourself.
|
||||
8. **No scope creep, no scope shrink.** NEVER add work the user did not ask for. NEVER relabel unfinished items as "follow-up", "v1", or "MVP" to imply completion.
|
||||
9. **Subagents do not verify, lint, or format.** Every `task` assignment MUST instruct the subagent to skip all gates and formatters. Their job is the edit only. You — the orchestrator — run verification and formatting **once** at the end of the phase across the union of changed files. Avoids redundant runs and racing formatter passes.
|
||||
10. **Right-size the offload — do not micro-task.** Subagents are for substantial or parallelizable chunks, not every keystroke. A trivial, self-contained mechanical edit — deleting a redundant glob, fixing one line in a config, renaming a single symbol in one file — costs less to *do* than to describe in a Goal/Constraints assignment. Make those yourself with `edit`/`write` and move on; reserve `task`/`quick_task` for work large enough to justify the dispatch overhead.
|
||||
10. **Right-size the offload — do not micro-task.** Subagents are for substantial or parallelizable chunks, not every keystroke. A trivial, self-contained mechanical edit — deleting a redundant glob, fixing one line in a config, renaming a single symbol in one file — costs less to *do* than to describe in a Goal/Constraints assignment. Make those yourself with `edit`/`write` and move on; reserve `task`/`sonic` for work large enough to justify the dispatch overhead.
|
||||
</rules>
|
||||
|
||||
<workflow>
|
||||
@@ -30,7 +30,7 @@ You decompose, dispatch, verify, and iterate. Substantial and parallelizable wor
|
||||
|
||||
<anti-patterns>
|
||||
- Doing substantial or parallelizable work yourself instead of fanning it out to subagents.
|
||||
- Wrapping a single trivial edit (e.g. removing one redundant config line) in a `task`/`quick_task` with full Goal/Constraints scaffolding — just make the edit inline.
|
||||
- Wrapping a single trivial edit (e.g. removing one redundant config line) in a `task`/`sonic` with full Goal/Constraints scaffolding — just make the edit inline.
|
||||
- Yielding after phase 1 with "ready to continue?".
|
||||
- Dispatching one subagent at a time when five could run in parallel.
|
||||
- Skipping `bun check` between phases because "the change looked safe".
|
||||
|
||||
@@ -184,9 +184,7 @@ EXECUTION WORKFLOW
|
||||
|
||||
# 5. Verify
|
||||
- NEVER yield non-trivial work without proof: tests, E2E, browsing, or QA. Run only tests you added or modified unless asked otherwise.
|
||||
- Prefer unit or runnable E2E tests. NEVER create mocks.
|
||||
- Test behavior, not plumbing—things that can actually break.
|
||||
- Don't test defaults: a config or string change shouldn't break the test. Assert logical behavior, not current state.
|
||||
- Test behavior, using tester agent where available. Assert logical behavior, not current state.
|
||||
- Aim at conditional branches, edge values, invariants across fields, and error handling versus silent broken results.
|
||||
|
||||
# 6. Cleanup
|
||||
@@ -201,7 +199,6 @@ DELIVERY CONTRACT
|
||||
<contract>
|
||||
Inviolable.
|
||||
- NEVER yield unless the deliverable is complete. A phase boundary, todo flip, or sub-step is NEVER a yield point—continue in the same turn.
|
||||
- NEVER suppress tests to make code pass.
|
||||
- NEVER fabricate outputs. Claims about code, tools, tests, docs, or sources MUST be grounded.
|
||||
- NEVER substitute an easier or more familiar problem:
|
||||
- Don't infer extra scope—retries, validation, telemetry, abstraction “while you're at it”—because it changes the contract.
|
||||
@@ -223,7 +220,7 @@ Inviolable.
|
||||
- Output format MUST match the ask.
|
||||
- Every claim about code, tools, tests, docs, or sources MUST be grounded.
|
||||
- Mark any claim not directly observed or established as `[INFERENCE]`.
|
||||
- Verification claims MUST match what was exercised. Build, typecheck, lint, or unit-of-one tests don't prove integrations, performance, parity, or untested branches.
|
||||
- Verification claims MUST match what was exercised, preferably smoke tested.
|
||||
- No required tool lookup may be skipped when it would cut uncertainty.
|
||||
- Be brief in prose, not in evidence, verification, or blocking details.
|
||||
</evidence-and-output>
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
<system-interrupt reason="thinking_loop_detected">
|
||||
The loop guard interrupted your previous turn: your reasoning or response repeated near-identical content without making progress. Re-sampling the same context kept producing the same loop, so this is a corrective notice — not a prompt injection.
|
||||
|
||||
Restating the same plan, summary, or intention again will loop again. Break the pattern now:
|
||||
- STOP narrating what you are about to do. Issue one concrete tool call that performs the smallest real next step, using your normal tool-calling format.
|
||||
- If you were stuck deciding between options, pick the most boring viable one and act; do not deliberate further.
|
||||
- If the task is genuinely complete, emit your final answer instead of more reasoning.
|
||||
|
||||
Do something different from the looped content. Act, don't re-plan.
|
||||
</system-interrupt>
|
||||
@@ -13,7 +13,7 @@ Worth it when the task benefits from decomposition + parallel coverage, or from
|
||||
<helpers>
|
||||
State persists across eval calls, so scout in one call and fan out in the next. Every eval call has:
|
||||
|
||||
- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", "oracle", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts.
|
||||
- `agent(prompt, *, agent="task", model=None, label=None, schema=None, isolated=None, apply=None, merge=None, handle=False)` — run ONE subagent; returns its final text, or the validated object when `schema` (a JSON Schema dict) is given. With `schema` the subagent is forced to emit structured output that is validated for you — branch on the object, not on parsed prose. `agent` picks a discovered agent ("explore", "reviewer", …); `label` names the artifact. Shared background goes in a `local://` file referenced from each prompt, not a parameter. Subagents are told their final text IS the return value, so they hand back raw data. `agent()` blocks until the subagent finishes; eval-spawned agents nest at most 3 deep. Pass `isolated=True` to run the spawn in a copy-on-write worktree so parallel `agent()` calls can edit overlapping files safely — strict opt-in, mirrors the `task` tool, defaults off regardless of `task.isolation.mode`; `isolated=True` while the setting is `"none"` errors out instead of silently downgrading. With isolation, `apply=False` keeps changes in the worktree, and `merge=False` forces patch mode even when the setting is `"branch"`. Captured root patch path, branch name, nested repo patches, and apply summary reach the workflow through `handle=True` — combine it with `apply=False` (or `apply=False, schema=…`) and read `node["patch_path"]`, `node["branch_name"]`, `node["nested_patches"]`, `node["changes_applied"]`, `node["isolation_summary"]` (JS: same keys camelCased) to recover artifacts.
|
||||
- `parallel(thunks)` — run zero-arg callables concurrently through a bounded pool, preserving input order; returns once all finish. The pool is bounded by the session's `task` concurrency — don't hand-tune it; fan out as wide as the work divides. A thunk that raises propagates — wrap risky work in `try/except` inside the thunk to keep partial results. In a loop, bind each closure's value with a default arg (`lambda d=d: …`) or every thunk captures the last one.
|
||||
- `pipeline(items, *stages)` — map items through `stages` left-to-right. There is a BARRIER between stages: ALL items clear stage N before stage N+1 begins. Each stage is a one-arg callable; stage 1 gets the original item, later stages get the previous result. Same pool width as `parallel()`.
|
||||
- `completion(prompt, *, model="default", system=None, schema=None)` — oneshot, stateless model call (no tools, no history). Tiers: "smol", "default", "slow". Cheap classification/scoring inside a fan-out.
|
||||
|
||||
@@ -29,13 +29,10 @@ Anything below → `eval` cell, not bash:
|
||||
|
||||
<critical>
|
||||
- Bash invokes real binaries with simple args; it is NOT a scripting surface. Loops, conditionals, heredocs, inline interpreter scripts (`-e`/`-c`/`--eval`) when an eval runtime exists, several piped stages, or quote/JSON escaping mean you're writing a program → use `eval` cells: restartable, stateful, and free of shell-quoting traps.
|
||||
- NEVER shell out to search content or files: `grep/rg` → `grep`.
|
||||
- NEVER use `ls` or `find` to list or locate files — `ls` → `read` (a directory path lists entries), `find` → the `glob` tool (globbing). This is non-negotiable, even for a single quick listing.
|
||||
- Avoid head/tail/redirections: stderr already merged; long output auto-truncated, FULL capture kept at `artifact://<id>`.
|
||||
</critical>
|
||||
|
||||
<output>
|
||||
- Returns output; exit code shown on non-zero exit.
|
||||
- Returns output (stderr merged into stdout); exit code shown on non-zero exit.
|
||||
- Truncated output → `artifact://<id>` (linked in metadata).
|
||||
</output>
|
||||
|
||||
|
||||
@@ -7,11 +7,10 @@ Execution blocks your turn: the call only returns once the work is completely fi
|
||||
- **Sequence only when necessary:** The only reason to run A before B is if B strictly requires A's output to function (e.g., a core API contract or schema migration). {{#if ircEnabled}}If the missing piece is small, run them in parallel and have B ask A via `irc`!{{/if}}
|
||||
- **Role matching:** Assign each subagent a specific `role` (e.g. "Security Reviewer", "DB Migrator"). Do not spawn generic workers.
|
||||
- **No overhead:** Each assignment MUST instruct its agent to skip formatters, linters, and project-wide test suites. You will run those once at the end.
|
||||
- **Do your own thinking:** NEVER assign reasoning, architecture, or design to `quick_task` or `explore`. They are for mechanical lookups only. Keep hard decisions in your own context or use `task`, `plan`, or `oracle`.
|
||||
- **One-pass agents:** Prefer agents that investigate **and** edit in a single pass; only spin a read-only discovery step (e.g. `explore`) when the affected files are genuinely unknown.
|
||||
|
||||
# Inputs
|
||||
- `agent`: The base agent type to use (e.g., `task`, `explore`).
|
||||
- `agent` (optional): The base agent type to use (e.g., `explore`, `reviewer`). Defaults to `task` (the general-purpose worker) — omit it for the default worker instead of passing `agent: "task"`.
|
||||
{{#if batchEnabled}}
|
||||
- `context`: Shared project state, constraints, and contracts. Applies to the entire batch; do not duplicate this background into individual tasks.
|
||||
- `tasks[]`: Array of subagents to spawn.
|
||||
@@ -37,13 +36,7 @@ Subagents start blank. They have no access to your conversation history.
|
||||
{{#if batchEnabled}}
|
||||
- Pass large payloads using `local://<path>` URIs, never inline text.
|
||||
{{else}}
|
||||
- *Note: The single-spawn shape has no `context` field.* Write shared project state ONCE to a `local://` file (e.g., `local://ctx.md`) and reference that URL in your assignments. Pass large payloads using `local://<path>` URIs, never inline text.
|
||||
{{/if}}
|
||||
{{#if ircEnabled}}
|
||||
- Once spawned, coordinate with live agents via `irc` using their IDs. If task B depends on task A, B SHOULD message A directly.
|
||||
{{/if}}
|
||||
{{#if asyncEnabled}}
|
||||
- If you run out of things to do and are genuinely blocked waiting for a subagent, use `job poll`. Use `job cancel` only for stalled work.
|
||||
- Write shared project state ONCE to a `local://` file (e.g., `local://ctx.md`) and reference that URL in your assignments.
|
||||
{{/if}}
|
||||
|
||||
# Format Contracts
|
||||
|
||||
@@ -42,6 +42,7 @@ import {
|
||||
resolveModelRoleValue,
|
||||
} from "./config/model-resolver";
|
||||
import { loadPromptTemplates as loadPromptTemplatesInternal, type PromptTemplate } from "./config/prompt-templates";
|
||||
import { buildServiceTierByFamily } from "./config/service-tier";
|
||||
import { Settings, type SkillsSettings } from "./config/settings";
|
||||
import { CursorExecHandlers } from "./cursor";
|
||||
import "./discovery";
|
||||
@@ -1559,7 +1560,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
getModelString: () => (hasExplicitModel && model ? formatModelString(model) : undefined),
|
||||
getActiveModelString,
|
||||
getActiveModel: () => agent?.state.model ?? model,
|
||||
getServiceTier: () => session?.serviceTier,
|
||||
getServiceTierByFamily: () => session?.serviceTierByFamily,
|
||||
getImageAttachments: () => session?.getImageAttachments() ?? [],
|
||||
getPlanModeState: () => session?.getPlanModeState(),
|
||||
getPlanReferencePath: () => session?.getPlanReferencePath() ?? "local://PLAN.md",
|
||||
@@ -2547,13 +2548,13 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
const openaiWebsocketSetting = settings.get("providers.openaiWebsockets") ?? "off";
|
||||
const preferOpenAICodexWebsockets =
|
||||
openaiWebsocketSetting === "on" ? true : openaiWebsocketSetting === "off" ? false : undefined;
|
||||
const serviceTierSetting = settings.get("serviceTier");
|
||||
|
||||
const initialServiceTier = hasServiceTierEntry
|
||||
? existingSession.serviceTier
|
||||
: serviceTierSetting === "none"
|
||||
? undefined
|
||||
: serviceTierSetting;
|
||||
const initialServiceTierByFamily = hasServiceTierEntry
|
||||
? (existingSession.serviceTier ?? {})
|
||||
: buildServiceTierByFamily(
|
||||
settings.get("tier.openai"),
|
||||
settings.get("tier.anthropic"),
|
||||
settings.get("tier.google"),
|
||||
);
|
||||
|
||||
// One-shot launch-latency marker: fired the first time the loop dispatches
|
||||
// a chat request to the provider transport. See onFirstChatDispatch.
|
||||
@@ -2601,7 +2602,6 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
minP: settings.get("minP") >= 0 ? settings.get("minP") : undefined,
|
||||
presencePenalty: settings.get("presencePenalty") >= 0 ? settings.get("presencePenalty") : undefined,
|
||||
repetitionPenalty: settings.get("repetitionPenalty") >= 0 ? settings.get("repetitionPenalty") : undefined,
|
||||
serviceTier: initialServiceTier,
|
||||
hideThinkingSummary: settings.get("omitThinking"),
|
||||
kimiApiFormat: settings.get("providers.kimiApiFormat") ?? "anthropic",
|
||||
preferWebsockets: preferOpenAICodexWebsockets,
|
||||
@@ -2661,8 +2661,8 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
// classification persists its concrete effort once a real user turn runs.
|
||||
sessionManager.appendThinkingLevelChange(effectiveThinkingLevel);
|
||||
}
|
||||
if (initialServiceTier) {
|
||||
sessionManager.appendServiceTierChange(initialServiceTier);
|
||||
if (Object.keys(initialServiceTierByFamily).length > 0) {
|
||||
sessionManager.appendServiceTierChange(initialServiceTierByFamily);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2713,6 +2713,7 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
agent,
|
||||
pruneToolDescriptions: inlineToolDescriptors,
|
||||
thinkingLevel: autoThinking ? AUTO_THINKING : effectiveThinkingLevel,
|
||||
serviceTierByFamily: initialServiceTierByFamily,
|
||||
sessionManager,
|
||||
settings,
|
||||
autoApprove: options.autoApprove,
|
||||
|
||||
@@ -92,6 +92,8 @@ import type {
|
||||
ResetCreditRedeemOutcome,
|
||||
ResetCreditTarget,
|
||||
ServiceTier,
|
||||
ServiceTierByFamily,
|
||||
ServiceTierFamily,
|
||||
SimpleStreamOptions,
|
||||
TextContent,
|
||||
ToolCall,
|
||||
@@ -105,7 +107,9 @@ import {
|
||||
deriveClaudeDeviceId,
|
||||
Effort,
|
||||
parseRateLimitReason,
|
||||
resolveServiceTier,
|
||||
realizesPriorityServiceTier,
|
||||
resolveModelServiceTier,
|
||||
serviceTierFamily,
|
||||
streamSimple,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import * as AIError from "@oh-my-pi/pi-ai/error";
|
||||
@@ -168,7 +172,7 @@ import {
|
||||
} from "../config/model-resolver";
|
||||
import { MODEL_ROLE_IDS, MODEL_ROLES } from "../config/model-roles";
|
||||
import { expandPromptTemplate, type PromptTemplate } from "../config/prompt-templates";
|
||||
import { resolveServiceTierSetting } from "../config/service-tier";
|
||||
import { buildServiceTierByFamily, serviceTierForAllFamilies, serviceTierSettingToTier } from "../config/service-tier";
|
||||
import type { Settings, SkillsSettings } from "../config/settings";
|
||||
import { getDefault, onAppendOnlyModeChanged, validateProviderMaxInFlightRequests } from "../config/settings";
|
||||
import { RawSseDebugBuffer } from "../debug/raw-sse-buffer";
|
||||
@@ -246,6 +250,7 @@ import planModeToolDecisionReminderPrompt from "../prompts/system/plan-mode-tool
|
||||
type: "text",
|
||||
};
|
||||
import sideChannelNoToolsReminder from "../prompts/system/side-channel-no-tools.md" with { type: "text" };
|
||||
import thinkingLoopRedirectTemplate from "../prompts/system/thinking-loop-redirect.md" with { type: "text" };
|
||||
import ttsrInterruptTemplate from "../prompts/system/ttsr-interrupt.md" with { type: "text" };
|
||||
import ttsrToolReminderTemplate from "../prompts/system/ttsr-tool-reminder.md" with { type: "text" };
|
||||
import unexpectedStopRetryTemplate from "../prompts/system/unexpected-stop-retry.md" with { type: "text" };
|
||||
@@ -343,6 +348,9 @@ const SESSION_STOP_CONTINUATION_CAP = 8;
|
||||
const GEMINI_HEADER_INTERRUPT_REASON = "Interrupted: emit a tool call instead of more planning";
|
||||
/** `customType` for the hidden tool-call reminder injected after the interrupt. */
|
||||
const GEMINI_TOOL_REMINDER_TYPE = "gemini-tool-call-reminder";
|
||||
/** `customType` for the hidden redirect notice injected into a turn retried after a
|
||||
* thinking/response loop. Steers the model off the repeated content; never displayed. */
|
||||
const THINKING_LOOP_REDIRECT_TYPE = "thinking-loop-redirect";
|
||||
|
||||
// A side-channel assistant response is signed for the hidden prompt/history that
|
||||
// produced it. If we persist that response under a different user turn, native
|
||||
@@ -507,6 +515,8 @@ export interface AgentSessionConfig {
|
||||
scopedModels?: Array<{ model: Model; thinkingLevel?: ThinkingLevel }>;
|
||||
/** Initial session thinking selector. */
|
||||
thinkingLevel?: ConfiguredThinkingLevel;
|
||||
/** Initial per-family service tiers (OpenAI / Anthropic / Google) for the live session. */
|
||||
serviceTierByFamily?: ServiceTierByFamily;
|
||||
/** Prompt templates for expansion */
|
||||
promptTemplates?: PromptTemplate[];
|
||||
/** File-based slash commands for expansion */
|
||||
@@ -1788,6 +1798,7 @@ export class AgentSession {
|
||||
// toggle scopes priority to Fireworks alone, without mutating the shared
|
||||
// session `serviceTier` that drives `/fast` and OpenAI/Anthropic priority.
|
||||
this.agent.serviceTierResolver = model => this.#effectiveServiceTier(model);
|
||||
this.#serviceTierByFamily = config.serviceTierByFamily ?? {};
|
||||
this.#advisorTools = config.advisorTools;
|
||||
this.#advisorWatchdogPrompt = config.advisorWatchdogPrompt;
|
||||
this.#advisorSharedInstructions = config.advisorSharedInstructions;
|
||||
@@ -2046,15 +2057,20 @@ export class AgentSession {
|
||||
const legacy = !this.#advisorConfigs?.length;
|
||||
const roster: AdvisorConfig[] = legacy ? [{ name: "default" }] : this.#advisorConfigs!;
|
||||
|
||||
// Advisor service tier (`serviceTierAdvisor`): "none" (default) runs the
|
||||
// advisor on standard processing; "inherit" tracks the session's live tier
|
||||
// per request (like the main agent, including /fast toggles) via a resolver;
|
||||
// a concrete value pins the advisor to that tier. One value for all advisors.
|
||||
const advisorTierSetting = this.settings.get("serviceTierAdvisor");
|
||||
const advisorServiceTier =
|
||||
advisorTierSetting === "inherit" ? undefined : resolveServiceTierSetting(advisorTierSetting, undefined);
|
||||
const advisorServiceTierResolver =
|
||||
advisorTierSetting === "inherit" ? (model: Model) => this.#effectiveServiceTier(model) : undefined;
|
||||
// Advisor service tier (`tier.advisor`): "none" (default) runs the advisor
|
||||
// on standard processing; "inherit" tracks the session's live per-family
|
||||
// tiers per request (like the main agent, including /fast toggles); a
|
||||
// concrete value is broadcast across families and applied to the advisor
|
||||
// model's family. One value for all advisors.
|
||||
const advisorTierSetting = this.settings.get("tier.advisor");
|
||||
const advisorTierMap =
|
||||
advisorTierSetting === "inherit"
|
||||
? undefined
|
||||
: serviceTierForAllFamilies(serviceTierSettingToTier(advisorTierSetting));
|
||||
const advisorServiceTierResolver = (model: Model): ServiceTier | undefined =>
|
||||
advisorTierSetting === "inherit"
|
||||
? this.#effectiveServiceTier(model)
|
||||
: resolveModelServiceTier(advisorTierMap, model);
|
||||
|
||||
const usedSlugs = new Set<string>();
|
||||
for (const config of roster) {
|
||||
@@ -2153,7 +2169,7 @@ export class AgentSession {
|
||||
transformProviderContext: this.#transformProviderContext,
|
||||
intentTracing: false,
|
||||
telemetry: advisorTelemetry,
|
||||
serviceTier: advisorServiceTier,
|
||||
serviceTier: undefined,
|
||||
serviceTierResolver: advisorServiceTierResolver,
|
||||
});
|
||||
advisorAgent.setDisableReasoning(shouldDisableReasoning(advisorThinkingLevel));
|
||||
@@ -2871,10 +2887,10 @@ export class AgentSession {
|
||||
* the mid-run-compaction planner can ask "is this turn message already on
|
||||
* the branch?" in O(1) instead of re-walking the branch per check.
|
||||
*
|
||||
* The Map's value is the list of branch messages that share a key — almost
|
||||
* always one. We only need the LIST when content equality matters (rare
|
||||
* collision tiebreaker via {@link sameMessageContent}); the empty/single-
|
||||
* entry common case lets the caller's lookup short-circuit at presence.
|
||||
* The mid-run ordering check uses key identity alone: same-key content
|
||||
* variants are one logical message at this boundary, because otherwise a
|
||||
* display-side rewrite can make the assistant look missing after its tool
|
||||
* results have already persisted.
|
||||
*
|
||||
* Pre-#3629 the equivalent was `sessionManager.getBranch()` called twice
|
||||
* per turn message, each call rebuilding the path via O(n²) `unshift` and
|
||||
@@ -2882,17 +2898,14 @@ export class AgentSession {
|
||||
* per `onTurnEnd` on a long session and the load-bearing source of the
|
||||
* `ui.loop-blocked` warnings in the bug report.
|
||||
*/
|
||||
#indexPersistedMessagesByKey(): Map<string, AgentMessage[]> {
|
||||
const index = new Map<string, AgentMessage[]>();
|
||||
#indexPersistedMessageKeys(): Set<string> {
|
||||
const keys = new Set<string>();
|
||||
for (const entry of this.sessionManager.getBranch()) {
|
||||
if (entry.type !== "message") continue;
|
||||
const key = sessionMessagePersistenceKey(entry.message);
|
||||
if (key === undefined) continue;
|
||||
const existing = index.get(key);
|
||||
if (existing) existing.push(entry.message);
|
||||
else index.set(key, [entry.message]);
|
||||
if (key !== undefined) keys.add(key);
|
||||
}
|
||||
return index;
|
||||
return keys;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -2992,17 +3005,17 @@ export class AgentSession {
|
||||
// JSON-compared every entry per turn message, which on long sessions
|
||||
// turned each `onTurnEnd` into a seconds-long sync block (the
|
||||
// `ui.loop-blocked` warnings tagged `subagent:*` in the bug report).
|
||||
const branchIndex = this.#indexPersistedMessagesByKey();
|
||||
const branchKeys = this.#indexPersistedMessageKeys();
|
||||
const turnKeys = turnMessages.map(sessionMessagePersistenceKey);
|
||||
const persistedKeys = new Set<string>();
|
||||
for (let index = 0; index < turnMessages.length; index++) {
|
||||
const key = turnKeys[index];
|
||||
if (key === undefined) continue;
|
||||
const candidates = branchIndex.get(key);
|
||||
if (!candidates) continue;
|
||||
// Key match only counts when content also matches — two distinct
|
||||
// messages that collided on the cheap key must STILL be persisted.
|
||||
if (candidates.some(persisted => sameMessageContent(persisted, turnMessages[index]))) {
|
||||
// Mid-run ordering is keyed by logical identity. A persisted display
|
||||
// variant (for example, redacted/deobfuscated content) must still count;
|
||||
// otherwise the assistant can look missing while later tool results are
|
||||
// present, producing a false out-of-order skip.
|
||||
if (branchKeys.has(key)) {
|
||||
persistedKeys.add(key);
|
||||
}
|
||||
}
|
||||
@@ -3216,10 +3229,11 @@ export class AgentSession {
|
||||
if (event.message.role === "assistant") {
|
||||
this.#lastAssistantMessage = event.message;
|
||||
const assistantMsg = event.message as AssistantMessage;
|
||||
const currentGrantsAnthropicPriority =
|
||||
this.serviceTier === "priority" || this.serviceTier === "claude-only";
|
||||
if (assistantMsg.disabledFeatures?.includes("priority") && currentGrantsAnthropicPriority) {
|
||||
this.setServiceTier(undefined);
|
||||
if (
|
||||
assistantMsg.disabledFeatures?.includes("priority") &&
|
||||
this.#serviceTierByFamily.anthropic === "priority"
|
||||
) {
|
||||
this.setServiceTierFamily("anthropic", undefined);
|
||||
this.emitNotice(
|
||||
"warning",
|
||||
"Priority/fast mode rejected for this model; retried without it. Fast mode is now off.",
|
||||
@@ -5204,8 +5218,11 @@ export class AgentSession {
|
||||
return this.#autoResolvedLevel;
|
||||
}
|
||||
|
||||
get serviceTier(): ServiceTier | undefined {
|
||||
return this.agent.serviceTier;
|
||||
#serviceTierByFamily: ServiceTierByFamily = {};
|
||||
|
||||
/** Live per-family service tiers (OpenAI / Anthropic / Google). */
|
||||
get serviceTierByFamily(): ServiceTierByFamily {
|
||||
return this.#serviceTierByFamily;
|
||||
}
|
||||
|
||||
/** Whether agent is currently streaming a response */
|
||||
@@ -7895,7 +7912,7 @@ export class AgentSession {
|
||||
this.#scheduledHiddenNextTurnGeneration = undefined;
|
||||
|
||||
this.sessionManager.appendThinkingLevelChange(this.thinkingLevel, this.configuredThinkingLevel());
|
||||
this.sessionManager.appendServiceTierChange(this.serviceTier ?? null);
|
||||
this.sessionManager.appendServiceTierChange(this.#serviceTierEntry());
|
||||
if (nextDiscoverySessionToolNames) {
|
||||
await this.#applyActiveToolsByName(nextDiscoverySessionToolNames, { persistMCPSelection: false });
|
||||
if (this.getSelectedMCPToolNames().length > 0) {
|
||||
@@ -8074,7 +8091,7 @@ export class AgentSession {
|
||||
*/
|
||||
async setModelTemporary(
|
||||
model: Model,
|
||||
thinkingLevel?: ThinkingLevel,
|
||||
thinkingLevel?: ConfiguredThinkingLevel,
|
||||
options?: { ephemeral?: boolean },
|
||||
): Promise<void> {
|
||||
const previousEditMode = this.#resolveActiveEditMode();
|
||||
@@ -8437,38 +8454,36 @@ export class AgentSession {
|
||||
}
|
||||
|
||||
/**
|
||||
* True when *any* fast-mode-granting service tier is configured, regardless
|
||||
* of whether the active model's provider actually realizes it. Used by the
|
||||
* toggle (`/fast on|off`) so re-toggling a scoped tier (`openai-only`,
|
||||
* `claude-only`) doesn't silently broaden it to unscoped `priority`.
|
||||
* True when the currently selected model's family is set to `priority` — the
|
||||
* `/fast` on/off state for the active model. Returns false when no model is
|
||||
* selected or the model exposes no service-tier family (e.g. Fireworks, which
|
||||
* has its own Providers › Fireworks Tier toggle).
|
||||
*
|
||||
* For "is fast mode actually applied to the next request?" use
|
||||
* {@link isFastModeActive} instead — that one respects the model's provider.
|
||||
* For "is priority actually applied to the next request?" use
|
||||
* {@link isFastModeActive} instead.
|
||||
*/
|
||||
isFastModeEnabled(): boolean {
|
||||
return (
|
||||
this.serviceTier === "priority" || this.serviceTier === "claude-only" || this.serviceTier === "openai-only"
|
||||
);
|
||||
const family = this.model ? serviceTierFamily(this.model) : undefined;
|
||||
return family ? this.#serviceTierByFamily[family] === "priority" : false;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the configured `serviceTier` resolves to `"priority"` for the
|
||||
* *currently selected model's provider*. Returns false for scoped tiers
|
||||
* that don't match (e.g. `"openai-only"` on an anthropic model) and when
|
||||
* no model is selected.
|
||||
* True when `priority` is actually realized on the wire for the currently
|
||||
* selected model (OpenAI/Google `service_tier`, direct Anthropic fast mode,
|
||||
* or Fireworks priority). Returns false for tiers the active model can't
|
||||
* realize and when no model is selected.
|
||||
*/
|
||||
isFastModeActive(): boolean {
|
||||
return resolveServiceTier(this.#effectiveServiceTier(), this.model?.provider) === "priority";
|
||||
const model = this.model;
|
||||
return !!model && realizesPriorityServiceTier(this.#effectiveServiceTier(model), model);
|
||||
}
|
||||
|
||||
/**
|
||||
* Effective wire service-tier for a request to `model`. Fireworks models
|
||||
* take the Priority serving path only when the Providers › Fireworks Tier
|
||||
* setting is `"priority"` — that toggle is the sole opt-in, so a global
|
||||
* `serviceTier: "priority"` (for OpenAI/Anthropic) never silently incurs
|
||||
* Fireworks priority costs — and never for `-fast` variants, whose Fast
|
||||
* serving path is mutually exclusive with Priority. Every other provider
|
||||
* uses the session `serviceTier` unchanged.
|
||||
* Effective wire service-tier for a request to `model`. Fireworks models take
|
||||
* the Priority serving path only when the Providers › Fireworks Tier setting
|
||||
* is `"priority"` (and never for `-fast` variants, whose Fast serving path is
|
||||
* mutually exclusive with Priority). Every other model resolves the live
|
||||
* per-family tier map down to the entry for its family.
|
||||
*/
|
||||
#effectiveServiceTier(model: Model | undefined = this.model): ServiceTier | undefined {
|
||||
if (model?.provider === "fireworks") {
|
||||
@@ -8476,40 +8491,56 @@ export class AgentSession {
|
||||
? "priority"
|
||||
: undefined;
|
||||
}
|
||||
return this.serviceTier;
|
||||
if (!model) return undefined;
|
||||
return resolveModelServiceTier(this.#serviceTierByFamily, model);
|
||||
}
|
||||
|
||||
setServiceTier(serviceTier: ServiceTier | undefined): void {
|
||||
if (this.serviceTier === serviceTier) return;
|
||||
// Re-arming priority on Anthropic? Clear the per-session auto-fallback
|
||||
// sticky disable so the next request actually carries `speed: "fast"`
|
||||
// again. Without this, `/fast on` (or user switching to a tier that
|
||||
// grants anthropic priority) after an auto-disable is a silent no-op
|
||||
// and the warning notice fires every turn.
|
||||
if (serviceTier === "priority" || serviceTier === "claude-only") {
|
||||
/** The live per-family tier map, or `null` when empty (for session persistence). */
|
||||
#serviceTierEntry(): ServiceTierByFamily | null {
|
||||
return Object.keys(this.#serviceTierByFamily).length > 0 ? this.#serviceTierByFamily : null;
|
||||
}
|
||||
|
||||
/** Set one family's tier (or clear it with `undefined`); persists the change. */
|
||||
setServiceTierFamily(family: ServiceTierFamily, tier: ServiceTier | undefined): void {
|
||||
if (this.#serviceTierByFamily[family] === tier) return;
|
||||
const next: ServiceTierByFamily = { ...this.#serviceTierByFamily };
|
||||
if (tier) next[family] = tier;
|
||||
else delete next[family];
|
||||
this.#applyServiceTierByFamily(next);
|
||||
}
|
||||
|
||||
/** Replace the whole per-family tier map; persists + re-arms Anthropic fast mode. */
|
||||
#applyServiceTierByFamily(next: ServiceTierByFamily): void {
|
||||
// Re-arming Anthropic priority clears the per-session fast-mode auto-disable
|
||||
// so the next request actually carries `speed: "fast"` again.
|
||||
if (next.anthropic === "priority" && this.#serviceTierByFamily.anthropic !== "priority") {
|
||||
clearAnthropicFastModeFallback(this.#providerSessionState);
|
||||
}
|
||||
this.agent.serviceTier = serviceTier;
|
||||
this.sessionManager.appendServiceTierChange(serviceTier ?? null);
|
||||
this.#serviceTierByFamily = next;
|
||||
this.sessionManager.appendServiceTierChange(this.#serviceTierEntry());
|
||||
}
|
||||
|
||||
/**
|
||||
* `/fast on|off` targets the family of the currently selected model: it sets
|
||||
* (or clears) that family's `priority` tier. Models without a service-tier
|
||||
* family (Fireworks, or providers with no tier knob) have nothing to toggle.
|
||||
*/
|
||||
setFastMode(enabled: boolean): void {
|
||||
if (enabled && this.isFastModeEnabled()) {
|
||||
// Already on under any scope — keep the user's scoped value.
|
||||
const family = this.model ? serviceTierFamily(this.model) : undefined;
|
||||
if (!family) {
|
||||
this.emitNotice("info", "The current model has no service-tier control for /fast to toggle.", "priority");
|
||||
return;
|
||||
}
|
||||
if (!enabled) {
|
||||
this.setServiceTier(undefined);
|
||||
if (this.#serviceTierByFamily[family] === "priority") this.setServiceTierFamily(family, undefined);
|
||||
return;
|
||||
}
|
||||
const scope = this.settings.get("fastModeScope");
|
||||
this.setServiceTier(scope === "openai" ? "openai-only" : scope === "claude" ? "claude-only" : "priority");
|
||||
this.setServiceTierFamily(family, "priority");
|
||||
}
|
||||
|
||||
toggleFastMode(): boolean {
|
||||
const enabled = !this.isFastModeEnabled();
|
||||
this.setFastMode(enabled);
|
||||
return enabled;
|
||||
this.setFastMode(!this.isFastModeEnabled());
|
||||
return this.isFastModeEnabled();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -8941,6 +8972,22 @@ export class AgentSession {
|
||||
...(snapcompactShapeSetting === "auto" ? {} : { shape }),
|
||||
maxFrames,
|
||||
});
|
||||
const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult);
|
||||
if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) {
|
||||
logger.warn("Snapcompact exceeded the per-request frame payload budget", {
|
||||
model: this.model?.id,
|
||||
framePayloadBytes,
|
||||
budget: snapcompact.FRAME_DATA_BYTES_BUDGET,
|
||||
});
|
||||
this.emitNotice(
|
||||
"warning",
|
||||
"snapcompact produced too much standing image payload. No LLM fallback was attempted.",
|
||||
"compaction",
|
||||
);
|
||||
throw new Error(
|
||||
"snapcompact cannot run locally: standing image payload exceeds the per-request budget.",
|
||||
);
|
||||
}
|
||||
const ctxWindow = this.model?.contextWindow ?? 0;
|
||||
const budget =
|
||||
ctxWindow > 0
|
||||
@@ -10903,7 +10950,7 @@ export class AgentSession {
|
||||
*/
|
||||
#computeSnapcompactMaxFrames(preparation: CompactionPreparation, settings: CompactionSettings): number {
|
||||
const ctxWindow = this.model?.contextWindow ?? 0;
|
||||
if (ctxWindow <= 0) return snapcompact.MAX_FRAMES_DEFAULT;
|
||||
if (ctxWindow <= 0) return Math.min(snapcompact.MAX_FRAMES_DEFAULT, snapcompact.maxFramesForDataBudget());
|
||||
const reserve = effectiveReserveTokens(ctxWindow, settings);
|
||||
let baseTokens = computeNonMessageTokens(this);
|
||||
for (const message of preparation.recentMessages) {
|
||||
@@ -10942,7 +10989,16 @@ export class AgentSession {
|
||||
const capReserve = textEdgeTokens + SUMMARY_TEMPLATE_TOKENS;
|
||||
const frameBudget = totalBudget - baseTokens - capReserve;
|
||||
if (frameBudget < snapcompact.FRAME_TOKEN_ESTIMATE) return 1;
|
||||
return Math.min(Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE), snapcompact.MAX_FRAMES_DEFAULT);
|
||||
return Math.min(
|
||||
Math.floor(frameBudget / snapcompact.FRAME_TOKEN_ESTIMATE),
|
||||
snapcompact.MAX_FRAMES_DEFAULT,
|
||||
snapcompact.maxFramesForDataBudget(),
|
||||
);
|
||||
}
|
||||
|
||||
#snapcompactFramePayloadBytes(result: snapcompact.CompactionResult): number {
|
||||
const archive = snapcompact.getPreservedArchive(result.preserveData);
|
||||
return archive ? snapcompact.frameDataBytes(archive.frames) : 0;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -10955,7 +11011,9 @@ export class AgentSession {
|
||||
*/
|
||||
#projectSnapcompactContextTokens(preparation: CompactionPreparation, result: snapcompact.CompactionResult): number {
|
||||
const archive = snapcompact.getPreservedArchive(result.preserveData);
|
||||
const blocks = archive ? snapcompact.historyBlocks(archive) : undefined;
|
||||
const blocks = archive
|
||||
? snapcompact.historyBlocks(archive, { maxFrameDataBytes: snapcompact.FRAME_DATA_BYTES_BUDGET })
|
||||
: undefined;
|
||||
const summaryMessage = createCompactionSummaryMessage(
|
||||
result.summary,
|
||||
result.tokensBefore,
|
||||
@@ -11046,6 +11104,52 @@ export class AgentSession {
|
||||
return residualTokens <= fitBudget;
|
||||
}
|
||||
|
||||
/**
|
||||
* Last-resort reducer when {@link #runAutoCompaction} would otherwise dead-end.
|
||||
* The summarizer cut at the only available turn boundary, but the kept tail is
|
||||
* still over the recovery band because a single recent turn (a large
|
||||
* tool-result, a heavy fenced/XML block) is itself bigger than the band and
|
||||
* `findCutPoint` cannot cut inside one message. `shake("elide")` reaches INSIDE
|
||||
* that tail — it offloads heavy tool-result / block content to one
|
||||
* `artifact://` blob and leaves a recoverable placeholder — so residual context
|
||||
* genuinely drops instead of the guard pausing maintenance and looping the
|
||||
* warning. Without it the guard would pause/warn here; with it the caller
|
||||
* re-tests its progress predicate after the elide pass and only falls through
|
||||
* to the warning when residual stays over.
|
||||
*
|
||||
* Image-only tails are out of scope: `collectShakeRegions` skips image-only
|
||||
* tool results and user-message images aren't counted by the local estimate
|
||||
* that gates the dead-end, so those still surface the warning (remedy:
|
||||
* `/shake images`).
|
||||
*
|
||||
* Returns the elide {@link ShakeResult} when something was offloaded (so the
|
||||
* caller can re-test and report), or `undefined` when nothing was eligible or
|
||||
* the pass aborted/failed.
|
||||
*/
|
||||
async #tryShakeRescueForDeadEnd(signal: AbortSignal): Promise<ShakeResult | undefined> {
|
||||
if (signal.aborted) return undefined;
|
||||
try {
|
||||
const result = await this.shake("elide", { signal });
|
||||
return result.toolResultsDropped + result.blocksDropped > 0 ? result : undefined;
|
||||
} catch (error) {
|
||||
logger.warn("Dead-end shake rescue failed", {
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
});
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
/** Notice describing a successful dead-end elide rescue. */
|
||||
#emitShakeRescueNotice(result: ShakeResult): void {
|
||||
const elided = result.toolResultsDropped + result.blocksDropped;
|
||||
const sink = result.artifactId ? "an artifact" : "placeholders";
|
||||
this.emitNotice(
|
||||
"info",
|
||||
`Compaction dead-end recovery: elided ${elided} heavy block${elided === 1 ? "" : "s"} (~${result.tokensFreed.toLocaleString()} tokens) to ${sink} so maintenance could make progress.`,
|
||||
"compaction",
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Internal: Run auto-compaction with events.
|
||||
*
|
||||
@@ -11079,6 +11183,7 @@ export class AgentSession {
|
||||
const shouldAutoContinue =
|
||||
!suppressContinuation && options.autoContinue !== false && compactionSettings.autoContinue !== false;
|
||||
const suppressHandoff = options.suppressHandoff === true;
|
||||
let fallbackFromShake = false;
|
||||
// Shake runs inline (cheap, no remote LLM). On overflow recovery, if shake
|
||||
// reclaims nothing we fall through to the summary-compaction body below so
|
||||
// the oversized input still gets resolved.
|
||||
@@ -11092,6 +11197,7 @@ export class AgentSession {
|
||||
suppressContinuation,
|
||||
);
|
||||
if (outcome !== "fallback") return outcome;
|
||||
fallbackFromShake = true;
|
||||
}
|
||||
// "overflow" and "incomplete" force inline execution because they are recovery
|
||||
// paths the caller wants resolved before scheduling the next turn. "idle" is
|
||||
@@ -11320,6 +11426,17 @@ export class AgentSession {
|
||||
...(shapeSetting === "auto" ? {} : { shape }),
|
||||
maxFrames,
|
||||
});
|
||||
const framePayloadBytes = this.#snapcompactFramePayloadBytes(snapcompactResult);
|
||||
if (framePayloadBytes > snapcompact.FRAME_DATA_BYTES_BUDGET) {
|
||||
logger.warn("Snapcompact exceeded the per-request frame payload budget", {
|
||||
model: this.model?.id,
|
||||
framePayloadBytes,
|
||||
budget: snapcompact.FRAME_DATA_BYTES_BUDGET,
|
||||
});
|
||||
snapcompactBlocker =
|
||||
"snapcompact produced too much standing image payload; using context-full auto-compaction instead.";
|
||||
snapcompactResult = undefined;
|
||||
}
|
||||
if (snapcompactResult) {
|
||||
const ctxWindow = this.model?.contextWindow ?? 0;
|
||||
const budget =
|
||||
@@ -11579,7 +11696,15 @@ export class AgentSession {
|
||||
// won't include) is excluded. Reusing the auto-continue recovery band
|
||||
// here turned recoverable overflows into manual dead-ends (#3412 review),
|
||||
// so use the looser fit budget.
|
||||
if (this.#compactionCreatedRetryFit()) {
|
||||
let retryFits = this.#compactionCreatedRetryFit();
|
||||
if (!retryFits && !fallbackFromShake) {
|
||||
const rescue = await this.#tryShakeRescueForDeadEnd(autoCompactionSignal);
|
||||
if (rescue && this.#compactionCreatedRetryFit()) {
|
||||
retryFits = true;
|
||||
this.#emitShakeRescueNotice(rescue);
|
||||
}
|
||||
}
|
||||
if (retryFits) {
|
||||
this.#scheduleAgentContinue({ delayMs: 100, generation });
|
||||
continuationScheduled = true;
|
||||
} else {
|
||||
@@ -11593,7 +11718,15 @@ export class AgentSession {
|
||||
// when auto-continue is disabled, a no-headroom threshold pass must still
|
||||
// block later automatic continuations (todo reminders/session_stop hooks)
|
||||
// from re-entering the same oversized context.
|
||||
if (this.#compactionCreatedHeadroom()) {
|
||||
let hasHeadroom = this.#compactionCreatedHeadroom();
|
||||
if (!hasHeadroom && !fallbackFromShake) {
|
||||
const rescue = await this.#tryShakeRescueForDeadEnd(autoCompactionSignal);
|
||||
if (rescue && this.#compactionCreatedHeadroom()) {
|
||||
hasHeadroom = true;
|
||||
this.#emitShakeRescueNotice(rescue);
|
||||
}
|
||||
}
|
||||
if (hasHeadroom) {
|
||||
if (shouldAutoContinue) {
|
||||
this.#scheduleAutoContinuePrompt(generation);
|
||||
continuationScheduled = true;
|
||||
@@ -11617,7 +11750,7 @@ export class AgentSession {
|
||||
if (noProgressDeadEnd) {
|
||||
this.emitNotice(
|
||||
"warning",
|
||||
"Compaction freed too little context to make progress — pausing automatic maintenance to avoid a compaction loop. The most recent turn alone is too large to reduce further; shrink it (e.g. clear large tool output) or switch to a larger-context model.",
|
||||
"Compaction freed too little context to make progress — pausing automatic maintenance to avoid a compaction loop. The most recent turn alone is too large to reduce further; clear large tool output, run `/shake images` to drop attached images, or switch to a larger-context model.",
|
||||
"compaction",
|
||||
);
|
||||
}
|
||||
@@ -12453,6 +12586,11 @@ export class AgentSession {
|
||||
// Remove the failed assistant message from active context before retrying.
|
||||
this.#removeAssistantMessageFromActiveContext(message);
|
||||
|
||||
// A thinking/response loop retried into identical context loops again. Inject a
|
||||
// hidden redirect so the retried turn sees a directive to break the repeated
|
||||
// pattern instead of re-sampling the same stalled reasoning.
|
||||
this.#maybeInjectThinkingLoopRedirect(id);
|
||||
|
||||
// Wait with exponential backoff (abortable).
|
||||
const retryAbortController = new AbortController();
|
||||
this.#retryAbortController?.abort();
|
||||
@@ -12486,6 +12624,35 @@ export class AgentSession {
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Inject a hidden redirect notice when a thinking/response loop is being retried, so
|
||||
* the retried turn carries an instruction to break the repeated pattern instead of
|
||||
* re-sampling the same stalled context. Injected on every {@link AIError.Flag.ThinkingLoop}
|
||||
* retry (the failed assistant is dropped each attempt, so the notice does not accumulate
|
||||
* unboundedly). No-op unless `id` carries the ThinkingLoop flag and the loop guard is
|
||||
* enabled. The notice is generic on purpose — the detector's detail can quote raw model
|
||||
* text, which must not be interpolated into a higher-priority developer message.
|
||||
*/
|
||||
#maybeInjectThinkingLoopRedirect(id: number): void {
|
||||
if (!AIError.is(id, AIError.Flag.ThinkingLoop)) return;
|
||||
if (this.settings.get("model.loopGuard.enabled") !== true) return;
|
||||
this.agent.appendMessage({
|
||||
role: "custom",
|
||||
customType: THINKING_LOOP_REDIRECT_TYPE,
|
||||
content: thinkingLoopRedirectTemplate,
|
||||
display: false,
|
||||
attribution: "agent",
|
||||
timestamp: Date.now(),
|
||||
});
|
||||
this.sessionManager.appendCustomMessageEntry(
|
||||
THINKING_LOOP_REDIRECT_TYPE,
|
||||
thinkingLoopRedirectTemplate,
|
||||
false,
|
||||
undefined,
|
||||
"agent",
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel in-progress retry.
|
||||
*/
|
||||
@@ -12863,6 +13030,50 @@ export class AgentSession {
|
||||
// IRC Delivery
|
||||
// =========================================================================
|
||||
|
||||
/**
|
||||
* Surfaces (and consumes) IRC incoming asides that have reached this running
|
||||
* session but have not yet been folded into the next model step.
|
||||
*
|
||||
* The inbox tool injects the formatted body into the tool result, so the
|
||||
* model sees it once via the result. Leaving the record in
|
||||
* {@link #pendingIrcAsides} would let the aside provider deliver it a second
|
||||
* time at the next step boundary — including on `peek`, which is why peek
|
||||
* also drains here.
|
||||
*/
|
||||
drainPendingIrcInboxMessages(agentId: string): IrcMessage[] {
|
||||
const messages: IrcMessage[] = [];
|
||||
const remaining: CustomMessage[] = [];
|
||||
for (const record of this.#pendingIrcAsides) {
|
||||
if (record.customType !== "irc:incoming") {
|
||||
remaining.push(record);
|
||||
continue;
|
||||
}
|
||||
const details = record.details;
|
||||
if (!details || typeof details !== "object") {
|
||||
remaining.push(record);
|
||||
continue;
|
||||
}
|
||||
const id = Reflect.get(details, "id");
|
||||
const from = Reflect.get(details, "from");
|
||||
const body = Reflect.get(details, "message");
|
||||
const replyTo = Reflect.get(details, "replyTo");
|
||||
if (typeof id !== "string" || typeof from !== "string" || typeof body !== "string") {
|
||||
remaining.push(record);
|
||||
continue;
|
||||
}
|
||||
messages.push({
|
||||
id,
|
||||
from,
|
||||
to: agentId,
|
||||
body,
|
||||
ts: record.timestamp,
|
||||
...(typeof replyTo === "string" ? { replyTo } : {}),
|
||||
});
|
||||
}
|
||||
this.#pendingIrcAsides = remaining;
|
||||
return messages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver an IRC message into this session (recipient side; called by the
|
||||
* IrcBus). Emits the `irc_message` session event for UI cards and injects
|
||||
@@ -13174,7 +13385,15 @@ export class AgentSession {
|
||||
// Flush pending writes before switching so restore snapshots reflect committed state.
|
||||
await this.sessionManager.flush();
|
||||
const previousSessionState = this.sessionManager.captureState();
|
||||
const previousSessionContext = this.buildDisplaySessionContext();
|
||||
// Only same-session reloads compare against the prior context to detect
|
||||
// rollback edits (`#didSessionMessagesChange` below). Building it for a
|
||||
// different-session switch is a pure waste — and on huge pre-fix sessions
|
||||
// it materializes every persisted snapcompact frame plus the
|
||||
// `openaiRemoteCompaction.replacementHistory` payload into messages,
|
||||
// blowing the heap before the new session even loads (issue #3846). The
|
||||
// error-recovery path rebuilds the context on demand from the restored
|
||||
// state instead.
|
||||
const previousSessionContext = switchingToDifferentSession ? undefined : this.buildDisplaySessionContext();
|
||||
// switchSession replaces these arrays wholesale during load/rollback, so retaining
|
||||
// the existing message objects is sufficient and avoids structured-clone failures for
|
||||
// extension/custom metadata that is valid to persist but not cloneable.
|
||||
@@ -13187,7 +13406,7 @@ export class AgentSession {
|
||||
const previousThinkingLevel = this.#thinkingLevel;
|
||||
const previousAutoThinking = this.#autoThinking;
|
||||
const previousAutoResolvedLevel = this.#autoResolvedLevel;
|
||||
const previousServiceTier = this.agent.serviceTier;
|
||||
const previousServiceTierByFamily = this.#serviceTierByFamily;
|
||||
const previousSelectedMCPToolNames = new Set(this.#selectedMCPToolNames);
|
||||
const previousTools = [...this.agent.state.tools];
|
||||
const previousBaseSystemPrompt = this.#baseSystemPrompt;
|
||||
@@ -13213,7 +13432,7 @@ export class AgentSession {
|
||||
|
||||
const sessionContext = this.buildDisplaySessionContext();
|
||||
const didReloadConversationChange =
|
||||
!switchingToDifferentSession &&
|
||||
previousSessionContext !== undefined &&
|
||||
this.#didSessionMessagesChange(previousSessionContext.messages, sessionContext.messages);
|
||||
const fallbackSelectedMCPToolNames = this.#getSessionDefaultSelectedMCPToolNames(sessionPath);
|
||||
await this.#restoreMCPSelectionsForSessionContext(sessionContext, { fallbackSelectedMCPToolNames });
|
||||
@@ -13273,7 +13492,11 @@ export class AgentSession {
|
||||
.getBranch()
|
||||
.some(entry => entry.type === "service_tier_change");
|
||||
const defaultThinkingLevel = parseConfiguredThinkingLevel(this.settings.get("defaultThinkingLevel"));
|
||||
const configuredServiceTier = this.settings.get("serviceTier");
|
||||
const configuredServiceTierByFamily = buildServiceTierByFamily(
|
||||
this.settings.get("tier.openai"),
|
||||
this.settings.get("tier.anthropic"),
|
||||
this.settings.get("tier.google"),
|
||||
);
|
||||
// Restore the thinking selector. Each change persists the configured
|
||||
// selector (`auto` or a concrete level), so prefer it: an `auto` session
|
||||
// resumes in auto mode (reclassifying the next turn) instead of freezing at
|
||||
@@ -13302,11 +13525,9 @@ export class AgentSession {
|
||||
this.#thinkingLevel = resolveThinkingLevelForModel(this.model, restoredThinkingLevel);
|
||||
}
|
||||
this.#applyThinkingLevelToAgent(this.#thinkingLevel);
|
||||
this.agent.serviceTier = hasServiceTierEntry
|
||||
? sessionContext.serviceTier
|
||||
: configuredServiceTier === "none"
|
||||
? undefined
|
||||
: configuredServiceTier;
|
||||
this.#serviceTierByFamily = hasServiceTierEntry
|
||||
? (sessionContext.serviceTier ?? {})
|
||||
: configuredServiceTierByFamily;
|
||||
|
||||
if (switchingToDifferentSession) {
|
||||
await this.#resetMemoryContextForNewTranscript();
|
||||
@@ -13329,7 +13550,12 @@ export class AgentSession {
|
||||
this.#rekeyMnemopiMemoryForCurrentSessionId();
|
||||
let restoreMcpError: unknown;
|
||||
try {
|
||||
await this.#restoreMCPSelectionsForSessionContext(previousSessionContext, {
|
||||
// `previousSessionContext` was skipped on different-session switches to
|
||||
// avoid materializing the previous session's heavy compaction payload
|
||||
// in the success path; rebuild it here on demand from the restored
|
||||
// state so MCP selection restoration still has its inputs.
|
||||
const mcpRestoreContext = previousSessionContext ?? this.buildDisplaySessionContext();
|
||||
await this.#restoreMCPSelectionsForSessionContext(mcpRestoreContext, {
|
||||
fallbackSelectedMCPToolNames: previousFallbackSelectedMCPToolNames,
|
||||
});
|
||||
} catch (mcpError) {
|
||||
@@ -13358,7 +13584,7 @@ export class AgentSession {
|
||||
this.#autoThinking = previousAutoThinking;
|
||||
this.#autoResolvedLevel = previousAutoResolvedLevel;
|
||||
this.#applyThinkingLevelToAgent(previousThinkingLevel);
|
||||
this.agent.serviceTier = previousServiceTier;
|
||||
this.#serviceTierByFamily = previousServiceTierByFamily;
|
||||
this.#syncTodoPhasesFromBranch();
|
||||
this.#resetAllAdvisorRuntimes();
|
||||
this.#reconnectToAgent();
|
||||
@@ -14312,7 +14538,7 @@ export class AgentSession {
|
||||
const payload = {
|
||||
model: this.agent.state.model ?? null,
|
||||
thinkingLevel: this.#thinkingLevel ?? null,
|
||||
serviceTier: this.agent.serviceTier ?? null,
|
||||
serviceTier: this.#serviceTierEntry(),
|
||||
systemPrompt: this.agent.state.systemPrompt,
|
||||
tools: this.agent.state.tools.map(tool => ({
|
||||
name: tool.name,
|
||||
|
||||
@@ -488,7 +488,7 @@ export interface FileMentionMessage {
|
||||
/** File size in bytes, if known. */
|
||||
byteSize?: number;
|
||||
/** Why the file contents were omitted from auto-read. */
|
||||
skippedReason?: "tooLarge";
|
||||
skippedReason?: "tooLarge" | "binary";
|
||||
image?: ImageContent;
|
||||
}>;
|
||||
timestamp: number;
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ProviderPayload, ServiceTier } from "@oh-my-pi/pi-ai";
|
||||
import { coerceServiceTierByFamily, type ProviderPayload, type ServiceTierByFamily } from "@oh-my-pi/pi-ai";
|
||||
import * as snapcompact from "@oh-my-pi/snapcompact";
|
||||
import { createBranchSummaryMessage, createCompactionSummaryMessage, createCustomMessage } from "./messages";
|
||||
import { type CompactionEntry, EPHEMERAL_MODEL_CHANGE_ROLE, type SessionEntry } from "./session-entries";
|
||||
@@ -9,7 +9,7 @@ export interface SessionContext {
|
||||
thinkingLevel?: string;
|
||||
/** Configured thinking selector (`"auto"` or a concrete level) from the latest change. */
|
||||
configuredThinkingLevel?: string;
|
||||
serviceTier?: ServiceTier;
|
||||
serviceTier?: ServiceTierByFamily;
|
||||
/** Model roles: { default: "provider/modelId", small: "provider/modelId", ... } */
|
||||
models: Record<string, string>;
|
||||
/** Names of TTSR rules that have been injected this session */
|
||||
@@ -77,6 +77,17 @@ export interface BuildSessionContextOptions {
|
||||
* If leafId is provided, walks from that entry to root.
|
||||
* Handles compaction and branch summaries along the path.
|
||||
*/
|
||||
function snapcompactHistoryBlocksForContext(
|
||||
archive: snapcompact.Archive | undefined,
|
||||
options: BuildSessionContextOptions | undefined,
|
||||
) {
|
||||
if (!archive) return undefined;
|
||||
return snapcompact.historyBlocks(
|
||||
archive,
|
||||
options?.transcript ? undefined : { maxFrameDataBytes: snapcompact.FRAME_DATA_BYTES_BUDGET },
|
||||
);
|
||||
}
|
||||
|
||||
export function buildSessionContext(
|
||||
entries: SessionEntry[],
|
||||
leafId?: string | null,
|
||||
@@ -138,7 +149,7 @@ export function buildSessionContext(
|
||||
// Extract settings and find compaction
|
||||
let thinkingLevel: string | undefined = "off";
|
||||
let configuredThinkingLevel: string | undefined;
|
||||
let serviceTier: ServiceTier | undefined;
|
||||
let serviceTier: ServiceTierByFamily | undefined;
|
||||
const models: Record<string, string> = {};
|
||||
let compaction: CompactionEntry | null = null;
|
||||
const injectedTtsrRulesSet = new Set<string>();
|
||||
@@ -169,7 +180,7 @@ export function buildSessionContext(
|
||||
}
|
||||
}
|
||||
} else if (entry.type === "service_tier_change") {
|
||||
serviceTier = entry.serviceTier ?? undefined;
|
||||
serviceTier = coerceServiceTierByFamily(entry.serviceTier);
|
||||
} else if (entry.type === "message" && entry.message.role === "assistant") {
|
||||
// Legacy fallback: infer default model from assistant messages only
|
||||
// when no explicit `model_change` (role=default) entry has been
|
||||
@@ -273,7 +284,7 @@ export function buildSessionContext(
|
||||
entry.shortSummary,
|
||||
undefined,
|
||||
undefined,
|
||||
snapcompactArchive ? snapcompact.historyBlocks(snapcompactArchive) : undefined,
|
||||
snapcompactHistoryBlocksForContext(snapcompactArchive, options),
|
||||
),
|
||||
);
|
||||
} else {
|
||||
@@ -307,7 +318,7 @@ export function buildSessionContext(
|
||||
compaction.shortSummary,
|
||||
providerPayload,
|
||||
undefined,
|
||||
snapcompactArchive ? snapcompact.historyBlocks(snapcompactArchive) : undefined,
|
||||
snapcompactHistoryBlocksForContext(snapcompactArchive, options),
|
||||
),
|
||||
);
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ImageContent, MessageAttribution, ServiceTier, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import type { ImageContent, MessageAttribution, ServiceTierByFamily, TextContent } from "@oh-my-pi/pi-ai";
|
||||
|
||||
export const CURRENT_SESSION_VERSION = 3;
|
||||
|
||||
@@ -73,7 +73,7 @@ export interface ModelChangeEntry extends SessionEntryBase {
|
||||
|
||||
export interface ServiceTierChangeEntry extends SessionEntryBase {
|
||||
type: "service_tier_change";
|
||||
serviceTier: ServiceTier | null;
|
||||
serviceTier: ServiceTierByFamily | null;
|
||||
}
|
||||
|
||||
export interface CompactionEntry<T = unknown> extends SessionEntryBase {
|
||||
|
||||
@@ -1,6 +1,13 @@
|
||||
import * as fs from "node:fs";
|
||||
import * as path from "node:path";
|
||||
import type { ImageContent, Message, MessageAttribution, ServiceTier, TextContent, Usage } from "@oh-my-pi/pi-ai";
|
||||
import type {
|
||||
ImageContent,
|
||||
Message,
|
||||
MessageAttribution,
|
||||
ServiceTierByFamily,
|
||||
TextContent,
|
||||
Usage,
|
||||
} from "@oh-my-pi/pi-ai";
|
||||
import {
|
||||
directoryExists,
|
||||
getBlobsDir,
|
||||
@@ -1286,7 +1293,7 @@ export class SessionManager {
|
||||
return entry.id;
|
||||
}
|
||||
|
||||
appendServiceTierChange(serviceTier: ServiceTier | null): string {
|
||||
appendServiceTierChange(serviceTier: ServiceTierByFamily | null): string {
|
||||
const entry: ServiceTierChangeEntry = { type: "service_tier_change", ...this.#freshEntryFields(), serviceTier };
|
||||
this.#recordEntry(entry);
|
||||
return entry.id;
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
* Settings-aware stream wrapper shared by the main agent (sdk.ts) and the
|
||||
* advisor agent (AgentSession.#buildAdvisorRuntime).
|
||||
*
|
||||
* Reads OpenRouter / Antigravity routing variants, Responses-family text
|
||||
* verbosity, per-provider in-flight caps, and the loop guard out of `Settings`
|
||||
* verbosity, stream watchdog budgets, per-provider in-flight caps, and the loop
|
||||
* guard out of `Settings`
|
||||
* per request, layering them onto whatever options the caller passed. Before
|
||||
* this helper existed, advisor turns called bare `streamSimple` while the main
|
||||
* turn went through an inline closure that read these settings — so an advisor on
|
||||
@@ -14,6 +14,12 @@ import type { StreamFn } from "@oh-my-pi/pi-agent-core";
|
||||
import { type SimpleStreamOptions, streamSimple } from "@oh-my-pi/pi-ai";
|
||||
import { type Settings, validateProviderMaxInFlightRequests } from "../config/settings";
|
||||
|
||||
function timeoutSecondsToMs(value: number): number | undefined {
|
||||
if (!Number.isFinite(value) || value < 0) return undefined;
|
||||
if (value === 0) return 0;
|
||||
return Math.max(1, Math.trunc(value * 1000));
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a {@link StreamFn} that reads provider routing/guard settings from
|
||||
* `settings` per call and forwards to `base` (defaults to `streamSimple`).
|
||||
@@ -30,11 +36,15 @@ export function createSettingsAwareStreamFn(settings: Settings, base: StreamFn =
|
||||
model.api === "openai-codex-responses" || model.api === "openai-responses"
|
||||
? settings.get("textVerbosity")
|
||||
: undefined;
|
||||
const streamFirstEventTimeoutMs = timeoutSecondsToMs(settings.get("providers.streamFirstEventTimeoutSeconds"));
|
||||
const streamIdleTimeoutMs = timeoutSecondsToMs(settings.get("providers.streamIdleTimeoutSeconds"));
|
||||
const merged: SimpleStreamOptions = {
|
||||
...streamOptions,
|
||||
openrouterVariant: streamOptions?.openrouterVariant ?? openrouterVariant,
|
||||
antigravityEndpointMode: streamOptions?.antigravityEndpointMode ?? antigravityEndpointMode,
|
||||
textVerbosity: streamOptions?.textVerbosity ?? textVerbosity,
|
||||
streamFirstEventTimeoutMs: streamOptions?.streamFirstEventTimeoutMs ?? streamFirstEventTimeoutMs,
|
||||
streamIdleTimeoutMs: streamOptions?.streamIdleTimeoutMs ?? streamIdleTimeoutMs,
|
||||
maxInFlightRequests: validateProviderMaxInFlightRequests(
|
||||
streamOptions?.maxInFlightRequests ?? settings.get("providers.maxInFlightRequests"),
|
||||
),
|
||||
|
||||
@@ -73,17 +73,9 @@ function refreshStatusLine(ctx: InteractiveModeContext): void {
|
||||
ctx.ui.requestRender();
|
||||
}
|
||||
|
||||
/** `/fast status` label: "off", "on", or scope-qualified "on (… only)". */
|
||||
/** `/fast status` label for the active model: "on" when its family is priority, else "off". */
|
||||
function formatFastModeStatus(session: AgentSession): string {
|
||||
if (!session.isFastModeEnabled()) return "off";
|
||||
switch (session.serviceTier) {
|
||||
case "openai-only":
|
||||
return "on (OpenAI only)";
|
||||
case "claude-only":
|
||||
return "on (Claude only)";
|
||||
default:
|
||||
return "on";
|
||||
}
|
||||
return session.isFastModeEnabled() ? "on" : "off";
|
||||
}
|
||||
|
||||
const AUTOCOMPLETE_DETAIL_LIMIT = 48;
|
||||
|
||||
@@ -3,5 +3,6 @@ export * from "./asr-protocol";
|
||||
export * from "./downloader";
|
||||
export * from "./models";
|
||||
export * from "./stt-controller";
|
||||
export * from "./submit-trigger";
|
||||
export * from "./transcriber";
|
||||
export * from "./wav";
|
||||
|
||||
@@ -15,6 +15,7 @@ import {
|
||||
startStreamingRecording,
|
||||
verifyRecordingFile,
|
||||
} from "./recorder";
|
||||
import { evaluateSubmitTrigger } from "./submit-trigger";
|
||||
import { transcribe } from "./transcriber";
|
||||
|
||||
export type SttState = "idle" | "recording" | "transcribing";
|
||||
@@ -33,6 +34,8 @@ interface Editor {
|
||||
setVolatileText(text: string): void;
|
||||
clearVolatileText(): void;
|
||||
commitVolatileText(text: string): void;
|
||||
submit(): void;
|
||||
deleteBeforeCursor(count: number): void;
|
||||
}
|
||||
|
||||
export class STTController {
|
||||
@@ -53,6 +56,7 @@ export class STTController {
|
||||
#streamEditor: Editor | null = null;
|
||||
#streamCommitted = false;
|
||||
#streamAbort: AbortController | null = null;
|
||||
#streamUtterance = "";
|
||||
|
||||
get state(): SttState {
|
||||
return this.#state;
|
||||
@@ -190,6 +194,7 @@ export class STTController {
|
||||
const language = settings.get("stt.language") as string | undefined;
|
||||
this.#streamEditor = editor;
|
||||
this.#streamCommitted = false;
|
||||
this.#streamUtterance = "";
|
||||
this.#streamAbort = new AbortController();
|
||||
const stream = sttClient.startStream(modelKey, {
|
||||
language: language || undefined,
|
||||
@@ -205,6 +210,7 @@ export class STTController {
|
||||
if (prefixed) {
|
||||
this.#streamEditor?.commitVolatileText(prefixed);
|
||||
this.#streamCommitted = true;
|
||||
this.#streamUtterance += prefixed;
|
||||
} else {
|
||||
this.#streamEditor?.clearVolatileText();
|
||||
}
|
||||
@@ -266,13 +272,27 @@ export class STTController {
|
||||
return;
|
||||
}
|
||||
if (!this.#streamCommitted && finalText) {
|
||||
this.#streamEditor?.commitVolatileText(this.#prefixed(finalText));
|
||||
const prefixed = this.#prefixed(finalText);
|
||||
this.#streamEditor?.commitVolatileText(prefixed);
|
||||
this.#streamCommitted = true;
|
||||
this.#streamUtterance = prefixed;
|
||||
} else {
|
||||
this.#streamEditor?.clearVolatileText();
|
||||
}
|
||||
options.requestRender?.();
|
||||
if (!failed) options.showStatus(this.#streamCommitted ? "" : "No speech detected.");
|
||||
|
||||
if (this.#streamCommitted && !failed && this.#streamEditor) {
|
||||
const trigger = settings.get("stt.submitTrigger");
|
||||
const { submit, trimTrailing } = evaluateSubmitTrigger(this.#streamUtterance, trigger);
|
||||
if (trimTrailing > 0) {
|
||||
this.#streamEditor.deleteBeforeCursor(trimTrailing);
|
||||
}
|
||||
if (submit) {
|
||||
this.#streamEditor.submit();
|
||||
}
|
||||
}
|
||||
|
||||
this.#cleanupStream();
|
||||
this.#setState("idle", options);
|
||||
}
|
||||
@@ -283,6 +303,7 @@ export class STTController {
|
||||
this.#streamEditor = null;
|
||||
this.#streamCommitted = false;
|
||||
this.#streamAbort = null;
|
||||
this.#streamUtterance = "";
|
||||
}
|
||||
|
||||
// ── Batch (single-shot) ─────────────────────────────────────────
|
||||
@@ -327,8 +348,16 @@ export class STTController {
|
||||
this.#transcriptionAbort = null;
|
||||
if (this.#disposed) return;
|
||||
if (text.length > 0) {
|
||||
editor.insertText(text);
|
||||
const trigger = settings.get("stt.submitTrigger");
|
||||
const { submit, trimTrailing } = evaluateSubmitTrigger(text, trigger);
|
||||
const textToInsert = trimTrailing > 0 ? text.slice(0, -trimTrailing) : text;
|
||||
if (textToInsert.length > 0) {
|
||||
editor.insertText(textToInsert);
|
||||
}
|
||||
options.showStatus("");
|
||||
if (submit) {
|
||||
editor.submit();
|
||||
}
|
||||
} else {
|
||||
options.showStatus("No speech detected.");
|
||||
}
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
/**
|
||||
* TTS/STT Submit Trigger options and evaluation logic.
|
||||
*/
|
||||
|
||||
export const STT_SUBMIT_TRIGGER_VALUES = ["never", "release", "release-complete", "say-submit"] as const;
|
||||
|
||||
export type SttSubmitTrigger = (typeof STT_SUBMIT_TRIGGER_VALUES)[number];
|
||||
|
||||
export const STT_SUBMIT_TRIGGER_OPTIONS = [
|
||||
{
|
||||
value: "never",
|
||||
label: "Never",
|
||||
description: "Never automatically submit; insert dictation and remain in editor.",
|
||||
},
|
||||
{
|
||||
value: "release",
|
||||
label: "Release",
|
||||
description: "Submit on release if the utterance has 2+ words to avoid accidental sends.",
|
||||
},
|
||||
{
|
||||
value: "release-complete",
|
||||
label: "Release with complete sentence",
|
||||
description: "Submit on release if the utterance ends with sentence-terminal punctuation (. ? ! etc.).",
|
||||
},
|
||||
{
|
||||
value: "say-submit",
|
||||
label: "When I Say Submit",
|
||||
description: "Submit if the utterance ends with a word containing 'submit' (strips that word before submitting).",
|
||||
},
|
||||
] satisfies ReadonlyArray<{ value: SttSubmitTrigger; label: string; description: string }>;
|
||||
|
||||
/**
|
||||
* Evaluate the submit trigger against a transcribed utterance.
|
||||
* Returns whether to submit, and the number of characters to trim from the end of the utterance.
|
||||
*/
|
||||
export function evaluateSubmitTrigger(
|
||||
utterance: string,
|
||||
trigger: SttSubmitTrigger,
|
||||
): { submit: boolean; trimTrailing: number } {
|
||||
const trimmed = utterance.trim();
|
||||
if (!trimmed) {
|
||||
return { submit: false, trimTrailing: 0 };
|
||||
}
|
||||
|
||||
if (trigger === "never") {
|
||||
return { submit: false, trimTrailing: 0 };
|
||||
}
|
||||
|
||||
if (trigger === "release") {
|
||||
// Split by whitespace and count words
|
||||
const words = trimmed.split(/\s+/).filter(Boolean);
|
||||
const submit = words.length >= 2;
|
||||
return { submit, trimTrailing: 0 };
|
||||
}
|
||||
|
||||
if (trigger === "release-complete") {
|
||||
// Matches typical sentence terminators: . ? ! ... or full-width equivalents, optionally followed by space
|
||||
const hasTerminalPunctuation = /[.?!…。?!]\s*$/.test(trimmed);
|
||||
return { submit: hasTerminalPunctuation, trimTrailing: 0 };
|
||||
}
|
||||
|
||||
if (trigger === "say-submit") {
|
||||
// Matches space followed by any word containing "submit" (case-insensitive), optionally followed by punctuation/spaces
|
||||
// Also handles the case where "submit" is the only word in the utterance (no leading space)
|
||||
const match = utterance.match(/(?:^|\s+)(\S*submit\S*)[.?!…。?!]*\s*$/i);
|
||||
if (match && match.index !== undefined) {
|
||||
const trimTrailing = utterance.length - match.index;
|
||||
return { submit: true, trimTrailing };
|
||||
}
|
||||
return { submit: false, trimTrailing: 0 };
|
||||
}
|
||||
|
||||
return { submit: false, trimTrailing: 0 };
|
||||
}
|
||||
@@ -0,0 +1,158 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
|
||||
interface ProbeRunResult {
|
||||
elapsedMs: number;
|
||||
childElapsedMs: number;
|
||||
cached: unknown;
|
||||
count: number;
|
||||
}
|
||||
|
||||
async function runProbeScenario(options: {
|
||||
runs: number;
|
||||
sleepSeconds?: number;
|
||||
holdStdoutOpen?: boolean;
|
||||
descendantHoldsStdout?: boolean;
|
||||
validOutput?: string;
|
||||
}): Promise<ProbeRunResult> {
|
||||
const tempRoot = await fs.mkdtemp(path.join(os.tmpdir(), "omp-gpu-probe-"));
|
||||
try {
|
||||
const binDir = path.join(tempRoot, "bin");
|
||||
const cacheRoot = path.join(tempRoot, "cache");
|
||||
const probeCountPath = path.join(tempRoot, "probe-count");
|
||||
await fs.mkdir(binDir, { recursive: true });
|
||||
await fs.mkdir(path.join(cacheRoot, "omp"), { recursive: true });
|
||||
const lspciPath = path.join(binDir, "lspci");
|
||||
await Bun.write(
|
||||
lspciPath,
|
||||
'#!/usr/bin/env sh\nprintf x >> "$OMP_GPU_PROBE_COUNT"\nif [ -n "$OMP_GPU_PROBE_VALID_OUTPUT" ]; then printf "%s\\n" "$OMP_GPU_PROBE_VALID_OUTPUT"; fi\nif [ "$OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & exit 0; fi\nif [ "$OMP_GPU_PROBE_HOLD_STDOUT_OPEN" = "true" ]; then sleep "$OMP_GPU_PROBE_SLEEP" & wait "$!"; fi\nif [ -n "$OMP_GPU_PROBE_SLEEP" ]; then exec sleep "$OMP_GPU_PROBE_SLEEP"; fi\nexit 0\n',
|
||||
);
|
||||
await fs.chmod(lspciPath, 0o755);
|
||||
|
||||
const scenarioPath = path.join(tempRoot, "scenario.ts");
|
||||
await Bun.write(
|
||||
scenarioPath,
|
||||
`import { getGpuCachePath, refreshDirsFromEnv } from ${JSON.stringify(path.resolve(import.meta.dir, "../../utils/src/index.ts"))};
|
||||
import { buildSystemPrompt } from ${JSON.stringify(path.join(import.meta.dir, "system-prompt.ts"))};
|
||||
|
||||
refreshDirsFromEnv();
|
||||
const buildOptions = {
|
||||
contextFiles: [],
|
||||
skills: [],
|
||||
toolNames: [],
|
||||
workspaceTree: {
|
||||
rootPath: process.cwd(),
|
||||
rendered: "",
|
||||
truncated: false,
|
||||
totalLines: 0,
|
||||
agentsMdFiles: [],
|
||||
},
|
||||
activeRepoContext: null,
|
||||
};
|
||||
const startedAt = performance.now();
|
||||
for (let index = 0; index < Number(process.env.OMP_GPU_PROBE_RUNS ?? "1"); index += 1) {
|
||||
await buildSystemPrompt(buildOptions);
|
||||
}
|
||||
const cacheFile = Bun.file(getGpuCachePath());
|
||||
const cached = await cacheFile.exists() ? await cacheFile.json() : null;
|
||||
const countFile = Bun.file(process.env.OMP_GPU_PROBE_COUNT ?? "");
|
||||
const count = await countFile.exists() ? (await countFile.text()).length : 0;
|
||||
console.log(JSON.stringify({ elapsedMs: Math.round(performance.now() - startedAt), cached, count }));
|
||||
`,
|
||||
);
|
||||
|
||||
const env: Record<string, string | undefined> = {
|
||||
...process.env,
|
||||
PATH: `${binDir}:${process.env.PATH ?? ""}`,
|
||||
XDG_CACHE_HOME: cacheRoot,
|
||||
OMP_GPU_PROBE_COUNT: probeCountPath,
|
||||
OMP_GPU_PROBE_RUNS: String(options.runs),
|
||||
};
|
||||
// Strip inherited dirs-resolver overrides so XDG_CACHE_HOME above wins and
|
||||
// the test cannot touch the developer/CI profile's real gpu_cache.json.
|
||||
for (const key of ["PI_CODING_AGENT_DIR", "OMP_PROFILE", "PI_PROFILE", "PI_CONFIG_DIR"]) {
|
||||
delete env[key];
|
||||
}
|
||||
if (options.sleepSeconds === undefined) {
|
||||
delete env.OMP_GPU_PROBE_SLEEP;
|
||||
} else {
|
||||
env.OMP_GPU_PROBE_SLEEP = String(options.sleepSeconds);
|
||||
}
|
||||
if (options.holdStdoutOpen) {
|
||||
env.OMP_GPU_PROBE_HOLD_STDOUT_OPEN = "true";
|
||||
} else {
|
||||
delete env.OMP_GPU_PROBE_HOLD_STDOUT_OPEN;
|
||||
}
|
||||
if (options.descendantHoldsStdout) {
|
||||
env.OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT = "true";
|
||||
} else {
|
||||
delete env.OMP_GPU_PROBE_DESCENDANT_HOLDS_STDOUT;
|
||||
}
|
||||
if (options.validOutput !== undefined) {
|
||||
env.OMP_GPU_PROBE_VALID_OUTPUT = options.validOutput;
|
||||
} else {
|
||||
delete env.OMP_GPU_PROBE_VALID_OUTPUT;
|
||||
}
|
||||
|
||||
const childStartedAt = performance.now();
|
||||
const child = Bun.spawn([process.execPath, scenarioPath], { stdout: "pipe", stderr: "pipe", env });
|
||||
const [stdout, stderr, exitCode] = await Promise.all([
|
||||
new Response(child.stdout).text(),
|
||||
new Response(child.stderr).text(),
|
||||
child.exited,
|
||||
]);
|
||||
const childElapsedMs = Math.round(performance.now() - childStartedAt);
|
||||
if (exitCode !== 0) {
|
||||
throw new Error(`GPU probe scenario failed with exit ${exitCode}: ${stderr}`);
|
||||
}
|
||||
return { ...JSON.parse(stdout.trim()), childElapsedMs };
|
||||
} finally {
|
||||
await fs.rm(tempRoot, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
describe.skipIf(process.platform !== "linux")("system prompt GPU probe", () => {
|
||||
it("caches empty GPU probe results", async () => {
|
||||
const result = await runProbeScenario({ runs: 2 });
|
||||
|
||||
expect(result.cached).toEqual({ gpu: null });
|
||||
expect(result.count).toBe(1);
|
||||
}, 15_000);
|
||||
|
||||
it("kills the GPU probe at the prep deadline", async () => {
|
||||
const result = await runProbeScenario({ runs: 1, sleepSeconds: 7, holdStdoutOpen: true });
|
||||
|
||||
expect(result.cached).toEqual({ gpu: null });
|
||||
expect(result.elapsedMs).toBeLessThan(6500);
|
||||
// Codex#3838: the child process MUST exit shortly after the deadline,
|
||||
// not linger until a descendant holding stdout (sleep 7) exits on its own.
|
||||
expect(result.childElapsedMs).toBeLessThan(6500);
|
||||
}, 15_000);
|
||||
|
||||
it("does not wait on stdout held by a descendant after a successful probe", async () => {
|
||||
const result = await runProbeScenario({ runs: 1, sleepSeconds: 3, descendantHoldsStdout: true });
|
||||
|
||||
expect(result.cached).toEqual({ gpu: null });
|
||||
// Probe exits 0 immediately but leaves a backgrounded sleep holding the stdout
|
||||
// pipe. The success path MUST bound the drain wait, not block until sleep exits.
|
||||
expect(result.elapsedMs).toBeLessThan(2000);
|
||||
expect(result.childElapsedMs).toBeLessThan(2000);
|
||||
}, 15_000);
|
||||
|
||||
it("keeps probe output captured before a descendant delays EOF", async () => {
|
||||
const result = await runProbeScenario({
|
||||
runs: 1,
|
||||
sleepSeconds: 3,
|
||||
descendantHoldsStdout: true,
|
||||
validOutput: "00:02.0 VGA compatible controller: NVIDIA TestGPU",
|
||||
});
|
||||
|
||||
// Probe exited 0 with valid output before bg sleep held stdout open.
|
||||
// Captured stdout MUST be cached, not discarded as if the probe failed.
|
||||
expect(result.cached).toEqual({ gpu: "02.0 VGA compatible controller: NVIDIA TestGPU" });
|
||||
expect(result.elapsedMs).toBeLessThan(2000);
|
||||
expect(result.childElapsedMs).toBeLessThan(2000);
|
||||
}, 15_000);
|
||||
});
|
||||
@@ -7,7 +7,6 @@ import type { AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ToolExample, TSchema } from "@oh-my-pi/pi-ai";
|
||||
import { renderToolInventory } from "@oh-my-pi/pi-ai/dialect";
|
||||
import { $env, getGpuCachePath, getProjectDir, hasFsCode, isEnoent, logger, prompt } from "@oh-my-pi/pi-utils";
|
||||
import { $ } from "bun";
|
||||
import { contextFileCapability } from "./capability/context-file";
|
||||
import { systemPromptCapability } from "./capability/system-prompt";
|
||||
import { findConfigFile } from "./config";
|
||||
@@ -112,21 +111,60 @@ function parseWmicTable(output: string, header: string): string | null {
|
||||
}
|
||||
|
||||
const SYSTEM_PROMPT_PREP_TIMEOUT_MS = 5000;
|
||||
/** Kept below prep timeout so timed-out probes can still write the null cache before fallback. */
|
||||
const GPU_PROBE_TIMEOUT_MS = SYSTEM_PROMPT_PREP_TIMEOUT_MS - 500;
|
||||
/** Drop stdout from a probe descendant that inherited the pipe after the probe exited. */
|
||||
const GPU_PROBE_STDOUT_DRAIN_MS = 250;
|
||||
|
||||
async function runGpuProbe(cmd: string[]): Promise<string | null> {
|
||||
try {
|
||||
const proc = Bun.spawn({
|
||||
cmd,
|
||||
stdout: "pipe",
|
||||
stderr: "ignore",
|
||||
stdin: "ignore",
|
||||
timeout: GPU_PROBE_TIMEOUT_MS,
|
||||
// SIGKILL so a probe ignoring SIGTERM (PATH wrapper, wedged WMI) still
|
||||
// dies at the deadline and lets getCachedGpu reach the null-cache write.
|
||||
killSignal: "SIGKILL",
|
||||
});
|
||||
const stdoutReader = proc.stdout.getReader();
|
||||
let stdout = "";
|
||||
const decoder = new TextDecoder();
|
||||
const stdoutDone = (async () => {
|
||||
while (true) {
|
||||
const chunk = await stdoutReader.read();
|
||||
if (chunk.done) break;
|
||||
stdout += decoder.decode(chunk.value, { stream: true });
|
||||
}
|
||||
stdout += decoder.decode();
|
||||
})();
|
||||
const exitCode = await proc.exited;
|
||||
// Even on exit 0, a probe wrapper can leave a descendant holding stdout open.
|
||||
// Bound the EOF wait so getCachedGpu cannot outlive the probe in either path;
|
||||
// keep whatever bytes the reader already captured before cancelling.
|
||||
const drained = await Promise.race([
|
||||
stdoutDone.then(() => "ok" as const).catch(() => "err" as const),
|
||||
Bun.sleep(GPU_PROBE_STDOUT_DRAIN_MS).then(() => "timeout" as const),
|
||||
]);
|
||||
if (drained !== "ok") {
|
||||
await stdoutReader.cancel().catch(() => undefined);
|
||||
await stdoutDone.catch(() => undefined);
|
||||
}
|
||||
return exitCode === 0 ? stdout : null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function getGpuModel(): Promise<string | null> {
|
||||
switch (process.platform) {
|
||||
case "win32": {
|
||||
const output = await $`wmic path win32_VideoController get name`
|
||||
.quiet()
|
||||
.text()
|
||||
.catch(() => null);
|
||||
const output = await runGpuProbe(["wmic", "path", "win32_VideoController", "get", "name"]);
|
||||
return output ? parseWmicTable(output, "Name") : null;
|
||||
}
|
||||
case "linux": {
|
||||
const output = await $`lspci`
|
||||
.quiet()
|
||||
.text()
|
||||
.catch(() => null);
|
||||
const output = await runGpuProbe(["lspci"]);
|
||||
if (!output) return null;
|
||||
const gpus: Array<{ name: string; priority: number }> = [];
|
||||
for (const line of output.split("\n")) {
|
||||
@@ -176,20 +214,20 @@ function getTerminalName(): string | undefined {
|
||||
return term ?? undefined;
|
||||
}
|
||||
|
||||
/** Cached system info structure */
|
||||
/** Cached GPU probe result. */
|
||||
interface GpuCache {
|
||||
gpu: string;
|
||||
}
|
||||
|
||||
function getSystemInfoCachePath(): string {
|
||||
return getGpuCachePath();
|
||||
gpu: string | null;
|
||||
}
|
||||
|
||||
async function loadGpuCache(): Promise<GpuCache | null> {
|
||||
try {
|
||||
const cachePath = getSystemInfoCachePath();
|
||||
const cachePath = getGpuCachePath();
|
||||
const content = await Bun.file(cachePath).json();
|
||||
return content as GpuCache;
|
||||
if (content && typeof content === "object" && "gpu" in content) {
|
||||
const gpu = content.gpu;
|
||||
return { gpu: typeof gpu === "string" ? gpu : null };
|
||||
}
|
||||
return null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
@@ -197,7 +235,7 @@ async function loadGpuCache(): Promise<GpuCache | null> {
|
||||
|
||||
async function saveGpuCache(info: GpuCache): Promise<void> {
|
||||
try {
|
||||
const cachePath = getSystemInfoCachePath();
|
||||
const cachePath = getGpuCachePath();
|
||||
await Bun.write(cachePath, JSON.stringify(info, null, "\t"));
|
||||
} catch {
|
||||
// Silently ignore cache write failures
|
||||
@@ -206,15 +244,12 @@ async function saveGpuCache(info: GpuCache): Promise<void> {
|
||||
|
||||
async function getCachedGpu(): Promise<string | undefined> {
|
||||
const cached = await logger.time("getCachedGpu:loadGpuCache", loadGpuCache);
|
||||
if (cached) return cached.gpu;
|
||||
if (cached) return cached.gpu ?? undefined;
|
||||
const gpu = await logger.time("getCachedGpu:getGpuModel", getGpuModel);
|
||||
if (gpu) {
|
||||
await logger.time("getCachedGpu:saveGpuCache", saveGpuCache, { gpu });
|
||||
}
|
||||
await logger.time("getCachedGpu:saveGpuCache", saveGpuCache, { gpu });
|
||||
return gpu ?? undefined;
|
||||
}
|
||||
async function getEnvironmentInfo(): Promise<Array<{ label: string; value: string }>> {
|
||||
const gpu = await getCachedGpu();
|
||||
function getEnvironmentInfo(gpu: string | undefined): Array<{ label: string; value: string }> {
|
||||
let cpuModel: string | undefined;
|
||||
try {
|
||||
cpuModel = os.cpus()[0]?.model;
|
||||
@@ -500,9 +535,13 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}):
|
||||
agentsMdFiles: [],
|
||||
} satisfies WorkspaceTree,
|
||||
activeRepoContext: null as ActiveRepoContext | null,
|
||||
gpu: undefined as string | undefined,
|
||||
};
|
||||
|
||||
const deadline = Bun.sleep(SYSTEM_PROMPT_PREP_TIMEOUT_MS).then(() => "__timeout__" as const);
|
||||
const { promise: deadline, resolve: fireDeadline } = Promise.withResolvers<"__timeout__">();
|
||||
const deadlineTimer = setTimeout(() => fireDeadline("__timeout__"), SYSTEM_PROMPT_PREP_TIMEOUT_MS);
|
||||
// Unref so a fast prep does not hold a one-shot CLI alive waiting for this timer.
|
||||
deadlineTimer.unref();
|
||||
const timedOut: string[] = [];
|
||||
const failed: Array<{ name: string; error: unknown }> = [];
|
||||
|
||||
@@ -566,6 +605,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}):
|
||||
providedActiveRepoContext !== undefined
|
||||
? Promise.resolve(providedActiveRepoContext)
|
||||
: logger.time("resolveActiveRepoContext", () => resolveActiveRepoContext(resolvedCwd));
|
||||
const gpuPromise = logger.time("getCachedGpu", getCachedGpu);
|
||||
|
||||
const [
|
||||
resolvedCustomPrompt,
|
||||
@@ -575,6 +615,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}):
|
||||
skills,
|
||||
workspaceTree,
|
||||
activeRepoContext,
|
||||
gpu,
|
||||
] = await Promise.all([
|
||||
withDeadline(
|
||||
"customPrompt",
|
||||
@@ -597,7 +638,9 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}):
|
||||
withDeadline("loadSkills", skillsPromise, prepDefaults.skills),
|
||||
withDeadline("buildWorkspaceTree", workspaceTreePromise, prepDefaults.workspaceTree),
|
||||
withDeadline("resolveActiveRepoContext", activeRepoContextPromise, prepDefaults.activeRepoContext),
|
||||
withDeadline("getCachedGpu", gpuPromise, prepDefaults.gpu),
|
||||
]);
|
||||
clearTimeout(deadlineTimer);
|
||||
const agentsMdFiles = Array.from(new Set(workspaceTree.agentsMdFiles)).sort().slice(0, AGENTS_MD_LIMIT);
|
||||
|
||||
if (timedOut.length > 0) {
|
||||
@@ -675,7 +718,7 @@ export async function buildSystemPrompt(options: BuildSystemPromptOptions = {}):
|
||||
];
|
||||
const injectedAlwaysApplyRules = dedupeAlwaysApplyRules(alwaysApplyRules, promptSources);
|
||||
|
||||
const environment = await logger.time("getEnvironmentInfo", getEnvironmentInfo);
|
||||
const environment = getEnvironmentInfo(gpu);
|
||||
const data = {
|
||||
systemPromptCustomization: effectiveSystemPromptCustomization,
|
||||
customPrompt: resolvedCustomPrompt,
|
||||
|
||||
@@ -11,11 +11,11 @@ import exploreMd from "../prompts/agents/explore.md" with { type: "text" };
|
||||
// Embed agent markdown files at build time
|
||||
import agentFrontmatterTemplate from "../prompts/agents/frontmatter.md" with { type: "text" };
|
||||
import librarianMd from "../prompts/agents/librarian.md" with { type: "text" };
|
||||
import oracleMd from "../prompts/agents/oracle.md" with { type: "text" };
|
||||
|
||||
import planMd from "../prompts/agents/plan.md" with { type: "text" };
|
||||
import reviewerMd from "../prompts/agents/reviewer.md" with { type: "text" };
|
||||
import taskMd from "../prompts/agents/task.md" with { type: "text" };
|
||||
import testerMd from "../prompts/agents/tester.md" with { type: "text" };
|
||||
|
||||
import type { AgentDefinition, AgentSource } from "./types";
|
||||
|
||||
@@ -47,7 +47,7 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [
|
||||
{ fileName: "designer.md", template: designerMd },
|
||||
{ fileName: "reviewer.md", template: reviewerMd },
|
||||
{ fileName: "librarian.md", template: librarianMd },
|
||||
{ fileName: "oracle.md", template: oracleMd },
|
||||
{ fileName: "tester.md", template: testerMd },
|
||||
{
|
||||
fileName: "task.md",
|
||||
frontmatter: {
|
||||
@@ -59,9 +59,9 @@ const EMBEDDED_AGENT_DEFS: EmbeddedAgentDef[] = [
|
||||
template: taskMd,
|
||||
},
|
||||
{
|
||||
fileName: "quick_task.md",
|
||||
fileName: "sonic.md",
|
||||
frontmatter: {
|
||||
name: "quick_task",
|
||||
name: "sonic",
|
||||
description: "Low-reasoning agent for strictly mechanical updates or data collection only",
|
||||
model: "pi/smol",
|
||||
thinkingLevel: Effort.Medium,
|
||||
|
||||
@@ -7,7 +7,7 @@
|
||||
import path from "node:path";
|
||||
import type { AgentEvent, AgentIdentity, AgentTelemetryConfig, ThinkingLevel } from "@oh-my-pi/pi-agent-core";
|
||||
import { recordHandoff, resolveTelemetry } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Api, Model, ServiceTier, Usage } from "@oh-my-pi/pi-ai";
|
||||
import type { Api, Model, ServiceTierByFamily, Usage } from "@oh-my-pi/pi-ai";
|
||||
import { logger, popLoopPhase, prompt, pushLoopPhase, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import type { Rule } from "../capability/rule";
|
||||
import { ModelRegistry } from "../config/model-registry";
|
||||
@@ -18,7 +18,7 @@ import {
|
||||
resolveModelOverrideWithAuthFallback,
|
||||
} from "../config/model-resolver";
|
||||
import type { PromptTemplate } from "../config/prompt-templates";
|
||||
import { resolveSubagentServiceTier } from "../config/service-tier";
|
||||
import { buildServiceTierByFamily, resolveSubagentServiceTier } from "../config/service-tier";
|
||||
import { Settings } from "../config/settings";
|
||||
import { SETTINGS_SCHEMA, type SettingPath } from "../config/settings-schema";
|
||||
import type { ToolPathWithSource } from "../extensibility/custom-tools";
|
||||
@@ -87,7 +87,7 @@ const MCP_CALL_TIMEOUT_MS = 60_000;
|
||||
*/
|
||||
export const SOFT_REQUEST_BUDGET: Record<string, number> = {
|
||||
explore: 40,
|
||||
quick_task: 40,
|
||||
sonic: 40,
|
||||
default: 90,
|
||||
};
|
||||
|
||||
@@ -344,12 +344,12 @@ export interface ExecutorOptions {
|
||||
modelRegistry?: ModelRegistry;
|
||||
settings?: Settings;
|
||||
/**
|
||||
* Parent session's live effective service tier, the source of truth for a
|
||||
* subagent whose `serviceTierSubagent` is `"inherit"`. `null` = the parent
|
||||
* Parent session's live per-family service tiers, the source of truth for a
|
||||
* subagent whose `tier.subagent` is `"inherit"`. `null` = the parent
|
||||
* explicitly has no tier (e.g. `/fast off`); omitted = no live session, so
|
||||
* inherit falls back to the configured `serviceTier` setting.
|
||||
* inherit falls back to the subagent's configured `tier.*` settings.
|
||||
*/
|
||||
parentServiceTier?: ServiceTier | null;
|
||||
parentServiceTier?: ServiceTierByFamily | null;
|
||||
/** Override local:// protocol options so subagent shares parent's local:// root */
|
||||
localProtocolOptions?: LocalProtocolOptions;
|
||||
/**
|
||||
@@ -739,21 +739,28 @@ export function createMCPProxyTools(mcpManager: MCPManager): CustomTool[] {
|
||||
export function createSubagentSettings(
|
||||
baseSettings: Settings,
|
||||
overrides?: Partial<Record<SettingPath, unknown>>,
|
||||
inheritedServiceTier?: ServiceTier | null,
|
||||
inheritedServiceTier?: ServiceTierByFamily | null,
|
||||
): Settings {
|
||||
const snapshot: Partial<Record<SettingPath, unknown>> = {};
|
||||
for (const key of Object.keys(SETTINGS_SCHEMA) as SettingPath[]) {
|
||||
snapshot[key] = baseSettings.get(key);
|
||||
}
|
||||
// Resolve the subagent's service tier from `serviceTierSubagent` ("inherit" =
|
||||
// match the parent's live tier when a live session supplied one, else the
|
||||
// configured `serviceTier`). The result is stamped back onto the snapshot so
|
||||
// createAgentSession's `settings.get("serviceTier")` read picks it up.
|
||||
snapshot.serviceTier = resolveSubagentServiceTier(
|
||||
baseSettings.get("serviceTierSubagent"),
|
||||
baseSettings.get("serviceTier"),
|
||||
inheritedServiceTier,
|
||||
);
|
||||
// Resolve the subagent's per-family tiers from `tier.subagent` ("inherit" =
|
||||
// match the parent's live tiers when a live session supplied them, else the
|
||||
// subagent's own configured tier.* settings). The result is stamped back onto
|
||||
// the snapshot so createAgentSession's tier.* reads pick it up.
|
||||
const inheritedTiers =
|
||||
inheritedServiceTier === undefined
|
||||
? buildServiceTierByFamily(
|
||||
baseSettings.get("tier.openai"),
|
||||
baseSettings.get("tier.anthropic"),
|
||||
baseSettings.get("tier.google"),
|
||||
)
|
||||
: (inheritedServiceTier ?? {});
|
||||
const subagentTiers = resolveSubagentServiceTier(baseSettings.get("tier.subagent"), inheritedTiers);
|
||||
snapshot["tier.openai"] = subagentTiers.openai ?? "none";
|
||||
snapshot["tier.anthropic"] = subagentTiers.anthropic ?? "none";
|
||||
snapshot["tier.google"] = subagentTiers.google ?? "none";
|
||||
return Settings.isolated({
|
||||
...snapshot,
|
||||
"async.enabled": false,
|
||||
|
||||
@@ -243,8 +243,11 @@ function validateShapeParams(batchEnabled: boolean, params: TaskParams): string
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate the spawn parameter contract against the wire shapes. `agent` is
|
||||
* always required. With `task.batch` the model-facing shape is
|
||||
* Validate the spawn parameter contract against the wire shapes. `agent`
|
||||
* defaults to `task` (the schema default; `execute` normalizes the same way for
|
||||
* direct callers), so the missing-`agent` guard only fires for callers that
|
||||
* invoke this validator with an unnormalized blank agent. With `task.batch` the
|
||||
* model-facing shape is
|
||||
* `{ agent, context, tasks[] }` — `tasks` non-empty with per-item assignments
|
||||
* and unique ids, `context` non-empty, no top-level `assignment` alongside.
|
||||
* The flat `{ agent, ...item }` form stays accepted at runtime under either
|
||||
@@ -328,14 +331,17 @@ function spawnParamsFor(params: TaskParams, item: TaskItem): TaskParams {
|
||||
return spawn;
|
||||
}
|
||||
|
||||
/** Agent type spawned when a `task` call omits `agent`; mirrors the schema default in `getTaskSchema`. */
|
||||
const DEFAULT_TASK_AGENT = "task";
|
||||
|
||||
/** Generic worker agents whose output sharpens with a tailored `role` rather than the bare type. */
|
||||
const GENERIC_SPAWN_AGENTS: ReadonlySet<string> = new Set(["task", "quick_task"]);
|
||||
const GENERIC_SPAWN_AGENTS: ReadonlySet<string> = new Set(["task", "sonic"]);
|
||||
|
||||
/**
|
||||
* Advisory — never a rejection — nudging the spawner toward tailored
|
||||
* specialists when it spawns generic role-less workers and still holds spawn
|
||||
* capacity (DepthCapacity: it currently has the `task` tool). Fires when a
|
||||
* generic `task`/`quick_task` spawn carries no `role`, or when one call clones
|
||||
* generic `task`/`sonic` spawn carries no `role`, or when one call clones
|
||||
* the same agent ≥2× all without roles. Returns undefined when no nudge applies.
|
||||
*/
|
||||
export function buildSpecializationAdvisory(
|
||||
@@ -559,7 +565,14 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
|
||||
signal?: AbortSignal,
|
||||
onUpdate?: AgentToolUpdateCallback<TaskToolDetails>,
|
||||
): Promise<AgentToolResult<TaskToolDetails>> {
|
||||
const params = repairTaskParams(rawParams as TaskParams);
|
||||
const repaired = repairTaskParams(rawParams as TaskParams);
|
||||
// The schema defaults `agent` to `task` for model calls, but internal
|
||||
// callers and stale transcripts build params directly and bypass arktype.
|
||||
// Normalize once here so every downstream path sees the resolved agent.
|
||||
const params =
|
||||
typeof repaired.agent === "string" && repaired.agent.trim() !== ""
|
||||
? repaired
|
||||
: { ...repaired, agent: DEFAULT_TASK_AGENT };
|
||||
const batchEnabled = this.#isBatchEnabled();
|
||||
const validationError = validateShapeParams(batchEnabled, params) ?? validateSpawnParams(params, batchEnabled);
|
||||
if (validationError) {
|
||||
@@ -1296,11 +1309,13 @@ export class TaskTool implements AgentTool<TaskToolSchemaInstance, TaskToolDetai
|
||||
parentTelemetry: this.session.getTelemetry?.(),
|
||||
parentEvalSessionId,
|
||||
parentAgentId: this.session.getAgentId?.() ?? MAIN_AGENT_ID,
|
||||
// Live source of truth for `serviceTierSubagent: inherit`. When the
|
||||
// session exposes a tier accessor, pass tier-or-null (null = explicit
|
||||
// none, e.g. /fast off); otherwise leave undefined so inherit falls
|
||||
// back to the configured serviceTier setting.
|
||||
parentServiceTier: this.session.getServiceTier ? (this.session.getServiceTier() ?? null) : undefined,
|
||||
// Live source of truth for `tier.subagent: inherit`. When the session
|
||||
// exposes a tier accessor, pass the per-family map or null (null =
|
||||
// explicit none, e.g. /fast off); otherwise leave undefined so inherit
|
||||
// falls back to the subagent's configured tier.* settings.
|
||||
parentServiceTier: this.session.getServiceTierByFamily
|
||||
? (this.session.getServiceTierByFamily() ?? null)
|
||||
: undefined,
|
||||
};
|
||||
|
||||
const runTask = async (): Promise<SingleResult> => {
|
||||
|
||||
@@ -154,6 +154,7 @@ export async function runIsolatedSubprocess(opts: IsolatedRunOptions): Promise<S
|
||||
return {
|
||||
...result,
|
||||
branchName: commitResult?.branchName,
|
||||
branchBaseSha: commitResult?.baseSha,
|
||||
nestedPatches: commitResult?.nestedPatches,
|
||||
};
|
||||
} catch (mergeErr) {
|
||||
@@ -222,6 +223,14 @@ export async function mergeIsolatedChanges(opts: IsolationMergeOptions): Promise
|
||||
const { result, repoRoot, mergeMode } = opts;
|
||||
try {
|
||||
if (mergeMode === "branch") {
|
||||
if (!result.branchName && result.exitCode === 0 && !result.aborted && result.error) {
|
||||
return {
|
||||
summary: `\n\n<system-notification>Branch merge failed before a task branch could be created: ${result.error}\nTask outputs are preserved but changes were not applied.</system-notification>`,
|
||||
changesApplied: false,
|
||||
hadAnyChanges: false,
|
||||
mergedBranchForNestedPatches: false,
|
||||
};
|
||||
}
|
||||
const canApplyNestedOnly =
|
||||
!result.branchName && result.exitCode === 0 && !result.aborted && (result.nestedPatches?.length ?? 0) > 0;
|
||||
if (!result.branchName || result.exitCode !== 0 || result.aborted) {
|
||||
@@ -235,7 +244,12 @@ export async function mergeIsolatedChanges(opts: IsolationMergeOptions): Promise
|
||||
};
|
||||
}
|
||||
const mergeResult = await mergeTaskBranches(repoRoot, [
|
||||
{ branchName: result.branchName, taskId: result.id, description: result.description },
|
||||
{
|
||||
branchName: result.branchName,
|
||||
taskId: result.id,
|
||||
description: result.description,
|
||||
baseSha: result.branchBaseSha,
|
||||
},
|
||||
]);
|
||||
const mergedBranchForNestedPatches = mergeResult.merged.includes(result.branchName);
|
||||
const changesApplied = mergeResult.failed.length === 0;
|
||||
|
||||
@@ -111,7 +111,7 @@ export interface TaskItem {
|
||||
}
|
||||
|
||||
export const taskSchema = type({
|
||||
agent: "string",
|
||||
agent: "string = 'task'",
|
||||
"id?": "string",
|
||||
"description?": "string",
|
||||
"role?": ROLE_INPUT_SCHEMA,
|
||||
@@ -120,7 +120,7 @@ export const taskSchema = type({
|
||||
"+": "delete",
|
||||
});
|
||||
const taskSchemaNoIsolation = type({
|
||||
agent: "string",
|
||||
agent: "string = 'task'",
|
||||
"id?": "string",
|
||||
"description?": "string",
|
||||
"role?": ROLE_INPUT_SCHEMA,
|
||||
@@ -128,13 +128,13 @@ const taskSchemaNoIsolation = type({
|
||||
"+": "delete",
|
||||
});
|
||||
const taskSchemaBatch = type({
|
||||
agent: "string",
|
||||
agent: "string = 'task'",
|
||||
context: "string",
|
||||
tasks: taskItemSchemaIsolated.array(),
|
||||
"+": "delete",
|
||||
});
|
||||
const taskSchemaBatchNoIsolation = type({
|
||||
agent: "string",
|
||||
agent: "string = 'task'",
|
||||
context: "string",
|
||||
tasks: taskItemSchema.array(),
|
||||
"+": "delete",
|
||||
@@ -160,7 +160,7 @@ export function getTaskSchema(options: { isolationEnabled: boolean; batchEnabled
|
||||
* transcripts using the flat form keep working under either setting.
|
||||
*/
|
||||
export interface TaskParams {
|
||||
/** Agent type; required. */
|
||||
/** Agent type to spawn; defaults to `"task"` (the general-purpose worker) when omitted. */
|
||||
agent?: string;
|
||||
/** Stable agent id (flat form); default = generated AdjectiveNoun. */
|
||||
id?: string;
|
||||
@@ -383,6 +383,12 @@ export interface SingleResult {
|
||||
patchPath?: string;
|
||||
/** Branch name for isolated branch-mode output */
|
||||
branchName?: string;
|
||||
/**
|
||||
* Baseline commit SHA the task branch was created from. Passed to
|
||||
* `mergeTaskBranches` so cherry-pick uses the inclusive range
|
||||
* `branchBaseSha..branchName` and preserves every agent commit's message.
|
||||
*/
|
||||
branchBaseSha?: string;
|
||||
/** Nested repo patches to apply after parent merge */
|
||||
nestedPatches?: NestedRepoPatch[];
|
||||
/** Data extracted by registered subprocess tool handlers (keyed by tool name) */
|
||||
|
||||
@@ -148,21 +148,33 @@ export async function captureBaseline(repoRoot: string): Promise<WorktreeBaselin
|
||||
return { root, nested };
|
||||
}
|
||||
|
||||
async function captureRepoDeltaPatch(repoDir: string, rb: RepoBaseline): Promise<string> {
|
||||
async function captureRepoDeltaPatch(repoDir: string, rb: RepoBaseline, objectRepoDir = repoDir): Promise<string> {
|
||||
const currentHead = (await git.head.sha(repoDir)) ?? "";
|
||||
const currentStaged = await git.diff(repoDir, { binary: true, cached: true });
|
||||
const currentUnstaged = await git.diff(repoDir, { binary: true });
|
||||
const currentUntracked = await git.ls.untracked(repoDir);
|
||||
const currentUntrackedPatch = await captureUntrackedPatch(repoDir, currentUntracked);
|
||||
const committedPatch =
|
||||
currentHead && currentHead !== rb.headCommit
|
||||
? await git.diff.tree(repoDir, rb.headCommit, currentHead, {
|
||||
allowFailure: true,
|
||||
binary: true,
|
||||
})
|
||||
: "";
|
||||
|
||||
const baselineTree = await writeSyntheticTree(repoDir, rb.headCommit, [rb.staged, rb.unstaged, rb.untrackedPatch]);
|
||||
const currentTree = await writeSyntheticTree(repoDir, currentHead, [
|
||||
const baselineTree = await writeSyntheticTree(objectRepoDir, rb.headCommit, [
|
||||
rb.staged,
|
||||
rb.unstaged,
|
||||
rb.untrackedPatch,
|
||||
]);
|
||||
const currentTree = await writeSyntheticTree(objectRepoDir, rb.headCommit, [
|
||||
committedPatch,
|
||||
currentStaged,
|
||||
currentUnstaged,
|
||||
currentUntrackedPatch,
|
||||
]);
|
||||
|
||||
return git.diff.tree(repoDir, baselineTree, currentTree, {
|
||||
return git.diff.tree(objectRepoDir, baselineTree, currentTree, {
|
||||
allowFailure: true,
|
||||
binary: true,
|
||||
});
|
||||
@@ -212,7 +224,7 @@ export interface DeltaPatchResult {
|
||||
}
|
||||
|
||||
export async function captureDeltaPatch(isolationDir: string, baseline: WorktreeBaseline): Promise<DeltaPatchResult> {
|
||||
const rootPatch = await captureRepoDeltaPatch(isolationDir, baseline.root);
|
||||
const rootPatch = await captureRepoDeltaPatch(isolationDir, baseline.root, baseline.root.repoRoot);
|
||||
const nestedPatches: NestedRepoPatch[] = [];
|
||||
|
||||
for (const { relativePath, baseline: nb } of baseline.nested) {
|
||||
@@ -222,7 +234,7 @@ export async function captureDeltaPatch(isolationDir: string, baseline: Worktree
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
const patch = await captureRepoDeltaPatch(nestedDir, nb);
|
||||
const patch = await captureRepoDeltaPatch(nestedDir, nb, nb.repoRoot);
|
||||
if (patch.trim()) nestedPatches.push({ relativePath, patch });
|
||||
}
|
||||
|
||||
@@ -460,12 +472,137 @@ export async function cleanupIsolation(handle: IsolationHandle): Promise<void> {
|
||||
export interface CommitToBranchResult {
|
||||
branchName?: string;
|
||||
nestedPatches: NestedRepoPatch[];
|
||||
/**
|
||||
* SHA of the parent-repo commit the task branch was created on top of, so
|
||||
* {@link mergeTaskBranches} can cherry-pick the range `baseSha..branchName`
|
||||
* and preserve every agent commit's message and author.
|
||||
*/
|
||||
baseSha?: string;
|
||||
}
|
||||
|
||||
function baselineHasRootWip(baseline: RepoBaseline): boolean {
|
||||
return !!(baseline.staged.trim() || baseline.unstaged.trim() || baseline.untrackedPatch.trim());
|
||||
}
|
||||
|
||||
async function commitPatchToBranchWorktree(
|
||||
tmpDir: string,
|
||||
taskId: string,
|
||||
patchText: string,
|
||||
message: string,
|
||||
author?: git.CommitAuthor,
|
||||
): Promise<void> {
|
||||
try {
|
||||
await git.patch.applyText(tmpDir, patchText);
|
||||
} catch (err) {
|
||||
if (!(err instanceof git.GitCommandError)) throw err;
|
||||
// Plain apply rejects when the parent checkout carries unrelated dirty
|
||||
// context near the task's edits; retry with --3way before giving up.
|
||||
try {
|
||||
await git.patch.applyText(tmpDir, patchText, { threeWay: true });
|
||||
} catch (threeWayErr) {
|
||||
if (threeWayErr instanceof git.GitCommandError) {
|
||||
const stderr = threeWayErr.result.stderr.slice(0, 2000);
|
||||
logger.error("commitToBranch: git apply --3way failed", {
|
||||
taskId,
|
||||
exitCode: threeWayErr.result.exitCode,
|
||||
stderr,
|
||||
initialStderr: err.result.stderr.slice(0, 2000),
|
||||
patchSize: patchText.length,
|
||||
patchHead: patchText.slice(0, 500),
|
||||
});
|
||||
throw new Error(`git apply --3way failed for task ${taskId}: ${stderr}`);
|
||||
}
|
||||
throw threeWayErr;
|
||||
}
|
||||
}
|
||||
await git.stage.files(tmpDir);
|
||||
await git.commit(tmpDir, message, author ? { author } : {});
|
||||
}
|
||||
|
||||
interface FilteredAgentReplayOptions {
|
||||
baseline: WorktreeBaseline;
|
||||
branchName: string;
|
||||
commitMessage?: (diff: string) => Promise<string | null>;
|
||||
fallbackMessage: string;
|
||||
isolationDir: string;
|
||||
isolationHead: string;
|
||||
repoRoot: string;
|
||||
rootPatch: string;
|
||||
taskId: string;
|
||||
}
|
||||
|
||||
async function replayFilteredAgentCommits(opts: FilteredAgentReplayOptions): Promise<void> {
|
||||
const baselineSha = opts.baseline.root.headCommit;
|
||||
await git.branch.create(opts.repoRoot, opts.branchName, baselineSha);
|
||||
|
||||
const tmpDir = path.join(os.tmpdir(), `omp-branch-${Snowflake.next()}`);
|
||||
try {
|
||||
await git.worktree.add(opts.repoRoot, tmpDir, opts.branchName);
|
||||
const agentCommits = await git.revList.range(opts.isolationDir, baselineSha, opts.isolationHead);
|
||||
const dirtyBaselineTree = await writeSyntheticTree(opts.isolationDir, baselineSha, [
|
||||
opts.baseline.root.staged,
|
||||
opts.baseline.root.unstaged,
|
||||
opts.baseline.root.untrackedPatch,
|
||||
]);
|
||||
let previousFilteredTree = baselineSha;
|
||||
|
||||
for (const commitSha of agentCommits) {
|
||||
const taskStatePatch = await git.diff.tree(opts.isolationDir, dirtyBaselineTree, `${commitSha}^{tree}`, {
|
||||
allowFailure: true,
|
||||
binary: true,
|
||||
});
|
||||
const currentFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [taskStatePatch]);
|
||||
const commitPatch = await git.diff.tree(opts.repoRoot, previousFilteredTree, currentFilteredTree, {
|
||||
allowFailure: true,
|
||||
binary: true,
|
||||
});
|
||||
if (commitPatch.trim()) {
|
||||
const details = await git.commitDetails(opts.isolationDir, commitSha);
|
||||
await commitPatchToBranchWorktree(
|
||||
tmpDir,
|
||||
opts.taskId,
|
||||
commitPatch,
|
||||
details.message || commitSha,
|
||||
details.author,
|
||||
);
|
||||
}
|
||||
previousFilteredTree = currentFilteredTree;
|
||||
}
|
||||
|
||||
const finalFilteredTree = await writeSyntheticTree(opts.repoRoot, baselineSha, [opts.rootPatch]);
|
||||
const leftoverPatch = await git.diff.tree(opts.repoRoot, previousFilteredTree, finalFilteredTree, {
|
||||
allowFailure: true,
|
||||
binary: true,
|
||||
});
|
||||
if (leftoverPatch.trim()) {
|
||||
const msg = (opts.commitMessage && (await opts.commitMessage(leftoverPatch))) || opts.fallbackMessage;
|
||||
await commitPatchToBranchWorktree(tmpDir, opts.taskId, leftoverPatch, msg);
|
||||
}
|
||||
} finally {
|
||||
await git.worktree.tryRemove(opts.repoRoot, tmpDir);
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Commit task-only changes to a new branch.
|
||||
* Only root repo changes go on the branch. Nested repo patches are returned
|
||||
* separately since the parent git can't track files inside gitlinks.
|
||||
* Capture task-only changes from the isolation worktree onto a parent-repo
|
||||
* branch named `omp/task/${taskId}`. Only root-repo changes go on the branch;
|
||||
* nested-repo patches are returned separately because the parent git can't
|
||||
* track files inside gitlinks.
|
||||
*
|
||||
* If the agent committed inside isolation (HEAD moved past
|
||||
* `baseline.root.headCommit`), clean-baseline runs fetch the raw commit range
|
||||
* into the parent repo and later cherry-pick `baseSha..branchName`, preserving
|
||||
* every message and author verbatim. Dirty-baseline runs rewrite each agent
|
||||
* commit against the captured baseline WIP before committing it to the task
|
||||
* branch, so user staged/unstaged/untracked changes present at isolation
|
||||
* start are not replayed into the parent commit history.
|
||||
*
|
||||
* If the agent did not commit, the captured delta is collapsed onto a single
|
||||
* branch commit with an AI-generated (or fallback) message — the legacy
|
||||
* behaviour.
|
||||
*
|
||||
* Returns `null` when no root or nested changes exist.
|
||||
*/
|
||||
export async function commitToBranch(
|
||||
isolationDir: string,
|
||||
@@ -474,45 +611,84 @@ export async function commitToBranch(
|
||||
description: string | undefined,
|
||||
commitMessage?: (diff: string) => Promise<string | null>,
|
||||
): Promise<CommitToBranchResult | null> {
|
||||
const baselineSha = baseline.root.headCommit;
|
||||
const isolationHead = (await git.head.sha(isolationDir)) ?? "";
|
||||
const agentCommitted = isolationHead !== "" && isolationHead !== baselineSha;
|
||||
|
||||
const { rootPatch, nestedPatches } = await captureDeltaPatch(isolationDir, baseline);
|
||||
if (!rootPatch.trim() && nestedPatches.length === 0) return null;
|
||||
if (!rootPatch.trim()) return { nestedPatches };
|
||||
|
||||
const repoRoot = baseline.root.repoRoot;
|
||||
const branchName = `omp/task/${taskId}`;
|
||||
const fallbackMessage = description || taskId;
|
||||
|
||||
// Only create a branch if the root repo has changes
|
||||
if (rootPatch.trim()) {
|
||||
await git.branch.create(repoRoot, branchName);
|
||||
let branchCreated = false;
|
||||
|
||||
if (agentCommitted) {
|
||||
if (baselineHasRootWip(baseline.root)) {
|
||||
await replayFilteredAgentCommits({
|
||||
baseline,
|
||||
branchName,
|
||||
commitMessage,
|
||||
fallbackMessage,
|
||||
isolationDir,
|
||||
isolationHead,
|
||||
repoRoot,
|
||||
rootPatch,
|
||||
taskId,
|
||||
});
|
||||
} else {
|
||||
// Transfer the agent's commit objects (which live in isolation's `.git`,
|
||||
// stranded once `cleanupIsolation` tears the overlay down) into the parent
|
||||
// repo's object DB and create the branch at the agent's HEAD. `+HEAD:…`
|
||||
// force-overwrites a stale branch from a prior run.
|
||||
await git.fetch(repoRoot, isolationDir, "HEAD", `refs/heads/${branchName}`);
|
||||
|
||||
// Leftover = anything still uncommitted in isolation on top of the
|
||||
// agent's last commit (staged, unstaged, untracked). The agent didn't
|
||||
// commit it, so it goes in as one AI-summarized trailing commit.
|
||||
const leftoverPatch = await captureRepoDeltaPatch(isolationDir, {
|
||||
repoRoot: isolationDir,
|
||||
headCommit: isolationHead,
|
||||
staged: "",
|
||||
unstaged: "",
|
||||
untracked: [],
|
||||
untrackedPatch: "",
|
||||
});
|
||||
if (leftoverPatch.trim()) {
|
||||
const tmpDir = path.join(os.tmpdir(), `omp-branch-${Snowflake.next()}`);
|
||||
try {
|
||||
await git.worktree.add(repoRoot, tmpDir, branchName);
|
||||
const msg = (commitMessage && (await commitMessage(leftoverPatch))) || fallbackMessage;
|
||||
await commitPatchToBranchWorktree(tmpDir, taskId, leftoverPatch, msg);
|
||||
} finally {
|
||||
await git.worktree.tryRemove(repoRoot, tmpDir);
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
}
|
||||
branchCreated = true;
|
||||
} else if (rootPatch.trim()) {
|
||||
await git.branch.create(repoRoot, branchName, baselineSha);
|
||||
branchCreated = true;
|
||||
const tmpDir = path.join(os.tmpdir(), `omp-branch-${Snowflake.next()}`);
|
||||
try {
|
||||
await git.worktree.add(repoRoot, tmpDir, branchName);
|
||||
try {
|
||||
await git.patch.applyText(tmpDir, rootPatch);
|
||||
} catch (err) {
|
||||
if (err instanceof git.GitCommandError) {
|
||||
const stderr = err.result.stderr.slice(0, 2000);
|
||||
logger.error("commitToBranch: git apply failed", {
|
||||
taskId,
|
||||
exitCode: err.result.exitCode,
|
||||
stderr,
|
||||
patchSize: rootPatch.length,
|
||||
patchHead: rootPatch.slice(0, 500),
|
||||
});
|
||||
throw new Error(`git apply failed for task ${taskId}: ${stderr}`);
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
await git.stage.files(tmpDir);
|
||||
|
||||
const msg = (commitMessage && (await commitMessage(rootPatch))) || fallbackMessage;
|
||||
await git.commit(tmpDir, msg);
|
||||
await commitPatchToBranchWorktree(tmpDir, taskId, rootPatch, msg);
|
||||
} finally {
|
||||
await git.worktree.tryRemove(repoRoot, tmpDir);
|
||||
await fs.rm(tmpDir, { recursive: true, force: true });
|
||||
}
|
||||
}
|
||||
|
||||
return { branchName: rootPatch.trim() ? branchName : undefined, nestedPatches };
|
||||
return {
|
||||
branchName: branchCreated ? branchName : undefined,
|
||||
baseSha: baselineSha,
|
||||
nestedPatches,
|
||||
};
|
||||
}
|
||||
|
||||
export interface MergeBranchResult {
|
||||
@@ -524,13 +700,17 @@ export interface MergeBranchResult {
|
||||
}
|
||||
|
||||
/**
|
||||
* Cherry-pick task branch commits sequentially onto HEAD.
|
||||
* Each branch has a single commit that gets replayed cleanly.
|
||||
* Stops on first conflict and reports which branches succeeded.
|
||||
* Cherry-pick task branch commits sequentially onto HEAD. When `baseSha` is
|
||||
* provided the cherry-pick uses the inclusive range `baseSha..branchName`,
|
||||
* replaying every commit individually and preserving each commit's message
|
||||
* and author. When omitted, the branch is cherry-picked as a single commit
|
||||
* (legacy callers).
|
||||
*
|
||||
* Stops on the first conflict and reports which branches succeeded.
|
||||
*/
|
||||
export async function mergeTaskBranches(
|
||||
repoRoot: string,
|
||||
branches: Array<{ branchName: string; taskId: string; description?: string }>,
|
||||
branches: Array<{ branchName: string; taskId: string; description?: string; baseSha?: string }>,
|
||||
): Promise<MergeBranchResult> {
|
||||
// Serialize against other in-process git mutations on this repo: concurrent
|
||||
// background merges interleaving stash push/pop + cherry-pick would corrupt
|
||||
@@ -546,9 +726,10 @@ export async function mergeTaskBranches(
|
||||
let conflictResult: MergeBranchResult | undefined;
|
||||
|
||||
try {
|
||||
for (const { branchName } of branches) {
|
||||
for (const { branchName, baseSha } of branches) {
|
||||
try {
|
||||
await git.cherryPick(repoRoot, branchName);
|
||||
const target = baseSha ? `${baseSha}..${branchName}` : branchName;
|
||||
await git.cherryPick(repoRoot, target);
|
||||
} catch (err) {
|
||||
try {
|
||||
await git.cherryPick.abort(repoRoot);
|
||||
|
||||
@@ -132,6 +132,15 @@ export const TINY_MEMORY_LOCAL_MODELS = [
|
||||
unsupportedReason:
|
||||
"onnxruntime-node does not support Qwen3 RotaryEmbedding cache updates in onnx-community/Qwen3-1.7B-ONNX",
|
||||
},
|
||||
{
|
||||
key: "llama3.2:3b",
|
||||
repo: "onnx-community/Llama-3.2-3B-Instruct-ONNX",
|
||||
dtype: "q4",
|
||||
label: "Llama 3.2 3B",
|
||||
description:
|
||||
"Larger Llama 3.2 option for local memory/classifier tasks; higher quality potential at higher disk/RAM/latency cost.",
|
||||
contextNote: "Use when larger model capacity is preferred over faster load times.",
|
||||
},
|
||||
{
|
||||
key: "gemma-3-1b",
|
||||
repo: "onnx-community/gemma-3-1b-it-ONNX",
|
||||
@@ -161,6 +170,7 @@ export const TINY_MEMORY_LOCAL_MODELS = [
|
||||
export const TINY_MEMORY_MODEL_VALUES = [
|
||||
ONLINE_MEMORY_MODEL_KEY,
|
||||
"qwen3-1.7b",
|
||||
"llama3.2:3b",
|
||||
"gemma-3-1b",
|
||||
"qwen2.5-1.5b",
|
||||
"lfm2-1.2b",
|
||||
|
||||
@@ -29,7 +29,12 @@ import type { TinyTitleProgressEvent, TinyTitleWorkerInbound, TinyTitleWorkerOut
|
||||
type PendingRequest =
|
||||
| { kind: "generate"; modelKey: TinyTitleLocalModelKey; resolve: (title: string | null) => void }
|
||||
| { kind: "complete"; modelKey: TinyMemoryLocalModelKey; resolve: (text: string | null) => void }
|
||||
| { kind: "download"; modelKey: TinyLocalModelKey; resolve: (ok: boolean) => void };
|
||||
| { kind: "download"; modelKey: TinyLocalModelKey; resolve: (result: TinyTitleDownloadResult) => void };
|
||||
|
||||
export interface TinyTitleDownloadResult {
|
||||
ok: boolean;
|
||||
error?: string;
|
||||
}
|
||||
|
||||
export interface TinyTitleDownloadOptions {
|
||||
signal?: AbortSignal;
|
||||
@@ -269,21 +274,21 @@ export class TinyTitleClient {
|
||||
}
|
||||
}
|
||||
|
||||
async downloadModel(modelKey: string, options: TinyTitleDownloadOptions = {}): Promise<boolean> {
|
||||
if (!isTinyLocalModelKey(modelKey)) return false;
|
||||
if (options.signal?.aborted) return false;
|
||||
async downloadModel(modelKey: string, options: TinyTitleDownloadOptions = {}): Promise<TinyTitleDownloadResult> {
|
||||
if (!isTinyLocalModelKey(modelKey)) return { ok: false };
|
||||
if (options.signal?.aborted) return { ok: false };
|
||||
|
||||
const unsubscribe = options.onProgress ? this.onProgress(options.onProgress) : undefined;
|
||||
try {
|
||||
const worker = this.#ensureWorker();
|
||||
const id = String(++this.#nextRequestId);
|
||||
const { promise, resolve } = Promise.withResolvers<boolean>();
|
||||
const { promise, resolve } = Promise.withResolvers<TinyTitleDownloadResult>();
|
||||
this.#addPending(id, { kind: "download", modelKey, resolve });
|
||||
const abort = (): void => {
|
||||
const pending = this.#pending.get(id);
|
||||
if (pending?.kind !== "download") return;
|
||||
this.#deletePending(id);
|
||||
pending.resolve(false);
|
||||
pending.resolve({ ok: false });
|
||||
};
|
||||
options.signal?.addEventListener("abort", abort, { once: true });
|
||||
try {
|
||||
@@ -294,11 +299,12 @@ export class TinyTitleClient {
|
||||
this.#deletePending(id);
|
||||
}
|
||||
} catch (error) {
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
logger.debug("tiny-title: local model download failed", {
|
||||
modelKey,
|
||||
error: error instanceof Error ? error.message : String(error),
|
||||
error: message,
|
||||
});
|
||||
return false;
|
||||
return { ok: false, error: message };
|
||||
} finally {
|
||||
unsubscribe?.();
|
||||
}
|
||||
@@ -314,7 +320,7 @@ export class TinyTitleClient {
|
||||
for (const pending of this.#pending.values()) {
|
||||
this.#emitProgress({ modelKey: pending.modelKey, status: "error" });
|
||||
if (pending.kind === "generate" || pending.kind === "complete") pending.resolve(null);
|
||||
else pending.resolve(false);
|
||||
else pending.resolve({ ok: false });
|
||||
}
|
||||
this.#pending.clear();
|
||||
this.#refed = false;
|
||||
@@ -379,7 +385,7 @@ export class TinyTitleClient {
|
||||
return;
|
||||
}
|
||||
if (message.type === "downloaded") {
|
||||
if (pending.kind === "download") pending.resolve(true);
|
||||
if (pending.kind === "download") pending.resolve({ ok: true });
|
||||
return;
|
||||
}
|
||||
if (message.type === "completion") {
|
||||
@@ -389,8 +395,8 @@ export class TinyTitleClient {
|
||||
logger.debug("tiny-title: worker returned error", { error: message.error });
|
||||
this.#markFailedModel(pending);
|
||||
this.#emitProgress({ modelKey: pending.modelKey, status: "error" });
|
||||
if (pending.kind === "generate" || pending.kind === "complete") pending.resolve(null);
|
||||
else pending.resolve(false);
|
||||
if (pending.kind === "download") pending.resolve({ ok: false, error: message.error });
|
||||
else pending.resolve(null);
|
||||
void this.terminate();
|
||||
}
|
||||
|
||||
@@ -407,7 +413,7 @@ export class TinyTitleClient {
|
||||
for (const pending of this.#pending.values()) {
|
||||
this.#emitProgress({ modelKey: pending.modelKey, status: "error" });
|
||||
if (pending.kind === "generate" || pending.kind === "complete") pending.resolve(null);
|
||||
else pending.resolve(false);
|
||||
else pending.resolve({ ok: false, error: error.message });
|
||||
}
|
||||
this.#pending.clear();
|
||||
void this.terminate();
|
||||
|
||||
@@ -85,8 +85,26 @@ const searchSchema = type({
|
||||
});
|
||||
|
||||
export type GrepToolInput = typeof searchSchema.infer;
|
||||
function parseStringEncodedPathArray(input: string): string[] | null {
|
||||
const trimmed = input.trim();
|
||||
if (!trimmed.startsWith("[") || !trimmed.endsWith("]")) return null;
|
||||
|
||||
let parsed: unknown;
|
||||
try {
|
||||
parsed = JSON.parse(trimmed);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
|
||||
if (!Array.isArray(parsed) || parsed.some(entry => typeof entry !== "string")) {
|
||||
return null;
|
||||
}
|
||||
return parsed;
|
||||
}
|
||||
|
||||
export function toPathList(input: string | string[] | undefined): string[] {
|
||||
return typeof input === "string" ? [input] : (input ?? []);
|
||||
if (typeof input === "string") return parseStringEncodedPathArray(input) ?? [input];
|
||||
return input ?? [];
|
||||
}
|
||||
|
||||
/** Maximum number of distinct files surfaced in a single response. The
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import type { InMemorySnapshotStore } from "@oh-my-pi/hashline";
|
||||
import type { AgentTelemetryConfig, AgentTool } from "@oh-my-pi/pi-agent-core";
|
||||
import type { FetchImpl, ImageContent, Model, ServiceTier, ToolChoice } from "@oh-my-pi/pi-ai";
|
||||
import type { FetchImpl, ImageContent, Model, ServiceTierByFamily, ToolChoice } from "@oh-my-pi/pi-ai";
|
||||
import { logger } from "@oh-my-pi/pi-utils";
|
||||
import type { AsyncJobManager } from "../async/job-manager";
|
||||
import type { Rule } from "../capability/rule";
|
||||
@@ -240,8 +240,8 @@ export interface ToolSession {
|
||||
getActiveModelString?: () => string | undefined;
|
||||
/** Get the current session model object (provider/api capabilities), regardless of how it was chosen. */
|
||||
getActiveModel?: () => Model | undefined;
|
||||
/** Get the session's live effective service tier (undefined = none). Source of truth for subagent `serviceTierSubagent: inherit`. */
|
||||
getServiceTier?: () => ServiceTier | undefined;
|
||||
/** Get the session's live per-family service tiers (undefined = none). Source of truth for subagent `tier.subagent: inherit`. */
|
||||
getServiceTierByFamily?: () => ServiceTierByFamily | undefined;
|
||||
/** Auth storage for passing to subagents (avoids re-discovery) */
|
||||
authStorage?: import("../session/auth-storage").AuthStorage;
|
||||
/** Model registry for passing to subagents (avoids re-discovery) */
|
||||
|
||||
@@ -172,7 +172,7 @@ export class IrcTool implements AgentTool<typeof ircSchema, IrcDetails> {
|
||||
case "wait":
|
||||
return this.#executeWait(senderId, params, signal);
|
||||
case "inbox":
|
||||
return this.#executeInbox(senderId, params);
|
||||
return this.#executeInbox(registry, senderId, params);
|
||||
default:
|
||||
return errorResult("Unknown irc op.", { op: params.op });
|
||||
}
|
||||
@@ -371,8 +371,14 @@ export class IrcTool implements AgentTool<typeof ircSchema, IrcDetails> {
|
||||
};
|
||||
}
|
||||
|
||||
#executeInbox(senderId: string, params: IrcParams): AgentToolResult<IrcDetails> {
|
||||
const messages = IrcBus.global().inbox(senderId, { peek: params.peek });
|
||||
#executeInbox(registry: AgentRegistry, senderId: string, params: IrcParams): AgentToolResult<IrcDetails> {
|
||||
const busMessages = IrcBus.global().inbox(senderId, { peek: params.peek });
|
||||
const session = registry.get(senderId)?.session;
|
||||
const pendingMessages =
|
||||
typeof session?.drainPendingIrcInboxMessages === "function"
|
||||
? session.drainPendingIrcInboxMessages(senderId)
|
||||
: [];
|
||||
const messages = [...busMessages, ...pendingMessages].sort((a, b) => a.ts - b.ts);
|
||||
if (messages.length === 0) {
|
||||
return {
|
||||
content: [{ type: "text", text: "Inbox empty." }],
|
||||
|
||||
@@ -21,6 +21,16 @@ export interface OutputValidator {
|
||||
validate(value: unknown): JsonSchemaValidationResult;
|
||||
/** Top-level required property names. Empty if the schema has no `required` array at root. */
|
||||
readonly requiredFields: readonly string[];
|
||||
/**
|
||||
* Per-label validators for incremental yields (`type: ["<label>"]`). Each entry validates the
|
||||
* `data` payload of a single section against the matching top-level property's sub-schema —
|
||||
* array-typed properties (e.g. `findings`) use the items schema since each yield contributes
|
||||
* one element, while scalar properties use the property schema directly. Unknown labels (not
|
||||
* top-level properties) have no entry and skip per-call validation. Lets the yield tool give
|
||||
* the model retry feedback on a section as soon as it arrives, instead of deferring every
|
||||
* mismatch to the parent's post-mortem `schema_violation`.
|
||||
*/
|
||||
readonly validateSection: ReadonlyMap<string, (value: unknown) => JsonSchemaValidationResult>;
|
||||
}
|
||||
|
||||
export interface BuildOutputValidatorResult {
|
||||
@@ -72,10 +82,38 @@ export function buildOutputValidator(schema: unknown): BuildOutputValidatorResul
|
||||
validator: {
|
||||
requiredFields: required,
|
||||
validate: value => validateJsonSchemaValue(jsonSchemaRecord, value),
|
||||
validateSection: buildSectionValidators(jsonSchemaRecord),
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Build per-top-level-property validators for incremental yields.
|
||||
*
|
||||
* Each entry validates the `data` payload of one `type: ["<label>"]` section against the
|
||||
* matching property's sub-schema — array-typed properties (e.g. `findings`, derived from JTD
|
||||
* `elements`) use the items schema since each yield contributes one element, while scalar
|
||||
* properties use the property schema directly. Unknown labels (anything not declared as a
|
||||
* top-level property) are deliberately omitted so user-defined section labels still pass.
|
||||
*/
|
||||
function buildSectionValidators(
|
||||
jsonSchema: Record<string, unknown>,
|
||||
): ReadonlyMap<string, (value: unknown) => JsonSchemaValidationResult> {
|
||||
const validators = new Map<string, (value: unknown) => JsonSchemaValidationResult>();
|
||||
const properties = jsonSchema.properties;
|
||||
if (properties === null || typeof properties !== "object") return validators;
|
||||
for (const [label, raw] of Object.entries(properties as Record<string, unknown>)) {
|
||||
if (raw === null || typeof raw !== "object") continue;
|
||||
const propRecord = raw as Record<string, unknown>;
|
||||
const sectionSchema =
|
||||
propRecord.type === "array" && propRecord.items !== undefined && propRecord.items !== null
|
||||
? (propRecord.items as Record<string, unknown>)
|
||||
: propRecord;
|
||||
validators.set(label, value => validateJsonSchemaValue(sectionSchema, value));
|
||||
}
|
||||
return validators;
|
||||
}
|
||||
|
||||
/** Produce the executor's headline+missing-required summary from a failed validation. */
|
||||
export function summarizeValidationFailure(
|
||||
result: JsonSchemaValidationResult,
|
||||
|
||||
@@ -14,7 +14,15 @@ import type { ImageContent, TextContent } from "@oh-my-pi/pi-ai";
|
||||
import { glob, type SummaryResult, summarizeCode } from "@oh-my-pi/pi-natives";
|
||||
import type { Component } from "@oh-my-pi/pi-tui";
|
||||
import { Text } from "@oh-my-pi/pi-tui";
|
||||
import { getRemoteDir, type ImageMetadata, logger, prompt, readImageMetadata, untilAborted } from "@oh-my-pi/pi-utils";
|
||||
import {
|
||||
getRemoteDir,
|
||||
type ImageMetadata,
|
||||
isProbablyBinary,
|
||||
logger,
|
||||
prompt,
|
||||
readImageMetadata,
|
||||
untilAborted,
|
||||
} from "@oh-my-pi/pi-utils";
|
||||
import { type } from "arktype";
|
||||
import { LRUCache } from "lru-cache/raw";
|
||||
import {
|
||||
@@ -2314,6 +2322,25 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
content = [{ type: "text", text: `[Cannot read ${ext} file: conversion failed]` }];
|
||||
}
|
||||
} else {
|
||||
// Binary sniff before any UTF-8 text materialization. A binary file
|
||||
// (font, object, archive, packed blob) decodes to NUL/control bytes and
|
||||
// U+FFFD mojibake that corrupts the terminal and burns context. Images,
|
||||
// notebooks, and markit-convertible documents were already routed above;
|
||||
// everything reaching here is meant to be plain text. `:raw` stays the
|
||||
// explicit escape hatch for reading bytes verbatim. This single guard
|
||||
// covers both the multi-range and single-range disk paths below.
|
||||
if (!isRawSelector(parsed) && (await isProbablyBinary(absolutePath))) {
|
||||
return toolResult<ReadToolDetails>({ resolvedPath: absolutePath, suffixResolution })
|
||||
.text(
|
||||
prependSuffixResolutionNotice(
|
||||
`[Cannot read binary file '${formatPathRelativeToCwd(absolutePath, this.session.cwd)}' (${formatBytes(fileSize)}); not valid UTF-8 text. Use ':raw' to read bytes verbatim.]`,
|
||||
suffixResolution,
|
||||
),
|
||||
)
|
||||
.sourcePath(absolutePath)
|
||||
.done();
|
||||
}
|
||||
|
||||
if (
|
||||
parsed.kind === "none" &&
|
||||
this.session.settings.get("read.summarize.enabled") &&
|
||||
@@ -2449,33 +2476,6 @@ export class ReadTool implements AgentTool<typeof readSchema, ReadToolDetails> {
|
||||
// counts in `truncation` keep reflecting the source, not the trimmed
|
||||
// view — column truncation surfaces separately via `.limits()`.
|
||||
const rawSelector = isRawSelector(parsed);
|
||||
// Binary sniff: NUL bytes mean the file is not displayable text
|
||||
// (binary, or UTF-16 which has NULs in the ASCII range) — emit a
|
||||
// notice instead of mojibake filling the line budget. `:raw`
|
||||
// stays an explicit escape hatch.
|
||||
//
|
||||
// `collectedLines` covers the common case where at least one
|
||||
// physical line terminates within the byte budget. Binary blobs
|
||||
// without newlines (videos, archives, packed JSON) leave it
|
||||
// empty; their bytes only land in `firstLinePreview`, which the
|
||||
// `firstLineExceedsLimit` branch below would otherwise emit
|
||||
// verbatim. Sniffing the preview here keeps the refusal uniform.
|
||||
if (!rawSelector) {
|
||||
const hasNul = (text: string): boolean => text.includes("\u0000");
|
||||
const binaryDetected =
|
||||
collectedLines.some(hasNul) || (firstLinePreview !== undefined && hasNul(firstLinePreview.text));
|
||||
if (binaryDetected) {
|
||||
return toolResult<ReadToolDetails>({ resolvedPath: absolutePath, suffixResolution })
|
||||
.text(
|
||||
prependSuffixResolutionNotice(
|
||||
`[Cannot read binary file '${formatPathRelativeToCwd(absolutePath, this.session.cwd)}' (${formatBytes(fileSize)}); content contains NUL bytes (binary or UTF-16 encoded)]`,
|
||||
suffixResolution,
|
||||
),
|
||||
)
|
||||
.sourcePath(absolutePath)
|
||||
.done();
|
||||
}
|
||||
}
|
||||
const maxColumns = resolveOutputMaxColumns(this.session.settings);
|
||||
// Column truncation is display-only. `collectedLines` MUST stay
|
||||
// byte-for-byte with the on-disk content so the snapshot recorded
|
||||
|
||||
@@ -98,6 +98,16 @@ function parseYieldType(value: unknown): string | string[] | undefined {
|
||||
throw new Error("type must be a string or non-empty array of strings");
|
||||
}
|
||||
|
||||
/**
|
||||
* Render an incremental yield's `type: [...]` labels as a quoted, comma-separated list for
|
||||
* model-facing retry messages — keeps the failed section labelled even when the yield carried
|
||||
* multiple labels at once.
|
||||
*/
|
||||
function formatYieldLabels(labels: readonly string[]): string {
|
||||
if (labels.length === 0) return '""';
|
||||
return labels.map(label => `"${label}"`).join(", ");
|
||||
}
|
||||
|
||||
/**
|
||||
* Expand a plain-object `data` schema into a strict union that ALSO accepts each
|
||||
* top-level section value (and array element) on its own. Agents that yield
|
||||
@@ -202,10 +212,12 @@ export class YieldTool implements AgentTool<TSchema, YieldDetails> {
|
||||
lenientArgValidation = true;
|
||||
|
||||
readonly #validate?: (value: unknown) => JsonSchemaValidationResult;
|
||||
readonly #validateSection?: ReadonlyMap<string, (value: unknown) => JsonSchemaValidationResult>;
|
||||
#schemaValidationFailures = 0;
|
||||
|
||||
constructor(session: ToolSession) {
|
||||
let validate: ((value: unknown) => JsonSchemaValidationResult) | undefined;
|
||||
let validateSection: ReadonlyMap<string, (value: unknown) => JsonSchemaValidationResult> | undefined;
|
||||
let parameters: TSchema;
|
||||
|
||||
try {
|
||||
@@ -217,6 +229,7 @@ export class YieldTool implements AgentTool<TSchema, YieldDetails> {
|
||||
} = buildOutputValidator(session.outputSchema);
|
||||
if (validator) {
|
||||
validate = value => validator.validate(value);
|
||||
validateSection = validator.validateSection;
|
||||
}
|
||||
|
||||
const schemaHint = formatSchema(normalizedSchema ?? session.outputSchema);
|
||||
@@ -266,6 +279,7 @@ export class YieldTool implements AgentTool<TSchema, YieldDetails> {
|
||||
}
|
||||
|
||||
this.#validate = validate;
|
||||
this.#validateSection = validateSection;
|
||||
this.parameters = parameters;
|
||||
}
|
||||
|
||||
@@ -307,22 +321,25 @@ export class YieldTool implements AgentTool<TSchema, YieldDetails> {
|
||||
if (data === null) {
|
||||
throw new Error("data is required when yield indicates success");
|
||||
}
|
||||
if (this.#validate && !isIncremental) {
|
||||
const parsed = this.#validate(data);
|
||||
if (!parsed.success) {
|
||||
this.#schemaValidationFailures++;
|
||||
if (this.#schemaValidationFailures <= MAX_SCHEMA_RETRIES) {
|
||||
const remaining = MAX_SCHEMA_RETRIES - this.#schemaValidationFailures;
|
||||
const retryHint =
|
||||
remaining > 0
|
||||
? ` Call yield again with the corrected shape — ${remaining} retry attempt(s) remain before the schema constraint is dropped.`
|
||||
: " Call yield again with the corrected shape — this is the final retry before the schema constraint is dropped.";
|
||||
throw new Error(
|
||||
`Output does not match schema: ${formatAllValidationIssues(parsed.issues)}.${retryHint}`,
|
||||
);
|
||||
}
|
||||
schemaValidationOverridden = true;
|
||||
const sectionFailure = isIncremental
|
||||
? this.#validateIncrementalSection(yieldType as string[], data)
|
||||
: this.#validate
|
||||
? this.#validate(data)
|
||||
: undefined;
|
||||
if (sectionFailure && !sectionFailure.success) {
|
||||
this.#schemaValidationFailures++;
|
||||
if (this.#schemaValidationFailures <= MAX_SCHEMA_RETRIES) {
|
||||
const remaining = MAX_SCHEMA_RETRIES - this.#schemaValidationFailures;
|
||||
const retryHint =
|
||||
remaining > 0
|
||||
? ` Call yield again with the corrected shape — ${remaining} retry attempt(s) remain before the schema constraint is dropped.`
|
||||
: " Call yield again with the corrected shape — this is the final retry before the schema constraint is dropped.";
|
||||
const scope = isIncremental ? `Section ${formatYieldLabels(yieldType as string[])}` : "Output";
|
||||
throw new Error(
|
||||
`${scope} does not match schema: ${formatAllValidationIssues(sectionFailure.issues)}.${retryHint}`,
|
||||
);
|
||||
}
|
||||
schemaValidationOverridden = true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -344,6 +361,26 @@ export class YieldTool implements AgentTool<TSchema, YieldDetails> {
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate the `data` payload of an incremental yield (`type: ["<label>", …]`) against
|
||||
* the matching property's sub-validator. Returns the first failure across all known labels,
|
||||
* or `undefined` when no label is recognised (user-defined section labels stay loose) or
|
||||
* when all known labels accept the value. Lets the model see the same retry feedback that
|
||||
* the terminal-yield path already produces, instead of leaking the mismatch through to
|
||||
* the parent's post-mortem `schema_violation`.
|
||||
*/
|
||||
#validateIncrementalSection(labels: string[], data: unknown): JsonSchemaValidationResult | undefined {
|
||||
const subValidators = this.#validateSection;
|
||||
if (!subValidators || subValidators.size === 0) return undefined;
|
||||
for (const label of labels) {
|
||||
const sub = subValidators.get(label);
|
||||
if (!sub) continue;
|
||||
const parsed = sub(data);
|
||||
if (!parsed.success) return parsed;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
}
|
||||
|
||||
// Register subprocess tool handler for extraction + termination.
|
||||
|
||||
@@ -10,7 +10,7 @@ import path from "node:path";
|
||||
import { formatHashlineHeader, formatNumberedLines, type SnapshotStore } from "@oh-my-pi/hashline";
|
||||
import type { AgentMessage } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ImageContent } from "@oh-my-pi/pi-ai";
|
||||
import { formatAge, formatBytes, readImageMetadata } from "@oh-my-pi/pi-utils";
|
||||
import { formatAge, formatBytes, isProbablyBinary, readImageMetadata } from "@oh-my-pi/pi-utils";
|
||||
import { canonicalSnapshotKey } from "../edit/file-snapshot-store";
|
||||
import { normalizeToLF } from "../edit/normalize";
|
||||
import type { FileMentionMessage } from "../session/messages";
|
||||
@@ -257,6 +257,15 @@ export async function generateFileMentionMessages(
|
||||
});
|
||||
continue;
|
||||
}
|
||||
if (await isProbablyBinary(absolutePath)) {
|
||||
files.push({
|
||||
path: resolvedPath,
|
||||
content: `(skipped auto-read: binary file, ${formatBytes(stat.size)})`,
|
||||
byteSize: stat.size,
|
||||
skippedReason: "binary",
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
const content = await Bun.file(absolutePath).text();
|
||||
const snapshotStore = options?.useHashLines ? options.snapshotStore : undefined;
|
||||
|
||||
@@ -74,8 +74,20 @@ export interface StatusOptions {
|
||||
readonly z?: boolean;
|
||||
}
|
||||
|
||||
export interface CommitAuthor {
|
||||
readonly date?: string;
|
||||
readonly email: string;
|
||||
readonly name: string;
|
||||
}
|
||||
|
||||
export interface CommitDetails {
|
||||
readonly author: CommitAuthor;
|
||||
readonly message: string;
|
||||
}
|
||||
|
||||
export interface CommitOptions {
|
||||
readonly allowEmpty?: boolean;
|
||||
readonly author?: CommitAuthor;
|
||||
readonly files?: readonly string[];
|
||||
readonly signal?: AbortSignal;
|
||||
}
|
||||
@@ -91,6 +103,7 @@ export interface PatchOptions {
|
||||
readonly cached?: boolean;
|
||||
readonly check?: boolean;
|
||||
readonly env?: Record<string, string | undefined>;
|
||||
readonly threeWay?: boolean;
|
||||
readonly signal?: AbortSignal;
|
||||
}
|
||||
|
||||
@@ -359,6 +372,7 @@ function buildApplyArgs(patchPath: string, options: PatchOptions): string[] {
|
||||
const args = ["apply"];
|
||||
if (options.check) args.push("--check");
|
||||
if (options.cached) args.push("--cached");
|
||||
if (options.threeWay) args.push("--3way");
|
||||
args.push("--binary", patchPath);
|
||||
return args;
|
||||
}
|
||||
@@ -1122,6 +1136,10 @@ export const stage = {
|
||||
/** Create a commit with the given message (passed via stdin). */
|
||||
export async function commit(cwd: string, message: string, options: CommitOptions = {}): Promise<GitCommandResult> {
|
||||
const args = ["commit", "-F", "-"];
|
||||
if (options.author) {
|
||||
args.push(`--author=${options.author.name} <${options.author.email}>`);
|
||||
if (options.author.date) args.push(`--date=${options.author.date}`);
|
||||
}
|
||||
if (options.allowEmpty) args.push("--allow-empty");
|
||||
if (options.files?.length) args.push("--", ...options.files);
|
||||
return runChecked(cwd, args, { signal: options.signal, stdin: message });
|
||||
@@ -1195,6 +1213,19 @@ export const show = Object.assign(
|
||||
},
|
||||
);
|
||||
|
||||
/** Read commit message and author metadata for replay/rewrite flows. */
|
||||
export async function commitDetails(cwd: string, revision: string, signal?: AbortSignal): Promise<CommitDetails> {
|
||||
const raw = await runText(cwd, ["show", "-s", "--format=%an%x00%ae%x00%aI%x00%B", revision], {
|
||||
readOnly: true,
|
||||
signal,
|
||||
});
|
||||
const [name = "", email = "", date = "", ...messageParts] = raw.split("\0");
|
||||
return {
|
||||
author: { date, email, name },
|
||||
message: messageParts.join("\0").replace(/\n$/, ""),
|
||||
};
|
||||
}
|
||||
|
||||
// ════════════════════════════════════════════════════════════════════════════
|
||||
// API: log
|
||||
// ════════════════════════════════════════════════════════════════════════════
|
||||
@@ -1212,6 +1243,13 @@ export const log = {
|
||||
},
|
||||
};
|
||||
|
||||
export const revList = {
|
||||
/** Commits in `base..head`, oldest first. */
|
||||
async range(cwd: string, base: string, head: string, signal?: AbortSignal): Promise<string[]> {
|
||||
return splitLines(await runText(cwd, ["rev-list", "--reverse", `${base}..${head}`], { readOnly: true, signal }));
|
||||
},
|
||||
};
|
||||
|
||||
// ════════════════════════════════════════════════════════════════════════════
|
||||
// API: branch
|
||||
// ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
@@ -35,7 +35,7 @@ const RECENCY_TO_DDG_DF: Record<NonNullable<SearchParams["recency"]>, string> =
|
||||
* the orchestrator can fall through to the next provider with context.
|
||||
*/
|
||||
const BROWSER_USER_AGENT =
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36";
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/149.0.0.0 Safari/537.36";
|
||||
|
||||
interface ParsedResult {
|
||||
title: string;
|
||||
@@ -130,15 +130,29 @@ async function callDuckDuckGoHtml(params: SearchParams): Promise<string> {
|
||||
const form = new URLSearchParams({ q: params.query, kl: "us-en" });
|
||||
const df = params.recency ? RECENCY_TO_DDG_DF[params.recency] : undefined;
|
||||
if (df) form.set("df", df);
|
||||
// Add b: "" parameter as specified in the browser fetch template to match real browser form submission
|
||||
form.set("b", "");
|
||||
|
||||
const response = await (params.fetch ?? fetch)(DUCKDUCKGO_HTML_URL, {
|
||||
method: "POST",
|
||||
body: form.toString(),
|
||||
headers: {
|
||||
Accept:
|
||||
"text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
|
||||
"Accept-Language": "en,en-US;q=0.9",
|
||||
"Cache-Control": "max-age=0",
|
||||
"Content-Type": "application/x-www-form-urlencoded",
|
||||
Priority: "u=0, i",
|
||||
"Sec-Ch-Ua": '"Google Chrome";v="149", "Chromium";v="149", "Not)A;Brand";v="24"',
|
||||
"Sec-Ch-Ua-Mobile": "?0",
|
||||
"Sec-Ch-Ua-Platform": '"macOS"',
|
||||
"Sec-Fetch-Dest": "document",
|
||||
"Sec-Fetch-Mode": "navigate",
|
||||
"Sec-Fetch-Site": "same-origin",
|
||||
"Sec-Fetch-User": "?1",
|
||||
"Upgrade-Insecure-Requests": "1",
|
||||
"User-Agent": BROWSER_USER_AGENT,
|
||||
Accept: "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "en-US,en;q=0.5",
|
||||
Referer: "https://html.duckduckgo.com/",
|
||||
},
|
||||
signal: withHardTimeout(params.signal),
|
||||
});
|
||||
|
||||
@@ -694,4 +694,81 @@ describe("AgentSession auto-compaction progress guard", () => {
|
||||
expect(noProgress.length).toBe(1);
|
||||
expect(noProgress[0].level).toBe("warning");
|
||||
});
|
||||
|
||||
it("auto-continues (no warning) when a shake rescue frees the oversized tail", async () => {
|
||||
// The escalation contract: compaction cut at the only turn boundary but the
|
||||
// kept tail (e.g. a huge tool result) still sits over the recovery band. The
|
||||
// guard now runs an elide shake INSIDE that tail; once it frees enough, the
|
||||
// auto-continue proceeds instead of pausing with the no-progress warning.
|
||||
const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined as never);
|
||||
vi.spyOn(session.agent, "continue").mockResolvedValue();
|
||||
// Residual is over the band until the rescue elides the tail, then drops.
|
||||
let shaken = false;
|
||||
vi.spyOn(session, "getContextUsage").mockImplementation(() =>
|
||||
shaken
|
||||
? { tokens: 1000, contextWindow: 200000, percent: 0.5 }
|
||||
: { tokens: 190000, contextWindow: 200000, percent: 95 },
|
||||
);
|
||||
const shakeSpy = vi.spyOn(session, "shake").mockImplementation(async () => {
|
||||
shaken = true;
|
||||
return { mode: "elide", toolResultsDropped: 1, blocksDropped: 0, tokensFreed: 160000, artifactId: "art-1" };
|
||||
});
|
||||
|
||||
const notices = collectNotices();
|
||||
|
||||
const { promise: compactionDone, resolve: onCompactionDone } = Promise.withResolvers<void>();
|
||||
session.subscribe(event => {
|
||||
if (event.type === "auto_compaction_end") onCompactionDone();
|
||||
});
|
||||
|
||||
const assistantMsg = highUsageAssistant();
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: assistantMsg });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMsg] });
|
||||
|
||||
await compactionDone;
|
||||
await session.waitForIdle();
|
||||
|
||||
expect(shakeSpy).toHaveBeenCalledWith("elide", expect.anything());
|
||||
expect(promptSpy).toHaveBeenCalledTimes(1);
|
||||
const noProgress = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes(NO_PROGRESS_FRAGMENT));
|
||||
expect(noProgress.length).toBe(0);
|
||||
const recovery = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes("dead-end recovery"));
|
||||
expect(recovery.length).toBe(1);
|
||||
expect(recovery[0].level).toBe("info");
|
||||
});
|
||||
|
||||
it("still warns when a shake rescue cannot free the irreducible tail", async () => {
|
||||
// When the oversized tail has nothing elide-eligible (image-only or plain
|
||||
// prose), the rescue frees nothing, the residual stays over the band, and
|
||||
// the guard MUST still pause with the single no-progress warning.
|
||||
const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined as never);
|
||||
vi.spyOn(session.agent, "continue").mockResolvedValue();
|
||||
vi.spyOn(session, "getContextUsage").mockReturnValue({ tokens: 190000, contextWindow: 200000, percent: 95 });
|
||||
// Nothing eligible: shake reports zero dropped, so residual is unchanged.
|
||||
const shakeSpy = vi
|
||||
.spyOn(session, "shake")
|
||||
.mockResolvedValue({ mode: "elide", toolResultsDropped: 0, blocksDropped: 0, tokensFreed: 0 });
|
||||
|
||||
const notices = collectNotices();
|
||||
|
||||
const { promise: compactionDone, resolve: onCompactionDone } = Promise.withResolvers<void>();
|
||||
session.subscribe(event => {
|
||||
if (event.type === "auto_compaction_end") onCompactionDone();
|
||||
});
|
||||
|
||||
const assistantMsg = highUsageAssistant();
|
||||
session.agent.emitExternalEvent({ type: "message_end", message: assistantMsg });
|
||||
session.agent.emitExternalEvent({ type: "agent_end", messages: [assistantMsg] });
|
||||
|
||||
await compactionDone;
|
||||
await session.waitForIdle();
|
||||
|
||||
expect(shakeSpy).toHaveBeenCalledWith("elide", expect.anything());
|
||||
expect(promptSpy).not.toHaveBeenCalled();
|
||||
const noProgress = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes(NO_PROGRESS_FRAGMENT));
|
||||
expect(noProgress.length).toBe(1);
|
||||
expect(noProgress[0].level).toBe("warning");
|
||||
const recovery = notices.filter(n => n.source === NOTICE_SOURCE && n.message.includes("dead-end recovery"));
|
||||
expect(recovery.length).toBe(0);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -6,12 +6,14 @@ import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream"
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import type { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions";
|
||||
import { ExtensionRuntime, loadExtensionFromFactory } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/loader";
|
||||
import { ExtensionRunner } from "@oh-my-pi/pi-coding-agent/extensibility/extensions/runner";
|
||||
import type { GoalModeState } from "@oh-my-pi/pi-coding-agent/goals/state";
|
||||
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { EventBus } from "@oh-my-pi/pi-coding-agent/utils/event-bus";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import { type } from "arktype";
|
||||
|
||||
@@ -265,6 +267,43 @@ describe("AgentSession mid-run threshold compaction", () => {
|
||||
expect(persistedToolTurnRoles).toEqual(["assistant", "toolResult"]);
|
||||
});
|
||||
|
||||
it("treats same-key assistant content variants as persisted before mid-run compaction", async () => {
|
||||
const extensionRuntime = new ExtensionRuntime();
|
||||
const extension = await loadExtensionFromFactory(
|
||||
pi => {
|
||||
pi.on("message_end", event => {
|
||||
if (event.message.role !== "assistant" || event.message.stopReason !== "toolUse") return;
|
||||
const [block] = event.message.content;
|
||||
if (block?.type !== "toolCall") return;
|
||||
event.message.content = [{ ...block, arguments: { cmd: "display-variant" } }];
|
||||
});
|
||||
},
|
||||
tempDir.path(),
|
||||
new EventBus(),
|
||||
extensionRuntime,
|
||||
"assistant-display-variant",
|
||||
);
|
||||
const extensionAuthStorage = await AuthStorage.create(path.join(tempDir.path(), "extension-auth-variant.db"));
|
||||
cleanups.push(async () => {
|
||||
extensionAuthStorage.close();
|
||||
});
|
||||
const extensionRunner = new ExtensionRunner(
|
||||
[extension],
|
||||
extensionRuntime,
|
||||
tempDir.path(),
|
||||
SessionManager.inMemory(),
|
||||
new ModelRegistry(extensionAuthStorage, path.join(tempDir.path(), "extension-models-variant.yml")),
|
||||
);
|
||||
const { session, observedContexts } = await createHarness({}, { extensionRunner });
|
||||
const compactSpy = mockCompaction("MID-RUN-COMPACTED-WITH-CONTENT-VARIANT");
|
||||
|
||||
await session.prompt("work on the release");
|
||||
|
||||
expect(compactSpy).toHaveBeenCalledTimes(1);
|
||||
expect(observedContexts.length).toBeGreaterThanOrEqual(2);
|
||||
expect(observedContexts[1].join("\n")).toContain("MID-RUN-COMPACTED-WITH-CONTENT-VARIANT");
|
||||
});
|
||||
|
||||
it("does not compact mid-run outside goal mode when disabled", async () => {
|
||||
const { session } = await createHarness({ "compaction.midTurnEnabled": false });
|
||||
const compactSpy = mockCompaction("SHOULD-NOT-RUN");
|
||||
|
||||
@@ -17,6 +17,7 @@ import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
import * as snapcompact from "@oh-my-pi/snapcompact";
|
||||
|
||||
const HANDOFF_SECRET = "HANDOFF_SECRET_TOKEN_12345";
|
||||
const UNRENDERABLE_SNAPCOMPACT_TEXT = "\uE000\uE001\uE002\uE003\uE004\uE005\uE006\uE007\uE008\uE009";
|
||||
|
||||
describe("AgentSession handoff", () => {
|
||||
// Immutable across the whole file: the model registry's synchronous bundled-model
|
||||
@@ -451,7 +452,11 @@ describe("AgentSession handoff", () => {
|
||||
const fixedPreparation: compactionModule.CompactionPreparation = {
|
||||
firstKeptEntryId: lastEntryId,
|
||||
messagesToSummarize: [
|
||||
{ role: "user", content: [{ type: "text", text: "中文内容".repeat(100) }], timestamp: 1 },
|
||||
{
|
||||
role: "user",
|
||||
content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }],
|
||||
timestamp: 1,
|
||||
},
|
||||
],
|
||||
turnPrefixMessages: [],
|
||||
recentMessages: [],
|
||||
@@ -478,7 +483,11 @@ describe("AgentSession handoff", () => {
|
||||
const fixedPreparation: compactionModule.CompactionPreparation = {
|
||||
firstKeptEntryId: lastEntryId,
|
||||
messagesToSummarize: [
|
||||
{ role: "user", content: [{ type: "text", text: "中文内容".repeat(100) }], timestamp: 1 },
|
||||
{
|
||||
role: "user",
|
||||
content: [{ type: "text", text: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(100) }],
|
||||
timestamp: 1,
|
||||
},
|
||||
],
|
||||
turnPrefixMessages: [],
|
||||
recentMessages: [],
|
||||
|
||||
@@ -774,7 +774,7 @@ describe("AgentSession MCP discovery", () => {
|
||||
settings: Settings.isolated({
|
||||
"mcp.discoveryMode": true,
|
||||
defaultThinkingLevel: "high",
|
||||
serviceTier: "priority",
|
||||
"tier.openai": "priority",
|
||||
}),
|
||||
modelRegistry: {} as never,
|
||||
toolRegistry,
|
||||
@@ -789,10 +789,10 @@ describe("AgentSession MCP discovery", () => {
|
||||
|
||||
expect(session.getSelectedMCPToolNames()).toEqual(["mcp__docs_search"]);
|
||||
sessionManager.appendThinkingLevelChange(ThinkingLevel.High);
|
||||
sessionManager.appendServiceTierChange("flex");
|
||||
sessionManager.appendServiceTierChange({ openai: "flex" });
|
||||
sessionManager.appendMCPToolSelection(["mcp__docs_search"]);
|
||||
expect(sessionManager.buildSessionContext().thinkingLevel).toBe(ThinkingLevel.High);
|
||||
expect(sessionManager.buildSessionContext().serviceTier).toBe("flex");
|
||||
expect(sessionManager.buildSessionContext().serviceTier).toEqual({ openai: "flex" });
|
||||
expect(sessionManager.buildSessionContext().selectedMCPToolNames).toEqual(["mcp__docs_search"]);
|
||||
expect(sessionManager.buildSessionContext().hasPersistedMCPToolSelection).toBe(true);
|
||||
await sessionManager.rewriteEntries();
|
||||
@@ -803,7 +803,7 @@ describe("AgentSession MCP discovery", () => {
|
||||
await session.switchSession(olderSessionFile!);
|
||||
expect(session.sessionFile).toBe(olderSessionFile);
|
||||
expect(session.thinkingLevel).toBe(ThinkingLevel.Medium);
|
||||
expect(session.serviceTier).toBe("priority");
|
||||
expect(session.serviceTierByFamily).toEqual({ openai: "priority" });
|
||||
expect(session.getSelectedMCPToolNames()).toEqual([]);
|
||||
expect(session.getActiveToolNames()).toEqual(["read"]);
|
||||
expect(session.systemPrompt).toEqual(["tools:read"]);
|
||||
@@ -813,7 +813,7 @@ describe("AgentSession MCP discovery", () => {
|
||||
await session.switchSession(originalSessionFile!);
|
||||
expect(session.sessionFile).toBe(originalSessionFile);
|
||||
expect(session.thinkingLevel).toBe(ThinkingLevel.Medium);
|
||||
expect(session.serviceTier).toBe("flex");
|
||||
expect(session.serviceTierByFamily).toEqual({ openai: "flex" });
|
||||
expect(session.getSelectedMCPToolNames()).toEqual(["mcp__docs_search"]);
|
||||
expect(session.getActiveToolNames()).toEqual(["read", "mcp__docs_search"]);
|
||||
expect(session.systemPrompt).toEqual(["tools:read,mcp__docs_search"]);
|
||||
|
||||
@@ -4,6 +4,7 @@ import { scheduler } from "node:timers/promises";
|
||||
import { Agent } from "@oh-my-pi/pi-agent-core";
|
||||
import type { ApiKeyResolveContext, AssistantMessage, ToolCall } from "@oh-my-pi/pi-ai";
|
||||
import { createMockModel } from "@oh-my-pi/pi-ai/providers/mock";
|
||||
import * as aiStream from "@oh-my-pi/pi-ai/stream";
|
||||
import { AssistantMessageEventStream } from "@oh-my-pi/pi-ai/utils/event-stream";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
@@ -54,6 +55,9 @@ describe("AgentSession retry delay cap", () => {
|
||||
beforeEach(async () => {
|
||||
tempDir = TempDir.createSync("@pi-retry-cap-");
|
||||
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
|
||||
// A live env var now overrides a stored static api_key; these tests rotate stored Anthropic
|
||||
// credentials, so neutralize env resolution (ignores every provider's ambient env key).
|
||||
vi.spyOn(aiStream, "getEnvApiKey").mockReturnValue(undefined);
|
||||
authStorage.setRuntimeApiKey("anthropic", "anthropic-test-key");
|
||||
modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
});
|
||||
|
||||
@@ -11,6 +11,8 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
const UNRENDERABLE_SNAPCOMPACT_TEXT = "\uE000\uE001\uE002\uE003\uE004\uE005\uE006\uE007\uE008\uE009";
|
||||
|
||||
interface Harness {
|
||||
session: AgentSession;
|
||||
sessionManager: SessionManager;
|
||||
@@ -47,7 +49,7 @@ async function createHarness(tempDir: TempDir, authStorage: AuthStorage, options
|
||||
const settings = Settings.isolated({
|
||||
"compaction.strategy": "snapcompact",
|
||||
// Force a 1-token recent window so the post-turn cut always splits off the
|
||||
// last turn and summarizes the seeded (CJK) history. With the default
|
||||
// last turn and summarizes the seeded unrenderable history. With the default
|
||||
// 20k window the cut keeps both tiny messages, leaving nothing for
|
||||
// snapcompact's renderability preflight to scan.
|
||||
"compaction.keepRecentTokens": 1,
|
||||
@@ -164,7 +166,7 @@ describe("AgentSession auto-snapcompact local-blocker fallback", () => {
|
||||
seedMessages: [
|
||||
{
|
||||
role: "user",
|
||||
content: "你好,请帮我审查这段代码。它的逻辑似乎有问题,我无法理解为何返回空结果。",
|
||||
content: UNRENDERABLE_SNAPCOMPACT_TEXT.repeat(10),
|
||||
timestamp: Date.now(),
|
||||
},
|
||||
],
|
||||
|
||||
@@ -163,6 +163,7 @@ describe("AgentSession snapcompact frame-budget sizing", () => {
|
||||
const maxFrames = opts?.maxFrames;
|
||||
expect(maxFrames).toBeDefined();
|
||||
expect(maxFrames).toBeLessThan(snapcompact.MAX_FRAMES_DEFAULT);
|
||||
expect(maxFrames).toBeLessThanOrEqual(snapcompact.maxFramesForDataBudget());
|
||||
expect(maxFrames).toBeGreaterThan(0);
|
||||
|
||||
// Verify the FULL projection — base (non-message + kept-recent) +
|
||||
@@ -241,4 +242,41 @@ describe("AgentSession snapcompact frame-budget sizing", () => {
|
||||
// text-only `planArchive` path makes this case recoverable.
|
||||
expect(opts?.maxFrames).toBe(1);
|
||||
});
|
||||
|
||||
it("applies the frame byte cap when the model context window is unknown", async () => {
|
||||
const model = session.model;
|
||||
if (!model) throw new Error("Expected model");
|
||||
await session.dispose();
|
||||
const unknownWindowModel = { ...model, contextWindow: 0 };
|
||||
session = new AgentSession({
|
||||
agent: new Agent({
|
||||
initialState: { model: unknownWindowModel, systemPrompt: ["Test"], tools: [], messages: [] },
|
||||
}),
|
||||
sessionManager,
|
||||
settings: Settings.isolated({
|
||||
"compaction.strategy": "snapcompact",
|
||||
"compaction.autoContinue": false,
|
||||
"compaction.keepRecentTokens": 4000,
|
||||
}),
|
||||
modelRegistry,
|
||||
});
|
||||
|
||||
const branchEntries = sessionManager.getBranch();
|
||||
const lastEntry = branchEntries[branchEntries.length - 1];
|
||||
if (!lastEntry?.id) throw new Error("Expected branch entry with id");
|
||||
const compactSpy = vi.spyOn(snapcompact, "compact").mockResolvedValue({
|
||||
summary: "stubbed snapcompact",
|
||||
shortSummary: "stub",
|
||||
firstKeptEntryId: lastEntry.id,
|
||||
tokensBefore: 100_000,
|
||||
details: { readFiles: [], modifiedFiles: [] },
|
||||
preserveData: {
|
||||
snapcompact: { frames: [], totalChars: 0, truncatedChars: 0 },
|
||||
},
|
||||
});
|
||||
|
||||
await session.compact(undefined, { mode: "snapcompact" });
|
||||
|
||||
expect(compactSpy.mock.calls[0]?.[1]?.maxFrames).toBe(snapcompact.maxFramesForDataBudget());
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
import { afterAll, afterEach, beforeAll, describe, expect, it } from "bun:test";
|
||||
import * as path from "node:path";
|
||||
import { Agent } from "@oh-my-pi/pi-agent-core";
|
||||
import type { Model } from "@oh-my-pi/pi-ai";
|
||||
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
||||
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import type { BuildSessionContextOptions, SessionContext } from "@oh-my-pi/pi-coding-agent/session/session-context";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
/**
|
||||
* Regression for issue #3846: in-TUI `/resume` rebuilt the *previous*
|
||||
* session's display context before switching files. That call expands persisted
|
||||
* snapcompact archives and `openaiRemoteCompaction.replacementHistory` payloads
|
||||
* into messages, which can OOM on huge pre-fix sessions even though the loader
|
||||
* itself streams. The previous context is only needed for same-session reloads
|
||||
* (where `#didSessionMessagesChange` compares against the freshly rebuilt one);
|
||||
* different-session switches MUST skip that work.
|
||||
*/
|
||||
describe("AgentSession.switchSession previous-context build", () => {
|
||||
let sharedDir: TempDir;
|
||||
let authStorage: AuthStorage;
|
||||
let modelRegistry: ModelRegistry;
|
||||
let model: Model;
|
||||
const tempDirs: TempDir[] = [];
|
||||
const sessions: AgentSession[] = [];
|
||||
|
||||
beforeAll(async () => {
|
||||
sharedDir = TempDir.createSync("@pi-switch-prev-ctx-shared-");
|
||||
authStorage = await AuthStorage.create(path.join(sharedDir.path(), "testauth.db"));
|
||||
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
||||
modelRegistry = new ModelRegistry(authStorage);
|
||||
const bundled = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
if (!bundled) throw new Error("Expected built-in anthropic model to exist");
|
||||
model = bundled;
|
||||
});
|
||||
|
||||
afterAll(async () => {
|
||||
authStorage.close();
|
||||
try {
|
||||
await sharedDir.remove();
|
||||
} catch {}
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
while (sessions.length > 0) {
|
||||
await sessions.pop()?.dispose();
|
||||
}
|
||||
for (const dir of tempDirs.splice(0)) {
|
||||
try {
|
||||
await dir.remove();
|
||||
} catch {}
|
||||
}
|
||||
});
|
||||
|
||||
function buildSession(tempDir: TempDir): { session: AgentSession; sessionManager: SessionManager } {
|
||||
const sessionManager = SessionManager.create(tempDir.path(), tempDir.path());
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
});
|
||||
const session = new AgentSession({
|
||||
agent,
|
||||
sessionManager,
|
||||
settings: Settings.isolated({ "compaction.enabled": false }),
|
||||
modelRegistry,
|
||||
});
|
||||
sessions.push(session);
|
||||
return { session, sessionManager };
|
||||
}
|
||||
|
||||
/** Wrap `sessionManager.buildSessionContext` so each call's caller-visible
|
||||
* state (the manager's currently-loaded session file) is recorded in
|
||||
* invocation order. The constructor itself calls `buildSessionContext`
|
||||
* once; spying *after* construction means only switchSession-driven calls
|
||||
* are observed. */
|
||||
function instrumentBuildSessionContext(sessionManager: SessionManager): {
|
||||
calls: Array<{ sessionFile: string | undefined; transcript: boolean | undefined }>;
|
||||
restore: () => void;
|
||||
} {
|
||||
const calls: Array<{ sessionFile: string | undefined; transcript: boolean | undefined }> = [];
|
||||
const original = sessionManager.buildSessionContext.bind(sessionManager);
|
||||
const patched = (options?: BuildSessionContextOptions): SessionContext => {
|
||||
calls.push({ sessionFile: sessionManager.getSessionFile(), transcript: options?.transcript });
|
||||
return original(options);
|
||||
};
|
||||
sessionManager.buildSessionContext = patched as SessionManager["buildSessionContext"];
|
||||
return {
|
||||
calls,
|
||||
restore: () => {
|
||||
sessionManager.buildSessionContext = original;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
it("skips building the previous display context when switching to a different session", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-switch-prev-ctx-different-");
|
||||
tempDirs.push(tempDir);
|
||||
|
||||
const { session, sessionManager } = buildSession(tempDir);
|
||||
sessionManager.appendMessage({ role: "user", content: "previous", timestamp: 1 });
|
||||
await sessionManager.flush();
|
||||
const previousSessionFile = sessionManager.getSessionFile();
|
||||
expect(previousSessionFile).toBeString();
|
||||
|
||||
const otherManager = SessionManager.create(tempDir.path(), tempDir.path());
|
||||
otherManager.appendMessage({ role: "user", content: "target", timestamp: 2 });
|
||||
await otherManager.flush();
|
||||
const targetSessionFile = otherManager.getSessionFile();
|
||||
expect(targetSessionFile).toBeString();
|
||||
expect(targetSessionFile).not.toBe(previousSessionFile);
|
||||
await otherManager.close();
|
||||
|
||||
const { calls, restore } = instrumentBuildSessionContext(sessionManager);
|
||||
try {
|
||||
const switched = await session.switchSession(targetSessionFile!);
|
||||
expect(switched).toBe(true);
|
||||
expect(session.sessionFile).toBe(targetSessionFile);
|
||||
} finally {
|
||||
restore();
|
||||
}
|
||||
|
||||
// The previous session's display context MUST NOT be materialized. Only
|
||||
// the new target context (post-`setSessionFile`) should be built.
|
||||
expect(calls).toEqual([{ sessionFile: targetSessionFile!, transcript: undefined }]);
|
||||
});
|
||||
|
||||
it("builds the previous display context for same-session reloads", async () => {
|
||||
const tempDir = TempDir.createSync("@pi-switch-prev-ctx-reload-");
|
||||
tempDirs.push(tempDir);
|
||||
|
||||
const { session, sessionManager } = buildSession(tempDir);
|
||||
sessionManager.appendMessage({ role: "user", content: "current", timestamp: 1 });
|
||||
await sessionManager.flush();
|
||||
const sessionFile = sessionManager.getSessionFile();
|
||||
expect(sessionFile).toBeString();
|
||||
|
||||
const { calls, restore } = instrumentBuildSessionContext(sessionManager);
|
||||
try {
|
||||
const switched = await session.switchSession(sessionFile!);
|
||||
expect(switched).toBe(true);
|
||||
expect(session.sessionFile).toBe(sessionFile);
|
||||
} finally {
|
||||
restore();
|
||||
}
|
||||
|
||||
// Same-session reload must snapshot the pre-reload context so
|
||||
// `#didSessionMessagesChange` can detect rollback edits.
|
||||
expect(calls).toEqual([
|
||||
{ sessionFile: sessionFile!, transcript: undefined },
|
||||
{ sessionFile: sessionFile!, transcript: undefined },
|
||||
]);
|
||||
});
|
||||
});
|
||||
@@ -19,6 +19,7 @@ import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { AgentSession, type AgentSessionEvent } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { type CustomMessage, convertToLlm } from "@oh-my-pi/pi-coding-agent/session/messages";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
@@ -241,4 +242,136 @@ describe("AgentSession thinking-loop retry", () => {
|
||||
expect(assistants).toHaveLength(1);
|
||||
expect(assistants[0].content).toEqual([{ type: "text", text: "Recovered after retry." }]);
|
||||
});
|
||||
|
||||
it("injects a redirect notice into the retried turn after a thinking loop", async () => {
|
||||
const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
|
||||
const modelRegistry = new ModelRegistry(authStorage);
|
||||
const calls: string[] = [];
|
||||
const contexts: Context[] = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: requestedModel => `${requestedModel.provider}-test-key`,
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (requestedModel, context, _options?: SimpleStreamOptions) => {
|
||||
calls.push(`${requestedModel.provider}/${requestedModel.id}`);
|
||||
contexts.push(context);
|
||||
return calls.length === 1 ? errorIdOnlyThinkingLoopStream(requestedModel) : successStream(requestedModel);
|
||||
},
|
||||
});
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
"retry.enabled": true,
|
||||
"retry.baseDelayMs": 0,
|
||||
"retry.maxDelayMs": 5_000,
|
||||
"retry.maxRetries": 1,
|
||||
"retry.modelFallback": false,
|
||||
"todo.enabled": false,
|
||||
"model.loopGuard.enabled": true,
|
||||
});
|
||||
settings.setModelRole("default", `${model.provider}/${model.id}`);
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
modelRegistry,
|
||||
});
|
||||
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
|
||||
|
||||
await session.prompt("Trigger redirect injection after thinking loop");
|
||||
await session.waitForIdle();
|
||||
const retryContext = contexts[1];
|
||||
const extractText = (content: string | Array<{ type: string; text?: string }>): string =>
|
||||
typeof content === "string"
|
||||
? content
|
||||
: content.map(part => (part.type === "text" ? (part.text ?? "") : "")).join("");
|
||||
const redirectDevMsgs = retryContext.messages.filter(
|
||||
message => message.role === "developer" && extractText(message.content).includes("thinking_loop_detected"),
|
||||
);
|
||||
expect(redirectDevMsgs).toHaveLength(1);
|
||||
|
||||
const redirects = session.agent.state.messages.filter(
|
||||
(message): message is CustomMessage =>
|
||||
message.role === "custom" && message.customType === "thinking-loop-redirect",
|
||||
);
|
||||
expect(redirects).toHaveLength(1);
|
||||
expect(redirects[0].display).toBe(false);
|
||||
expect(redirects[0].attribution).toBe("agent");
|
||||
expect(typeof redirects[0].content).toBe("string");
|
||||
expect(redirects[0].content).toContain("thinking_loop_detected");
|
||||
|
||||
const assistants = session.agent.state.messages.filter(
|
||||
(message): message is AssistantMessage => message.role === "assistant",
|
||||
);
|
||||
expect(assistants).toHaveLength(1);
|
||||
expect(assistants[0].content).toEqual([{ type: "text", text: "Recovered after retry." }]);
|
||||
});
|
||||
|
||||
it("injects a redirect notice on each consecutive thinking-loop retry", async () => {
|
||||
const model = createMockModel({ provider: "openrouter", id: "google/gemini-3.5-flash" }).model;
|
||||
const modelRegistry = new ModelRegistry(authStorage);
|
||||
const calls: string[] = [];
|
||||
const contexts: Context[] = [];
|
||||
const agent = new Agent({
|
||||
getApiKey: requestedModel => `${requestedModel.provider}-test-key`,
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
convertToLlm,
|
||||
streamFn: (requestedModel, context, _options?: SimpleStreamOptions) => {
|
||||
calls.push(`${requestedModel.provider}/${requestedModel.id}`);
|
||||
contexts.push(context);
|
||||
return calls.length <= 2 ? errorIdOnlyThinkingLoopStream(requestedModel) : successStream(requestedModel);
|
||||
},
|
||||
});
|
||||
const settings = Settings.isolated({
|
||||
"compaction.enabled": false,
|
||||
"retry.enabled": true,
|
||||
"retry.baseDelayMs": 0,
|
||||
"retry.maxDelayMs": 5_000,
|
||||
"retry.maxRetries": 2,
|
||||
"retry.modelFallback": false,
|
||||
"todo.enabled": false,
|
||||
"model.loopGuard.enabled": true,
|
||||
});
|
||||
settings.setModelRole("default", `${model.provider}/${model.id}`);
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
modelRegistry,
|
||||
});
|
||||
vi.spyOn(scheduler, "wait").mockResolvedValue(undefined);
|
||||
|
||||
await session.prompt("Trigger redirect injection after two thinking loops");
|
||||
await session.waitForIdle();
|
||||
|
||||
expect(calls).toHaveLength(3);
|
||||
const redirects = session.agent.state.messages.filter(
|
||||
(message): message is CustomMessage =>
|
||||
message.role === "custom" && message.customType === "thinking-loop-redirect",
|
||||
);
|
||||
expect(redirects).toHaveLength(2);
|
||||
const extractText = (content: string | Array<{ type: string; text?: string }>): string =>
|
||||
typeof content === "string"
|
||||
? content
|
||||
: content.map(part => (part.type === "text" ? (part.text ?? "") : "")).join("");
|
||||
const thirdAttemptRedirectDevMsgs = contexts[2].messages.filter(
|
||||
message => message.role === "developer" && extractText(message.content).includes("thinking_loop_detected"),
|
||||
);
|
||||
expect(thirdAttemptRedirectDevMsgs).toHaveLength(2);
|
||||
|
||||
const assistants = session.agent.state.messages.filter(
|
||||
(message): message is AssistantMessage => message.role === "assistant",
|
||||
);
|
||||
expect(assistants).toHaveLength(1);
|
||||
expect(assistants[0].content).toEqual([{ type: "text", text: "Recovered after retry." }]);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -173,13 +173,16 @@ describe("bench empty-output guard", () => {
|
||||
|
||||
function settingsStub(serviceTier: string | undefined): Settings | undefined {
|
||||
if (serviceTier === undefined) return undefined;
|
||||
return { get: (key: string) => (key === "serviceTier" ? serviceTier : undefined) } as unknown as Settings;
|
||||
return {
|
||||
get: (key: string) =>
|
||||
key === "tier.openai" ? serviceTier : key === "tier.anthropic" || key === "tier.google" ? "none" : undefined,
|
||||
} as unknown as Settings;
|
||||
}
|
||||
|
||||
async function captureServiceTier(opts: {
|
||||
flag?: string;
|
||||
setting?: string;
|
||||
}): Promise<{ wire: SimpleStreamOptions["serviceTier"]; summary: BenchSummary["serviceTier"] }> {
|
||||
}): Promise<{ wire: SimpleStreamOptions["serviceTier"]; summary: BenchSummary["serviceTierByFamily"] }> {
|
||||
const registry = fakeRegistry({ models: [fakeModel("openai-codex", "gpt-5.5")], authedProviders: ["openai-codex"] });
|
||||
let captured: SimpleStreamOptions | undefined;
|
||||
const summary = await runBenchCommand(
|
||||
@@ -205,7 +208,7 @@ async function captureServiceTier(opts: {
|
||||
stdoutIsTTY: false,
|
||||
},
|
||||
);
|
||||
return { wire: captured?.serviceTier, summary: summary.serviceTier };
|
||||
return { wire: captured?.serviceTier, summary: summary.serviceTierByFamily };
|
||||
}
|
||||
|
||||
describe("bench provider session state and websocket preference", () => {
|
||||
@@ -245,24 +248,24 @@ describe("bench service tier", () => {
|
||||
it("sends the configured serviceTier setting when no flag is passed", async () => {
|
||||
const { wire, summary } = await captureServiceTier({ setting: "flex" });
|
||||
expect(wire).toBe("flex");
|
||||
expect(summary).toBe("flex");
|
||||
expect(summary).toEqual({ openai: "flex" });
|
||||
});
|
||||
|
||||
it("lets an explicit --service-tier override the configured setting", async () => {
|
||||
const { wire, summary } = await captureServiceTier({ flag: "priority", setting: "flex" });
|
||||
expect(wire).toBe("priority");
|
||||
expect(summary).toBe("priority");
|
||||
expect(summary).toEqual({ openai: "priority", anthropic: "priority", google: "priority" });
|
||||
});
|
||||
|
||||
it("omits service_tier when the setting is none and no flag is passed", async () => {
|
||||
const { wire, summary } = await captureServiceTier({ setting: "none" });
|
||||
expect(wire).toBeUndefined();
|
||||
expect(summary).toBeUndefined();
|
||||
expect(summary).toEqual({});
|
||||
});
|
||||
|
||||
it("omits service_tier when neither flag nor settings are present", async () => {
|
||||
const { wire, summary } = await captureServiceTier({});
|
||||
expect(wire).toBeUndefined();
|
||||
expect(summary).toBeUndefined();
|
||||
expect(summary).toEqual({});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -111,6 +111,56 @@ describe("hashline executor", () => {
|
||||
});
|
||||
});
|
||||
|
||||
it("preserves UTF-8 BOM bytes when hashline edits decoded text", async () => {
|
||||
await withTempDir(async tempDir => {
|
||||
const filePath = path.join(tempDir, "Program.cs");
|
||||
const source = "using A;\n";
|
||||
await Bun.write(filePath, new Uint8Array([0xef, 0xbb, 0xbf, ...new TextEncoder().encode(source)]));
|
||||
const session = makeHashlineSession(tempDir);
|
||||
const sourceTag = recordFullSnapshot(getFileReadCache(session), filePath, source);
|
||||
const input = `${header("Program.cs", sourceTag)}\n${sameLineRange(tag(1, source))}\n${repl("using B;")}\n`;
|
||||
|
||||
await executeHashlineSingle(hashlineExecuteOptions(tempDir, input, undefined, session));
|
||||
|
||||
const bytes = await fs.readFile(filePath);
|
||||
expect(Array.from(bytes.subarray(0, 3))).toEqual([0xef, 0xbb, 0xbf]);
|
||||
expect(new TextDecoder().decode(bytes.subarray(3))).toBe("using B;\n");
|
||||
});
|
||||
});
|
||||
|
||||
it("edits BOM-prefixed notebooks through the virtual cell text", async () => {
|
||||
await withTempDir(async tempDir => {
|
||||
const filePath = path.join(tempDir, "notebook.ipynb");
|
||||
const notebook = {
|
||||
cells: [
|
||||
{
|
||||
cell_type: "markdown",
|
||||
metadata: { keep: true },
|
||||
source: ["# Title\n"],
|
||||
},
|
||||
],
|
||||
metadata: {},
|
||||
nbformat: 4,
|
||||
nbformat_minor: 5,
|
||||
};
|
||||
await Bun.write(
|
||||
filePath,
|
||||
new Uint8Array([0xef, 0xbb, 0xbf, ...new TextEncoder().encode(JSON.stringify(notebook))]),
|
||||
);
|
||||
const session = makeHashlineSession(tempDir);
|
||||
const editableText = "# %% [markdown] cell:0\n# Title\n";
|
||||
const sourceTag = recordFullSnapshot(getFileReadCache(session), filePath, editableText);
|
||||
const input = `${header("notebook.ipynb", sourceTag)}\n${sameLineRange(tag(2, "# Title"))}\n${repl("# Updated")}\n`;
|
||||
|
||||
await executeHashlineSingle(hashlineExecuteOptions(tempDir, input, undefined, session));
|
||||
|
||||
const updated = await Bun.file(filePath).json();
|
||||
expect(updated.cells).toHaveLength(1);
|
||||
expect(updated.cells[0].source).toEqual(["# Updated\n"]);
|
||||
expect(updated.cells[0].metadata).toEqual({ keep: true });
|
||||
});
|
||||
});
|
||||
|
||||
it("emits an actionable no-op diagnostic when the payload matches the file byte-for-byte", async () => {
|
||||
await withTempDir(async tempDir => {
|
||||
const filePath = path.join(tempDir, "a.ts");
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "bun:test";
|
||||
import { CustomEditor } from "@oh-my-pi/pi-coding-agent/modes/components/custom-editor";
|
||||
import { getEditorTheme, initTheme } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
|
||||
import { StdinBuffer } from "@oh-my-pi/pi-tui/stdin-buffer";
|
||||
|
||||
/**
|
||||
* Regression for #3857.
|
||||
*
|
||||
* A fast double-Esc lands as one `"\x1b\x1b"` chunk on stdin. Before the fix,
|
||||
* `StdinBuffer` held it as the buffered remainder, then timer-flushed it as
|
||||
* one sequence. `parseKey("\x1b\x1b")` returns `undefined`, so
|
||||
* `CustomEditor.handleInput` fell through to the base editor and never fired
|
||||
* the configured `onEscape` — breaking the double-escape gesture.
|
||||
*
|
||||
* The fix splits a bare `"\x1b\x1b"` into two ESC events only when no follower
|
||||
* arrives in the disambiguation window. If a follower arrives, the second ESC
|
||||
* remains attached to that follower so legacy Alt chords survive.
|
||||
*/
|
||||
describe("buffered double-Esc reaches CustomEditor.onEscape", () => {
|
||||
beforeAll(async () => {
|
||||
await initTheme();
|
||||
});
|
||||
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.useRealTimers();
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("fires onEscape twice when a fast double-Esc arrives as one buffered chunk", () => {
|
||||
const editor = new CustomEditor(getEditorTheme());
|
||||
const onEscape = vi.fn();
|
||||
editor.onEscape = onEscape;
|
||||
|
||||
const buf = new StdinBuffer({ timeout: 5, partialHoldTimeout: 5 });
|
||||
buf.on("data", chunk => editor.handleInput(chunk));
|
||||
|
||||
buf.process("\x1b\x1b");
|
||||
// Drain the flush timer chain (main timeout + zero-delay deferral).
|
||||
vi.runAllTimers();
|
||||
|
||||
expect(onEscape).toHaveBeenCalledTimes(2);
|
||||
buf.destroy();
|
||||
});
|
||||
|
||||
it("preserves a legacy Alt chord batched after a bare ESC", () => {
|
||||
const editor = new CustomEditor(getEditorTheme());
|
||||
const onEscape = vi.fn();
|
||||
editor.onEscape = onEscape;
|
||||
editor.setText("foo bar");
|
||||
|
||||
const buf = new StdinBuffer({ timeout: 5, partialHoldTimeout: 5 });
|
||||
buf.on("data", chunk => editor.handleInput(chunk));
|
||||
|
||||
buf.process("\x1b\x1b\x7f");
|
||||
vi.runAllTimers();
|
||||
|
||||
expect(onEscape).toHaveBeenCalledTimes(1);
|
||||
expect(editor.getText()).toBe("foo ");
|
||||
buf.destroy();
|
||||
});
|
||||
|
||||
it("does not split a meta-CSI arrow into two ESC events", () => {
|
||||
const editor = new CustomEditor(getEditorTheme());
|
||||
const onEscape = vi.fn();
|
||||
editor.onEscape = onEscape;
|
||||
|
||||
const buf = new StdinBuffer({ timeout: 5, partialHoldTimeout: 5 });
|
||||
buf.on("data", chunk => editor.handleInput(chunk));
|
||||
|
||||
buf.process("\x1b\x1b[A");
|
||||
vi.runAllTimers();
|
||||
|
||||
// alt+up is its own keypress and must never look like two ESC keys.
|
||||
expect(onEscape).not.toHaveBeenCalled();
|
||||
buf.destroy();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,24 @@
|
||||
import { describe, expect, it } from "bun:test";
|
||||
import { startCpuProfile } from "@oh-my-pi/pi-coding-agent/debug/profiler";
|
||||
|
||||
describe("startCpuProfile", () => {
|
||||
// Regression: `node:v8` `setFlagsFromString` throws on Bun
|
||||
// (oven-sh/bun#1702). The profiler used to call it unconditionally and
|
||||
// crash before connecting the inspector session. Running this test under
|
||||
// Bun guarantees the guard is in place — without it the call below would
|
||||
// reject with "node:v8 setFlagsFromString is not yet implemented in Bun".
|
||||
it("starts and stops successfully even when v8.setFlagsFromString is unavailable", async () => {
|
||||
const session = await startCpuProfile();
|
||||
// Run a tiny bit of work so the profile has at least one sample.
|
||||
let acc = 0;
|
||||
for (let i = 0; i < 10_000; i++) acc += i;
|
||||
expect(acc).toBeGreaterThan(0);
|
||||
|
||||
const profile = await session.stop();
|
||||
const parsed = JSON.parse(profile.data) as { nodes: unknown[] };
|
||||
expect(Array.isArray(parsed.nodes)).toBe(true);
|
||||
expect(parsed.nodes.length).toBeGreaterThan(0);
|
||||
expect(typeof profile.markdown).toBe("string");
|
||||
expect(profile.markdown.length).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
@@ -146,6 +146,83 @@ describe("builtin-defaults rule provider", () => {
|
||||
}),
|
||||
).toEqual([]);
|
||||
});
|
||||
it("go-new-expr matches value→pointer helpers (named + generic) but not real functions, only on *.go", async () => {
|
||||
const rules = await loadBuiltinRules();
|
||||
const rule = rules.find(r => r.name === "go-new-expr");
|
||||
if (!rule) throw new Error("go-new-expr rule missing");
|
||||
const manager = new TtsrManager();
|
||||
expect(manager.addRule(rule)).toBe(true);
|
||||
const ctx: TtsrMatchContext = { source: "tool", toolName: "edit", filePaths: ["pkg/foo.go"] };
|
||||
|
||||
const hits = [
|
||||
"package p\nfunc boolPtr(v bool) *bool { return &v }",
|
||||
"package p\nfunc Ptr[T any](v T) *T { return &v }",
|
||||
];
|
||||
for (const snippet of hits) {
|
||||
manager.resetBuffer();
|
||||
expect(
|
||||
(await manager.checkAstSnapshot(snippet, ctx)).map(m => m.name),
|
||||
snippet,
|
||||
).toEqual(["go-new-expr"]);
|
||||
}
|
||||
|
||||
const misses = [
|
||||
"package p\nfunc add(a int, b int) *int { return &a }",
|
||||
"package p\nfunc (s *S) Get() *int { return &s.x }",
|
||||
];
|
||||
for (const snippet of misses) {
|
||||
manager.resetBuffer();
|
||||
expect(await manager.checkAstSnapshot(snippet, ctx), snippet).toEqual([]);
|
||||
}
|
||||
|
||||
// AST conditions never reach a non-go path.
|
||||
manager.resetBuffer();
|
||||
expect(
|
||||
await manager.checkAstSnapshot(hits[0], { source: "tool", toolName: "edit", filePaths: ["pkg/foo.ts"] }),
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
it("go-bench-loop fires on a *testing.B b.N loop but not an ordinary .N counter", async () => {
|
||||
const rules = await loadBuiltinRules();
|
||||
const rule = rules.find(r => r.name === "go-bench-loop");
|
||||
if (!rule) throw new Error("go-bench-loop rule missing");
|
||||
const manager = new TtsrManager();
|
||||
expect(manager.addRule(rule)).toBe(true);
|
||||
const ctx: TtsrMatchContext = { source: "tool", toolName: "edit", filePaths: ["pkg/foo_test.go"] };
|
||||
|
||||
const bench =
|
||||
"package p\nfunc BenchmarkX(b *testing.B) {\n\tsetup()\n\tfor i := 0; i < b.N; i++ {\n\t\twork()\n\t}\n}";
|
||||
manager.resetBuffer();
|
||||
expect((await manager.checkAstSnapshot(bench, ctx)).map(m => m.name)).toEqual(["go-bench-loop"]);
|
||||
|
||||
// A `.N` selector on something that is not the benchmark receiver must not fire.
|
||||
const helper =
|
||||
"package p\nfunc TestThing(t *testing.T) {\n\treq := build()\n\tfor i := 0; i < req.N; i++ {\n\t\twork()\n\t}\n}";
|
||||
manager.resetBuffer();
|
||||
expect(await manager.checkAstSnapshot(helper, ctx)).toEqual([]);
|
||||
});
|
||||
|
||||
it("go-range-int fires only on *.go, never on a same-named non-go path", async () => {
|
||||
const rules = await loadBuiltinRules();
|
||||
const rule = rules.find(r => r.name === "go-range-int");
|
||||
if (!rule) throw new Error("go-range-int rule missing");
|
||||
const manager = new TtsrManager();
|
||||
expect(manager.addRule(rule)).toBe(true);
|
||||
|
||||
const loop = "package p\nfunc f(n int) {\n\tfor i := 0; i < n; i++ {\n\t\tuse(i)\n\t}\n}";
|
||||
manager.resetBuffer();
|
||||
expect(
|
||||
(await manager.checkAstSnapshot(loop, { source: "tool", toolName: "edit", filePaths: ["pkg/foo.go"] })).map(
|
||||
m => m.name,
|
||||
),
|
||||
).toEqual(["go-range-int"]);
|
||||
// A step-2 loop is not equivalent to range-over-int and must not fire.
|
||||
const step2 = "package p\nfunc f(n int) {\n\tfor i := 0; i < n; i += 2 {\n\t\tuse(i)\n\t}\n}";
|
||||
manager.resetBuffer();
|
||||
expect(
|
||||
await manager.checkAstSnapshot(step2, { source: "tool", toolName: "edit", filePaths: ["pkg/foo.go"] }),
|
||||
).toEqual([]);
|
||||
});
|
||||
|
||||
it("is the lowest-priority rule provider so user/project rules override defaults", () => {
|
||||
const { cap, provider } = ruleProvider();
|
||||
|
||||
@@ -299,6 +299,47 @@ describe("EventController working loader reconciliation", () => {
|
||||
expect(ctx.ensureLoadingAnimation).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("self-heals missing working loader when a task subagent finishes mid-turn (#3858)", async () => {
|
||||
// `task` subagents run inside the parent's streaming turn. While the task is
|
||||
// running a transient overlay (auto-compaction / auto-retry) can drop the
|
||||
// working loader by clearing the status container, and the overlay's end
|
||||
// handler is the only restorer keyed off the missing loader. If the task
|
||||
// finishes between the overlay's start and end (or any other branch where
|
||||
// the loader was nulled without a follow-up overlay-end), `tool_execution_end`
|
||||
// is the next streaming event that lands and must heal the loader, mirroring
|
||||
// the `tool_execution_update` reconciler. Without this the spinner stays
|
||||
// gone for the remainder of the parent turn even though the agent keeps
|
||||
// streaming (the user-visible regression in #3858).
|
||||
const { controller, ctx } = createFixture();
|
||||
(ctx.viewSession as unknown as { isStreaming: boolean }).isStreaming = true;
|
||||
|
||||
await controller.handleEvent({
|
||||
type: "tool_execution_end",
|
||||
toolCallId: "task-1",
|
||||
toolName: "task",
|
||||
isError: false,
|
||||
result: { content: [{ type: "text", text: "ok" }], details: {} },
|
||||
} as Extract<AgentSessionEvent, { type: "tool_execution_end" }>);
|
||||
|
||||
expect(ctx.ensureLoadingAnimation).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it("does not restore the working loader while an overlay loader (auto-retry) owns the status container at tool_execution_end", async () => {
|
||||
const { controller, ctx } = createFixture();
|
||||
ctx.retryLoader = { stop: vi.fn() } as unknown as InteractiveModeContext["retryLoader"];
|
||||
(ctx.viewSession as unknown as { isStreaming: boolean }).isStreaming = true;
|
||||
|
||||
await controller.handleEvent({
|
||||
type: "tool_execution_end",
|
||||
toolCallId: "task-2",
|
||||
toolName: "task",
|
||||
isError: false,
|
||||
result: { content: [{ type: "text", text: "ok" }], details: {} },
|
||||
} as Extract<AgentSessionEvent, { type: "tool_execution_end" }>);
|
||||
|
||||
expect(ctx.ensureLoadingAnimation).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it("keeps transient retry status exclusive while a retry loader is visible", async () => {
|
||||
const { controller, ctx } = createFixture();
|
||||
ctx.retryLoader = { stop: vi.fn() } as unknown as InteractiveModeContext["retryLoader"];
|
||||
|
||||
@@ -21,6 +21,11 @@ function createContext() {
|
||||
updateEditorTopBorder: vi.fn(),
|
||||
clearPinnedError: vi.fn(),
|
||||
ensureLoadingAnimation: vi.fn(),
|
||||
// `viewSession.isStreaming` is read by `#ensureWorkingLoaderWhileStreaming`,
|
||||
// which runs at the top of `tool_execution_end` (and other streaming-event
|
||||
// handlers). Leaving it false matches the implicit assumption in this
|
||||
// fixture: the todo HUD lifecycle is independent of the working loader.
|
||||
viewSession: { isStreaming: false },
|
||||
todoReminderContainer,
|
||||
setTodos: vi.fn(),
|
||||
present,
|
||||
|
||||
@@ -0,0 +1,310 @@
|
||||
/**
|
||||
* Regression guard for issue #3827.
|
||||
*
|
||||
* `/mcp list` and the `/extensions` dashboard MUST agree on whether a given MCP
|
||||
* server is enabled or disabled. The two read paths historically diverged: the
|
||||
* dashboard's `loadAllExtensions` only consulted the dashboard-private
|
||||
* `disabledExtensions` settings array, while `/mcp list` (and the MCP runtime
|
||||
* itself) honored both the per-server `enabled` flag in `mcp.json` and the
|
||||
* user-level `disabledServers` denylist.
|
||||
*
|
||||
* The fixtures below cover both inputs and the round-trip helper the
|
||||
* dashboard's MCP toggle uses.
|
||||
*/
|
||||
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
|
||||
import * as fs from "node:fs/promises";
|
||||
import * as os from "node:os";
|
||||
import * as path from "node:path";
|
||||
import { resetSettingsForTest, Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { initializeWithSettings, reset as resetDiscoveryCache } from "@oh-my-pi/pi-coding-agent/discovery";
|
||||
import { readMCPConfigFile, setMcpServerEnabled, setServerDisabled } from "@oh-my-pi/pi-coding-agent/mcp/config-writer";
|
||||
import { loadAllExtensions } from "@oh-my-pi/pi-coding-agent/modes/components/extensions/state-manager";
|
||||
import { __resetDirsFromEnvForTests, getMCPConfigPath, removeWithRetries, setAgentDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
describe("loadAllExtensions MCP parity with /mcp list (issue #3827)", () => {
|
||||
let projectDir = "";
|
||||
let userAgentDir = "";
|
||||
|
||||
beforeEach(async () => {
|
||||
resetSettingsForTest();
|
||||
projectDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-3827-project-"));
|
||||
userAgentDir = await fs.mkdtemp(path.join(os.tmpdir(), "omp-3827-user-"));
|
||||
|
||||
// Redirect user-scoped mcp.json (resolved via getAgentDir() at the call
|
||||
// site) into the per-test temp directory so neither the discovery loader
|
||||
// nor the denylist reader touches the real user profile.
|
||||
setAgentDir(userAgentDir);
|
||||
|
||||
await fs.mkdir(path.join(projectDir, ".omp"), { recursive: true });
|
||||
await fs.writeFile(
|
||||
path.join(projectDir, ".omp", "mcp.json"),
|
||||
JSON.stringify({
|
||||
mcpServers: {
|
||||
"denylisted-server": { command: "echo", args: ["denylisted"] },
|
||||
"flag-disabled-server": { command: "echo", args: ["flag"], enabled: false },
|
||||
"active-server": { command: "echo", args: ["active"] },
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
// User-level mcp.json carries the denylist; this is what `/mcp disable`
|
||||
// writes through setServerDisabled().
|
||||
await fs.writeFile(
|
||||
path.join(userAgentDir, "mcp.json"),
|
||||
JSON.stringify({
|
||||
mcpServers: {},
|
||||
disabledServers: ["denylisted-server"],
|
||||
}),
|
||||
);
|
||||
|
||||
const settings = await Settings.init({ inMemory: true, cwd: projectDir });
|
||||
initializeWithSettings(settings);
|
||||
});
|
||||
|
||||
afterEach(async () => {
|
||||
resetSettingsForTest();
|
||||
__resetDirsFromEnvForTests();
|
||||
await removeWithRetries(projectDir);
|
||||
await removeWithRetries(userAgentDir);
|
||||
});
|
||||
|
||||
test("treats a server in user-level disabledServers as disabled (matches /mcp list)", async () => {
|
||||
const extensions = await loadAllExtensions(projectDir, []);
|
||||
const denylisted = extensions.find(e => e.id === "mcp:denylisted-server");
|
||||
expect(denylisted).toBeDefined();
|
||||
expect(denylisted!.state).toBe("disabled");
|
||||
expect(denylisted!.disabledReason).toBe("item-disabled");
|
||||
});
|
||||
|
||||
test("treats a server with enabled:false as disabled (matches /mcp list)", async () => {
|
||||
const extensions = await loadAllExtensions(projectDir, []);
|
||||
const flagDisabled = extensions.find(e => e.id === "mcp:flag-disabled-server");
|
||||
expect(flagDisabled).toBeDefined();
|
||||
expect(flagDisabled!.state).toBe("disabled");
|
||||
expect(flagDisabled!.disabledReason).toBe("item-disabled");
|
||||
});
|
||||
|
||||
test("leaves untouched servers active", async () => {
|
||||
const extensions = await loadAllExtensions(projectDir, []);
|
||||
const active = extensions.find(e => e.id === "mcp:active-server");
|
||||
expect(active).toBeDefined();
|
||||
expect(active!.state).toBe("active");
|
||||
expect(active!.disabledReason).toBeUndefined();
|
||||
});
|
||||
|
||||
test("setServerDisabled round-trips through the dashboard view", async () => {
|
||||
// Re-enable `denylisted-server` through the canonical writer the
|
||||
// dashboard's MCP toggle now calls. The dashboard view MUST flip to
|
||||
// active on the next load.
|
||||
await setServerDisabled(getMCPConfigPath("user", projectDir), "denylisted-server", false);
|
||||
const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:denylisted-server");
|
||||
expect(reenabled).toBeDefined();
|
||||
expect(reenabled!.state).toBe("active");
|
||||
|
||||
// The inverse path: disabling `active-server` via the writer flips the
|
||||
// dashboard view to disabled.
|
||||
await setServerDisabled(getMCPConfigPath("user", projectDir), "active-server", true);
|
||||
const disabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:active-server");
|
||||
expect(disabled).toBeDefined();
|
||||
expect(disabled!.state).toBe("disabled");
|
||||
expect(disabled!.disabledReason).toBe("item-disabled");
|
||||
});
|
||||
|
||||
test("dashboard re-enable flips enabled:false in mcp.json (PR #3829 review)", async () => {
|
||||
// The bug: when a server has `enabled: false` in mcp.json, the dashboard
|
||||
// toggle previously only removed it from the user-level denylist, so
|
||||
// state-manager's `server.enabled === false` check kept it disabled.
|
||||
// setMcpServerEnabled MUST overwrite the per-server flag.
|
||||
const projectMcpPath = path.join(projectDir, ".omp", "mcp.json");
|
||||
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
name: "flag-disabled-server",
|
||||
enabled: true,
|
||||
});
|
||||
|
||||
const projectConfig = await readMCPConfigFile(projectMcpPath);
|
||||
expect(projectConfig.mcpServers?.["flag-disabled-server"]?.enabled).toBe(true);
|
||||
|
||||
const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:flag-disabled-server");
|
||||
expect(reenabled).toBeDefined();
|
||||
expect(reenabled!.state).toBe("active");
|
||||
});
|
||||
|
||||
test("dashboard re-enable also clears a stale denylist entry on a config-resident server", async () => {
|
||||
// Manually disable `active-server` via BOTH the per-server flag and the
|
||||
// denylist, simulating a server that's been toggled off multiple ways.
|
||||
const projectMcpPath = path.join(projectDir, ".omp", "mcp.json");
|
||||
const initial = await readMCPConfigFile(projectMcpPath);
|
||||
await Bun.write(
|
||||
projectMcpPath,
|
||||
JSON.stringify({
|
||||
...initial,
|
||||
mcpServers: {
|
||||
...initial.mcpServers,
|
||||
"active-server": { ...initial.mcpServers!["active-server"], enabled: false },
|
||||
},
|
||||
}),
|
||||
);
|
||||
await setServerDisabled(getMCPConfigPath("user", projectDir), "active-server", true);
|
||||
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
name: "active-server",
|
||||
enabled: true,
|
||||
});
|
||||
|
||||
const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir));
|
||||
expect(userConfig.disabledServers ?? []).not.toContain("active-server");
|
||||
|
||||
const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:active-server");
|
||||
expect(reenabled).toBeDefined();
|
||||
expect(reenabled!.state).toBe("active");
|
||||
});
|
||||
|
||||
test("dashboard disable on a config-resident server writes enabled:false (not denylist)", async () => {
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
name: "active-server",
|
||||
enabled: false,
|
||||
});
|
||||
|
||||
const projectConfig = await readMCPConfigFile(path.join(projectDir, ".omp", "mcp.json"));
|
||||
expect(projectConfig.mcpServers?.["active-server"]?.enabled).toBe(false);
|
||||
|
||||
// The denylist is reserved for discovered (config-less) servers; a
|
||||
// config-resident server's `enabled: false` flag is the canonical signal.
|
||||
const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir));
|
||||
expect(userConfig.disabledServers ?? []).not.toContain("active-server");
|
||||
|
||||
const disabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:active-server");
|
||||
expect(disabled).toBeDefined();
|
||||
expect(disabled!.state).toBe("disabled");
|
||||
});
|
||||
|
||||
test("dashboard re-enable updates the row's non-primary source mcp.json before denylisting", async () => {
|
||||
const alternatePath = path.join(projectDir, ".omp", ".mcp.json");
|
||||
await Bun.write(
|
||||
alternatePath,
|
||||
JSON.stringify({
|
||||
mcpServers: {
|
||||
"alternate-server": { command: "echo", args: ["alternate"], enabled: false },
|
||||
},
|
||||
}),
|
||||
);
|
||||
|
||||
const disabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:alternate-server");
|
||||
expect(disabled).toBeDefined();
|
||||
expect(disabled!.state).toBe("disabled");
|
||||
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
sourcePath: alternatePath,
|
||||
name: "alternate-server",
|
||||
enabled: true,
|
||||
});
|
||||
|
||||
const alternateConfig = await readMCPConfigFile(alternatePath);
|
||||
expect(alternateConfig.mcpServers?.["alternate-server"]?.enabled).toBe(true);
|
||||
|
||||
const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir));
|
||||
expect(userConfig.disabledServers ?? []).not.toContain("alternate-server");
|
||||
|
||||
const reenabled = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:alternate-server");
|
||||
expect(reenabled).toBeDefined();
|
||||
expect(reenabled!.state).toBe("active");
|
||||
});
|
||||
test("dashboard re-enable force-enables a tool-owned source (opencode.json) via enabledServers", async () => {
|
||||
// OpenCode is a non-writable source: the dashboard must NOT mutate
|
||||
// opencode.json, but the user-level enabledServers allowlist still has
|
||||
// to flip the row active. Modeled after the codex review on PR #3829.
|
||||
const opencodePath = path.join(projectDir, "opencode.json");
|
||||
await Bun.write(
|
||||
opencodePath,
|
||||
JSON.stringify({
|
||||
mcp: {
|
||||
"opencode-server": {
|
||||
type: "local",
|
||||
command: ["echo", "opencode"],
|
||||
enabled: false,
|
||||
},
|
||||
},
|
||||
}),
|
||||
);
|
||||
// beforeEach's Settings.init() already cached an absent opencode.json
|
||||
// for this projectDir, so drop the capability fs cache before the first
|
||||
// dashboard load picks the file up.
|
||||
resetDiscoveryCache();
|
||||
|
||||
const before = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:opencode-server");
|
||||
expect(before).toBeDefined();
|
||||
expect(before!.source.provider).toBe("opencode");
|
||||
expect(before!.state).toBe("disabled");
|
||||
|
||||
// The dashboard withholds sourcePath for tool-owned sources, mirroring
|
||||
// the #writableMcpSourcePath gate.
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
name: "opencode-server",
|
||||
enabled: true,
|
||||
});
|
||||
|
||||
// opencode.json MUST stay untouched.
|
||||
const opencodeRaw = JSON.parse(await Bun.file(opencodePath).text()) as {
|
||||
mcp: { "opencode-server": { enabled: boolean } };
|
||||
};
|
||||
expect(opencodeRaw.mcp["opencode-server"].enabled).toBe(false);
|
||||
|
||||
// The override lands in the user mcp.json's enabledServers list.
|
||||
const userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir));
|
||||
expect(userConfig.enabledServers ?? []).toContain("opencode-server");
|
||||
|
||||
const after = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:opencode-server");
|
||||
expect(after).toBeDefined();
|
||||
expect(after!.state).toBe("active");
|
||||
|
||||
// Disabling again clears the override.
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
name: "opencode-server",
|
||||
enabled: false,
|
||||
});
|
||||
|
||||
const userConfigAfter = await readMCPConfigFile(getMCPConfigPath("user", projectDir));
|
||||
expect(userConfigAfter.enabledServers ?? []).not.toContain("opencode-server");
|
||||
expect(userConfigAfter.disabledServers ?? []).toContain("opencode-server");
|
||||
|
||||
const offAgain = (await loadAllExtensions(projectDir, [])).find(e => e.id === "mcp:opencode-server");
|
||||
expect(offAgain).toBeDefined();
|
||||
expect(offAgain!.state).toBe("disabled");
|
||||
});
|
||||
|
||||
test("dashboard toggles on a discovered (config-less) server use the denylist", async () => {
|
||||
// `phantom-server` is not in any config; only the denylist can suppress it.
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
name: "phantom-server",
|
||||
enabled: false,
|
||||
});
|
||||
|
||||
let userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir));
|
||||
expect(userConfig.disabledServers ?? []).toContain("phantom-server");
|
||||
|
||||
await setMcpServerEnabled({
|
||||
userPath: getMCPConfigPath("user", projectDir),
|
||||
projectPath: getMCPConfigPath("project", projectDir),
|
||||
name: "phantom-server",
|
||||
enabled: true,
|
||||
});
|
||||
|
||||
userConfig = await readMCPConfigFile(getMCPConfigPath("user", projectDir));
|
||||
expect(userConfig.disabledServers ?? []).not.toContain("phantom-server");
|
||||
});
|
||||
});
|
||||
@@ -9,9 +9,7 @@ import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { TempDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
type FastModeScope = "both" | "openai" | "claude";
|
||||
|
||||
describe("fast mode scope", () => {
|
||||
describe("/fast targets the current model's service-tier family", () => {
|
||||
let tempDir: TempDir;
|
||||
let authStorage: AuthStorage;
|
||||
let session: AgentSession;
|
||||
@@ -29,77 +27,54 @@ describe("fast mode scope", () => {
|
||||
tempDir.removeSync();
|
||||
});
|
||||
|
||||
async function createSession(fastModeScope?: FastModeScope): Promise<AgentSession> {
|
||||
const model = getBundledModel("anthropic", "claude-sonnet-4-5");
|
||||
async function createSession(provider: "anthropic" | "openai", modelId: string): Promise<AgentSession> {
|
||||
const model = getBundledModel(provider, modelId);
|
||||
if (!model) {
|
||||
throw new Error("Expected bundled test model to exist");
|
||||
throw new Error(`Expected bundled test model ${provider}/${modelId} to exist`);
|
||||
}
|
||||
|
||||
const settings = fastModeScope === undefined ? Settings.isolated() : Settings.isolated({ fastModeScope });
|
||||
const agent = new Agent({
|
||||
initialState: {
|
||||
model,
|
||||
systemPrompt: ["Test"],
|
||||
tools: [],
|
||||
messages: [],
|
||||
},
|
||||
initialState: { model, systemPrompt: ["Test"], tools: [], messages: [] },
|
||||
});
|
||||
|
||||
authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
|
||||
authStorage.setRuntimeApiKey(model.provider, "anthropic-token");
|
||||
authStorage.setRuntimeApiKey(model.provider, "token");
|
||||
modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
||||
|
||||
session = new AgentSession({
|
||||
agent,
|
||||
sessionManager: SessionManager.inMemory(),
|
||||
settings,
|
||||
settings: Settings.isolated(),
|
||||
modelRegistry,
|
||||
});
|
||||
session.subscribe(() => {});
|
||||
return session;
|
||||
}
|
||||
|
||||
it("scopes enabled fast mode to OpenAI when configured", async () => {
|
||||
const session = await createSession("openai");
|
||||
|
||||
it("enables priority on the Anthropic family for a Claude model", async () => {
|
||||
const session = await createSession("anthropic", "claude-sonnet-4-5");
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("openai-only");
|
||||
expect(session.serviceTierByFamily).toEqual({ anthropic: "priority" });
|
||||
expect(session.isFastModeEnabled()).toBe(true);
|
||||
});
|
||||
|
||||
it("scopes enabled fast mode to Claude when configured", async () => {
|
||||
const session = await createSession("claude");
|
||||
|
||||
it("enables priority on the OpenAI family for an OpenAI model", async () => {
|
||||
const session = await createSession("openai", "gpt-5.2");
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("claude-only");
|
||||
expect(session.serviceTierByFamily).toEqual({ openai: "priority" });
|
||||
expect(session.isFastModeEnabled()).toBe(true);
|
||||
});
|
||||
|
||||
it("defaults enabled fast mode to priority for both providers", async () => {
|
||||
const session = await createSession();
|
||||
|
||||
it("clears only the current model's family when disabled", async () => {
|
||||
const session = await createSession("anthropic", "claude-sonnet-4-5");
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("priority");
|
||||
});
|
||||
|
||||
it("clears the service tier when disabled", async () => {
|
||||
const session = await createSession("openai");
|
||||
session.setFastMode(true);
|
||||
|
||||
session.setFastMode(false);
|
||||
|
||||
expect(session.serviceTier).toBeUndefined();
|
||||
expect(session.serviceTierByFamily).toEqual({});
|
||||
expect(session.isFastModeEnabled()).toBe(false);
|
||||
});
|
||||
|
||||
it("does not broaden an already enabled scoped tier", async () => {
|
||||
const session = await createSession("claude");
|
||||
session.setFastMode(true);
|
||||
expect(session.serviceTier).toBe("claude-only");
|
||||
session.settings.set("fastModeScope", "both");
|
||||
|
||||
session.setFastMode(true);
|
||||
|
||||
expect(session.serviceTier).toBe("claude-only");
|
||||
it("toggle reports the resulting state", async () => {
|
||||
const session = await createSession("anthropic", "claude-sonnet-4-5");
|
||||
expect(session.toggleFastMode()).toBe(true);
|
||||
expect(session.serviceTierByFamily.anthropic).toBe("priority");
|
||||
expect(session.toggleFastMode()).toBe(false);
|
||||
expect(session.serviceTierByFamily.anthropic).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -98,4 +98,26 @@ describe("generateFileMentionMessages path resolution", () => {
|
||||
expect(message.files).toHaveLength(1);
|
||||
expect(message.files[0]?.path).toBe("My Folder/my file.png");
|
||||
});
|
||||
|
||||
test("skips auto-reading a binary file instead of injecting raw bytes", async () => {
|
||||
const cwd = await createTempDir();
|
||||
// TTF header begins with a NUL run; auto-reading it as text would leak
|
||||
// control bytes into the conversation (the reported bug).
|
||||
await Bun.write(path.join(cwd, "Silver.ttf"), Buffer.from([0x00, 0x01, 0x00, 0x00, 0x00, 0x0c, 0x4f, 0x53]));
|
||||
// A non-NUL invalid-UTF8 blob must be refused too, not just NUL-bearing files.
|
||||
await Bun.write(path.join(cwd, "blob.bin"), Buffer.from([0x4d, 0x5a, 0xff, 0xfe, 0xc0, 0xc0]));
|
||||
|
||||
const messages = await generateFileMentionMessages(["Silver.ttf", "blob.bin"], cwd);
|
||||
expect(messages).toHaveLength(1);
|
||||
const message = messages[0];
|
||||
if (message?.role !== "fileMention") {
|
||||
throw new Error("expected file mention message");
|
||||
}
|
||||
expect(message.files).toHaveLength(2);
|
||||
for (const file of message.files) {
|
||||
expect(file.skippedReason).toBe("binary");
|
||||
expect(file.content).toContain("binary file");
|
||||
expect(file.content).not.toContain("\u0000");
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
@@ -17,6 +17,7 @@ import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
||||
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
||||
import { SILENT_ABORT_MARKER, USER_INTERRUPT_LABEL } from "@oh-my-pi/pi-coding-agent/session/messages";
|
||||
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
||||
import { AUTO_THINKING } from "@oh-my-pi/pi-coding-agent/thinking";
|
||||
import { setKeybindings } from "@oh-my-pi/pi-tui";
|
||||
import { formatNumber, TempDir } from "@oh-my-pi/pi-utils";
|
||||
|
||||
@@ -836,6 +837,46 @@ describe("InteractiveMode plan review rendering", () => {
|
||||
expect(defaultApply?.[0]?.explicitThinkingLevel).toBe(true);
|
||||
});
|
||||
|
||||
it("preserves DEFAULT(auto) when plan approval restores the default tier", async () => {
|
||||
const sonnet = session.modelRegistry.find("anthropic", "claude-sonnet-4-5");
|
||||
const opus = session.modelRegistry.find("anthropic", "claude-opus-4-5");
|
||||
if (!sonnet || !opus) throw new Error("Expected sonnet + opus to exist in registry");
|
||||
|
||||
session.settings.setModelRole("default", "anthropic/claude-sonnet-4-5");
|
||||
session.settings.setModelRole("slow", "anthropic/claude-opus-4-5");
|
||||
session.settings.setModelRole("plan", "anthropic/claude-opus-4-5");
|
||||
session.setThinkingLevel(AUTO_THINKING, true);
|
||||
|
||||
const planFilePath = "local://PLAN.md";
|
||||
const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, {
|
||||
getArtifactsDir: () => session.sessionManager.getArtifactsDir(),
|
||||
getSessionId: () => session.sessionManager.getSessionId(),
|
||||
});
|
||||
await Bun.write(resolvedPlanPath, "# Plan\n\nPreserve the configured auto selector.");
|
||||
|
||||
await mode.handlePlanModeCommand();
|
||||
expect(session.model?.id).toBe(opus.id);
|
||||
|
||||
vi.spyOn(session, "getContextUsage").mockReturnValue(undefined);
|
||||
vi.spyOn(session, "prompt").mockResolvedValue(undefined as never);
|
||||
|
||||
vi.spyOn(mode, "showPlanReview").mockImplementation(
|
||||
async (_planContent, _title, _options, _dialogOptions, extra?: { slider?: HookSelectorSlider }) => {
|
||||
const slider = extra?.slider;
|
||||
expect(slider).toBeDefined();
|
||||
const defaultIndex = slider!.segments.findIndex(segment => segment.label === "default");
|
||||
expect(defaultIndex).toBeGreaterThanOrEqual(0);
|
||||
slider!.onChange?.(defaultIndex);
|
||||
return "Approve and keep context";
|
||||
},
|
||||
);
|
||||
|
||||
await mode.handlePlanApproval({ planFilePath, planExists: true, title: "PLAN" });
|
||||
|
||||
expect(session.model?.id).toBe(sonnet.id);
|
||||
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
||||
});
|
||||
|
||||
it("falls back to the pre-plan model when only plan is configured and the slider is hidden", async () => {
|
||||
const sonnet = session.modelRegistry.find("anthropic", "claude-sonnet-4-5");
|
||||
const opus = session.modelRegistry.find("anthropic", "claude-opus-4-5");
|
||||
|
||||
@@ -171,10 +171,30 @@ describe("issue #3291 — tiny-model downloads keep the worker referenced", () =
|
||||
|
||||
worker.emit({ type: "downloaded", id: downloadRequestId });
|
||||
|
||||
expect(await download).toBe(true);
|
||||
expect(await download).toEqual({ ok: true });
|
||||
expect(worker.unrefCalls).toBe(1);
|
||||
} finally {
|
||||
await client.terminate();
|
||||
}
|
||||
});
|
||||
|
||||
it("returns the worker error for failed download requests", async () => {
|
||||
let downloadRequestId = "";
|
||||
const worker = new FakeTinyWorker(message => {
|
||||
if (message.type === "download") downloadRequestId = message.id;
|
||||
});
|
||||
const client = new TinyTitleClient(() => worker);
|
||||
|
||||
try {
|
||||
const download = client.downloadModel("lfm2-700m");
|
||||
|
||||
expect(downloadRequestId).not.toBe("");
|
||||
worker.emit({ type: "error", id: downloadRequestId, error: "Error: runtime install failed" });
|
||||
|
||||
expect(await download).toEqual({ ok: false, error: "Error: runtime install failed" });
|
||||
expect(worker.terminated).toBe(true);
|
||||
} finally {
|
||||
await client.terminate();
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
@@ -48,7 +48,7 @@ describe("job renderer task-result preview", () => {
|
||||
|
||||
it("previews the envelope body, not the wrapper markup", () => {
|
||||
const summary = prompt.render(taskSummaryTemplate, {
|
||||
agentName: "quick_task",
|
||||
agentName: "sonic",
|
||||
id: "SpawnProbe",
|
||||
status: "completed",
|
||||
duration: "8.7s",
|
||||
@@ -83,7 +83,7 @@ describe("job renderer task-result preview", () => {
|
||||
|
||||
it("flattens a pretty-printed JSON body instead of previewing a lone brace", () => {
|
||||
const summary = prompt.render(taskSummaryTemplate, {
|
||||
agentName: "quick_task",
|
||||
agentName: "sonic",
|
||||
id: "EchoAlpha",
|
||||
status: "completed",
|
||||
duration: "11.6s",
|
||||
|
||||
@@ -43,6 +43,9 @@ function restoreEnvValue(name: string, value: string | undefined): void {
|
||||
}
|
||||
function createController(authStorage: AuthStorage, mcpManagerOverrides: Record<string, unknown> = {}) {
|
||||
const showError = vi.fn();
|
||||
const showStatus = vi.fn();
|
||||
const present = vi.fn();
|
||||
const editor: { onEscape?: () => void } = {};
|
||||
const prepareConfig = vi.fn(async (config: MCPServerConfig) => config);
|
||||
const mcpManager = {
|
||||
prepareConfig,
|
||||
@@ -55,11 +58,11 @@ function createController(authStorage: AuthStorage, mcpManagerOverrides: Record<
|
||||
};
|
||||
const controller = new MCPCommandController({
|
||||
chatContainer: { addChild: vi.fn() },
|
||||
present: vi.fn(),
|
||||
present,
|
||||
ui: { requestRender: vi.fn() },
|
||||
editor: {},
|
||||
editor,
|
||||
showError,
|
||||
showStatus: vi.fn(),
|
||||
showStatus,
|
||||
oauthManualInput: {
|
||||
hasPending: vi.fn(() => false),
|
||||
pendingProviderId: undefined,
|
||||
@@ -72,7 +75,7 @@ function createController(authStorage: AuthStorage, mcpManagerOverrides: Record<
|
||||
mcpManager,
|
||||
} as never);
|
||||
|
||||
return { controller, showError, prepareConfig, mcpManager };
|
||||
return { controller, showError, showStatus, present, editor, prepareConfig, mcpManager };
|
||||
}
|
||||
|
||||
describe("/mcp auth commands", () => {
|
||||
@@ -219,6 +222,109 @@ describe("/mcp auth commands", () => {
|
||||
});
|
||||
});
|
||||
|
||||
test("Esc aborts the OAuth flow during /mcp reauth", async () => {
|
||||
const authStorage = freshAuthStorage();
|
||||
await authStorage.reload();
|
||||
vi.spyOn(mcpClient, "connectToServer").mockRejectedValue(AUTH_ERROR);
|
||||
|
||||
// Simulate the real flow: login hangs waiting for the OAuth callback and
|
||||
// only resolves when the controller's signal aborts. Mirrors what
|
||||
// OAuthCallbackFlow.#waitForCallback does in production.
|
||||
vi.spyOn(oauthFlow.MCPOAuthFlow.prototype, "login").mockImplementation(function (this: oauthFlow.MCPOAuthFlow) {
|
||||
const pending = Promise.withResolvers<never>();
|
||||
this.ctrl.signal?.addEventListener("abort", () => {
|
||||
pending.reject(new Error(`OAuth callback cancelled: ${String(this.ctrl.signal?.reason ?? "aborted")}`));
|
||||
});
|
||||
return pending.promise;
|
||||
});
|
||||
|
||||
const { controller, showError, showStatus, editor } = createController(authStorage);
|
||||
|
||||
const reauthPromise = controller.handle("/mcp reauth envserver");
|
||||
|
||||
// Wait for #handleOAuthFlow to install its editor.onEscape hook.
|
||||
const deadline = Date.now() + 1_000;
|
||||
while (typeof editor.onEscape !== "function" && Date.now() < deadline) {
|
||||
await Bun.sleep(10);
|
||||
}
|
||||
expect(typeof editor.onEscape).toBe("function");
|
||||
|
||||
const installedEscape = editor.onEscape;
|
||||
editor.onEscape?.();
|
||||
|
||||
// Cancellation must resolve the reauth promise promptly (well under the
|
||||
// 5-minute production timeout); a 2s race exposes a hung flow as a test
|
||||
// failure rather than a suite hang.
|
||||
await Promise.race([
|
||||
reauthPromise,
|
||||
Bun.sleep(2_000).then(() => {
|
||||
throw new Error("reauth did not resolve within 2s of Esc");
|
||||
}),
|
||||
]);
|
||||
|
||||
expect(showError).not.toHaveBeenCalled();
|
||||
expect(showStatus).toHaveBeenCalledWith(expect.stringMatching(/cancel/i));
|
||||
// onEscape must be restored to its previous value so subsequent user
|
||||
// input does not keep aborting the (now-finished) flow.
|
||||
expect(editor.onEscape).not.toBe(installedEscape);
|
||||
});
|
||||
|
||||
test("Esc cancels even when OAuth login has not registered its signal listener yet", async () => {
|
||||
const authStorage = freshAuthStorage();
|
||||
await authStorage.reload();
|
||||
vi.spyOn(mcpClient, "connectToServer").mockRejectedValue(AUTH_ERROR);
|
||||
|
||||
// Simulates the review race: Esc aborts oauthTimeout before
|
||||
// OAuthCallbackFlow.#waitForCallback has registered its abort listener
|
||||
// (e.g. during dynamic client registration or metadata discovery).
|
||||
// The login promise itself never observes ctrl.signal; #handleOAuthFlow
|
||||
// must race it against oauthTimeout.signal.
|
||||
vi.spyOn(oauthFlow.MCPOAuthFlow.prototype, "login").mockReturnValue(Promise.withResolvers<never>().promise);
|
||||
const { controller, showError, showStatus, editor } = createController(authStorage);
|
||||
|
||||
const reauthPromise = controller.handle("/mcp reauth envserver");
|
||||
const deadline = Date.now() + 1_000;
|
||||
while (typeof editor.onEscape !== "function" && Date.now() < deadline) {
|
||||
await Bun.sleep(10);
|
||||
}
|
||||
expect(typeof editor.onEscape).toBe("function");
|
||||
editor.onEscape?.();
|
||||
|
||||
await Promise.race([
|
||||
reauthPromise,
|
||||
Bun.sleep(2_000).then(() => {
|
||||
throw new Error("reauth did not resolve within 2s of pre-wait Esc");
|
||||
}),
|
||||
]);
|
||||
|
||||
expect(showError).not.toHaveBeenCalled();
|
||||
expect(showStatus).toHaveBeenCalledWith(expect.stringMatching(/cancel/i));
|
||||
});
|
||||
|
||||
test("OAuth deadline still surfaces as a reauthorization error, not a cancellation", async () => {
|
||||
const authStorage = freshAuthStorage();
|
||||
await authStorage.reload();
|
||||
vi.spyOn(mcpClient, "connectToServer").mockRejectedValue(AUTH_ERROR);
|
||||
|
||||
// Deadline path bypasses both the editor's Esc hook and any external
|
||||
// signal: withTimeout aborts the controller with reason "MCP OAuth flow
|
||||
// timed out" and the login promise rejects with a "timed out" message.
|
||||
// Mirror that here. Keeping the surface distinct from the user-cancel
|
||||
// flag in #handleOAuthFlow is the whole point of this regression test.
|
||||
vi.spyOn(oauthFlow.MCPOAuthFlow.prototype, "login").mockRejectedValue(
|
||||
new Error("OAuth flow timed out after 5 minutes"),
|
||||
);
|
||||
const { controller, showError, showStatus } = createController(authStorage);
|
||||
|
||||
await controller.handle("/mcp reauth envserver");
|
||||
|
||||
// Deadline must read as "failed", not "cancelled" — they have different
|
||||
// surfaces (error banner vs status line) and the user expects a clear
|
||||
// timeout message rather than thinking they pressed Esc.
|
||||
expect(showStatus).not.toHaveBeenCalledWith(expect.stringMatching(/cancel/i));
|
||||
expect(showError).toHaveBeenCalledWith(expect.stringMatching(/timed out/i));
|
||||
});
|
||||
|
||||
test("clears both expanded and stale raw URL-keyed credentials on unauth", async () => {
|
||||
const authStorage = freshAuthStorage();
|
||||
await authStorage.reload();
|
||||
|
||||
@@ -7,7 +7,7 @@ import type { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-regis
|
||||
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
||||
import { ModelSelectorComponent } from "@oh-my-pi/pi-coding-agent/modes/components/model-selector";
|
||||
import { getThemeByName, setThemeInstance } from "@oh-my-pi/pi-coding-agent/modes/theme/theme";
|
||||
import type { ConfiguredThinkingLevel } from "@oh-my-pi/pi-coding-agent/thinking";
|
||||
import { AUTO_THINKING, type ConfiguredThinkingLevel } from "@oh-my-pi/pi-coding-agent/thinking";
|
||||
import type { TUI } from "@oh-my-pi/pi-tui";
|
||||
|
||||
function normalizeRenderedText(text: string): string {
|
||||
@@ -156,6 +156,26 @@ describe("ModelSelector role badge thinking display", () => {
|
||||
expect(rendered).not.toContain("low medium high max");
|
||||
});
|
||||
|
||||
test("reloads DEFAULT(auto) from defaultThinkingLevel", async () => {
|
||||
installTestTheme();
|
||||
const model = getBundledModel("openai", "gpt-5.5");
|
||||
if (!model) throw new Error("Expected bundled model openai/gpt-5.5");
|
||||
|
||||
const settings = Settings.isolated({
|
||||
defaultThinkingLevel: AUTO_THINKING,
|
||||
modelRoles: {
|
||||
default: `${model.provider}/${model.id}`,
|
||||
},
|
||||
});
|
||||
|
||||
const selector = createSelector(model, settings);
|
||||
await Bun.sleep(0);
|
||||
installTestTheme();
|
||||
|
||||
const rendered = normalizeRenderedText(selector.render(220).join("\n"));
|
||||
expect(rendered).toContain("DEFAULT (auto)");
|
||||
});
|
||||
|
||||
test("shows compact auto badges for unconfigured role defaults", async () => {
|
||||
installTestTheme();
|
||||
const settings = Settings.isolated({});
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user