diff --git a/packages/ai/test/azure-openai-responses-stream.test.ts b/packages/ai/test/azure-openai-responses-stream.test.ts index 5d9f09aee..1a940fc35 100644 --- a/packages/ai/test/azure-openai-responses-stream.test.ts +++ b/packages/ai/test/azure-openai-responses-stream.test.ts @@ -276,7 +276,7 @@ describe("azure openai responses streaming", () => { expect(result.stopReason).toBe("error"); expect(result.errorMessage).toContain("server_error: backend exploded late"); }); - it("preserves assistant message id and phase when rebuilding fallback replay history", async () => { + it("preserves assistant message phase when rebuilding fallback replay history", async () => { const payload = await captureAzurePayload({ messages: [ { role: "user", content: "first user", timestamp: Date.now() }, diff --git a/packages/coding-agent/CHANGELOG.md b/packages/coding-agent/CHANGELOG.md index 816036b45..4b8c96e71 100644 --- a/packages/coding-agent/CHANGELOG.md +++ b/packages/coding-agent/CHANGELOG.md @@ -558,9 +558,6 @@ - Mid-session `xd://` mount changes (e.g. MCP connect/disconnect) no longer rewrite the system prompt: the delta is announced to the model as a steered system notice ("these tools became available" / "no longer mounted"), so the provider prompt cache stays intact; device docs join the prompt on the next unrelated rebuild. ### Fixed -- Fixed prewalk never switching after a successful todo call; the todo gate now opens at the completed turn boundary without allowing the todo call itself to trigger the handoff. -- Fixed live pending tool headers retaining a stale spinner frame by rebuilding their cached display on each spinner tick. -- Fixed task and eval tool descriptions omitting the configured default or allowed subagent names under restricted spawn policies. - Fixed a bug where a nested configuration value (like `dev.autoqa.consent` / `dev.autoqaConsent`) would incorrectly satisfy a parent key lookup (like `dev.autoqa`), causing Auto QA to be enabled and prompt for consent by default when it should have been disabled. - Fixed compiled appserver startup deadlocking before socket creation when user extensions were present. diff --git a/packages/coding-agent/src/config/settings-schema.ts b/packages/coding-agent/src/config/settings-schema.ts index eb526cfc7..2e2346aa8 100644 --- a/packages/coding-agent/src/config/settings-schema.ts +++ b/packages/coding-agent/src/config/settings-schema.ts @@ -139,7 +139,6 @@ export const TAB_GROUPS: Record = { tools: [ "Available Tools", "Todos", - "Subagents", "Grep & Browser", "GitHub", "Output Limits", diff --git a/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts b/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts index a422141d2..8af6f5678 100644 --- a/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts +++ b/packages/coding-agent/src/eval/__tests__/julia-prelude.test.ts @@ -5,24 +5,21 @@ import { disposeJuliaKernelSessionsByOwner, executeJulia } from "../jl/executor" const HAS_JULIA = Boolean($which("julia")); const OWNER_ID = "julia-prelude-tests"; -const JULIA_TEST_TIMEOUT_MS = 60_000; describe.skipIf(!HAS_JULIA)("eval Julia prelude helpers", () => { afterEach(async () => { await disposeJuliaKernelSessionsByOwner(OWNER_ID); }, 30_000); - it( - "supports output ranges, JSON queries, metadata, and ANSI stripping", - async () => { - using tempDir = TempDir.createSync("@omp-eval-julia-output-"); - const artifactsDir = path.join(tempDir.path(), "session-artifacts"); - await Bun.write(path.join(artifactsDir, "alpha.md"), "one\ntwo\nthree\nfour"); - await Bun.write(path.join(artifactsDir, "json.md"), JSON.stringify({ items: [{ name: "a" }, { name: "b" }] })); - await Bun.write(path.join(artifactsDir, "ansi.md"), "\u001b[31mred\u001b[0m"); + it("supports output ranges, JSON queries, metadata, and ANSI stripping", async () => { + using tempDir = TempDir.createSync("@omp-eval-julia-output-"); + const artifactsDir = path.join(tempDir.path(), "session-artifacts"); + await Bun.write(path.join(artifactsDir, "alpha.md"), "one\ntwo\nthree\nfour"); + await Bun.write(path.join(artifactsDir, "json.md"), JSON.stringify({ items: [{ name: "a" }, { name: "b" }] })); + await Bun.write(path.join(artifactsDir, "ansi.md"), "\u001b[31mred\u001b[0m"); - const result = await executeJulia( - ` + const result = await executeJulia( + ` println("RANGE=", replace(output("alpha", offset=2, limit=2), "\\n" => "|")) println("QUERY=", output("json", query=".items[1].name")) println("STRIPPED=", output("ansi", format="stripped")) @@ -32,44 +29,38 @@ multi = output("alpha", "json") println("MULTI=", length(multi), ":", multi[1]["id"], ":", multi[2]["id"]) nothing `, - { - cwd: tempDir.path(), - artifactsDir, - sessionId: `julia-prelude-output:${crypto.randomUUID()}`, - kernelOwnerId: OWNER_ID, - reset: true, - }, - ); - - expect(result.exitCode).toBe(0); - expect(result.output).toContain("RANGE=two|three"); - expect(result.output).toContain('QUERY="b"'); - expect(result.output).toContain("STRIPPED=red"); - expect(result.output).toContain("META=alpha:true"); - expect(result.output).toContain("MULTI=2:alpha:json"); - }, - JULIA_TEST_TIMEOUT_MS, - ); - - it( - "surfaces the exception type and message in the error output, not just stack frames", - async () => { - using tempDir = TempDir.createSync("@omp-eval-julia-error-"); - const result = await executeJulia(`println("="^8)\nmissing_var_xyz + 1`, { + { cwd: tempDir.path(), - sessionId: `julia-prelude-error:${crypto.randomUUID()}`, + artifactsDir, + sessionId: `julia-prelude-output:${crypto.randomUUID()}`, kernelOwnerId: OWNER_ID, reset: true, - }); + }, + ); - // The rendered error must carry the actual exception, not only the - // runner-internal backtrace frames (regression: traceback-only output - // hid `ename`/`evalue`). - expect(result.output).toContain("UndefVarError"); - expect(result.output).toContain("missing_var_xyz"); - // Frames are still present alongside the message. - expect(result.output).toContain("top-level scope"); - }, - JULIA_TEST_TIMEOUT_MS, - ); + expect(result.exitCode).toBe(0); + expect(result.output).toContain("RANGE=two|three"); + expect(result.output).toContain('QUERY="b"'); + expect(result.output).toContain("STRIPPED=red"); + expect(result.output).toContain("META=alpha:true"); + expect(result.output).toContain("MULTI=2:alpha:json"); + }, 60_000); + + it("surfaces the exception type and message in the error output, not just stack frames", async () => { + using tempDir = TempDir.createSync("@omp-eval-julia-error-"); + const result = await executeJulia(`println("="^8)\nmissing_var_xyz + 1`, { + cwd: tempDir.path(), + sessionId: `julia-prelude-error:${crypto.randomUUID()}`, + kernelOwnerId: OWNER_ID, + reset: true, + }); + + // The rendered error must carry the actual exception, not only the + // runner-internal backtrace frames (regression: traceback-only output + // hid `ename`/`evalue`). + expect(result.output).toContain("UndefVarError"); + expect(result.output).toContain("missing_var_xyz"); + // Frames are still present alongside the message. + expect(result.output).toContain("top-level scope"); + }, 30_000); }); diff --git a/packages/coding-agent/src/modes/theme/mermaid-cache.ts b/packages/coding-agent/src/modes/theme/mermaid-cache.ts index a4e6caa5c..810287e9a 100644 --- a/packages/coding-agent/src/modes/theme/mermaid-cache.ts +++ b/packages/coding-agent/src/modes/theme/mermaid-cache.ts @@ -32,10 +32,6 @@ function asciiDisplayWidth(ascii: string): number { return max; } -type DirectionalMermaidAsciiRenderOptions = MermaidAsciiRenderOptions & { - direction: "TD" | "LR"; -}; - function renderVariant( source: string, baseOptions: MermaidAsciiRenderOptions, @@ -46,10 +42,7 @@ function renderVariant( const cached = cache.get(key); if (cached !== undefined) return cached; - const directionalOptions: DirectionalMermaidAsciiRenderOptions | MermaidAsciiRenderOptions = direction - ? { ...baseOptions, direction } - : baseOptions; - const ascii = renderMermaidAsciiSafe(source, directionalOptions); + const ascii = renderMermaidAsciiSafe(source, direction ? { ...baseOptions, direction } : baseOptions); cache.set(key, ascii); return ascii; } diff --git a/packages/coding-agent/src/prompts/tools/task.md b/packages/coding-agent/src/prompts/tools/task.md index 9a22f98c8..0e3be887f 100644 --- a/packages/coding-agent/src/prompts/tools/task.md +++ b/packages/coding-agent/src/prompts/tools/task.md @@ -11,8 +11,7 @@ Agents marked BLOCKING run inline — results return in this call; non-blocking {{/if}} # Task Design -- **Agent typing:** Pick each item's `agent` type. Read-only research MUST use `agent: "scout"` (faster model). Use the general-purpose worker (`{{defaultAgent}}`) only when no specialist fits. -{{#if allowedAgentsText}}Current spawn policy allows: {{allowedAgentsText}}.{{/if}} +- **Agent typing:** Pick each item's `agent` type. Read-only research MUST use `agent: "scout"` (faster model). Use default worker only when no specialist fits. - **No overhead:** Each `task` MUST instruct its agent to skip formatters, linters, and project-wide test suites. Run those once at the end. - **One-pass:** Prefer agents that investigate AND edit in one pass; spin a read-only scout only when affected files are genuinely unknown. diff --git a/packages/coding-agent/src/session/agent-session.ts b/packages/coding-agent/src/session/agent-session.ts index 6a09b8e33..fd2b35f47 100644 --- a/packages/coding-agent/src/session/agent-session.ts +++ b/packages/coding-agent/src/session/agent-session.ts @@ -2506,9 +2506,6 @@ export class AgentSession { const action = todoGateOpen ? context.toolResults.find(result => PREWALK_ACTION_TOOLS[result.toolName]) : undefined; - if (context.toolResults.some(result => result.toolName === "todo" && !result.isError)) { - this.#prewalkTodoSeen = true; - } if (!action) { if (!this.#prewalkPlanInjected) { this.#prewalkPlanInjected = true; diff --git a/packages/coding-agent/src/task/index.ts b/packages/coding-agent/src/task/index.ts index 75944afd5..e3046ba92 100644 --- a/packages/coding-agent/src/task/index.ts +++ b/packages/coding-agent/src/task/index.ts @@ -187,7 +187,6 @@ function renderDescription( agents: renderedAgents, spawningDisabled, defaultAgent: spawnPolicy.defaultAgent, - allowedAgentsText: spawnPolicy.allowedPromptText, isolationEnabled, applyIsolatedChanges, batchEnabled, diff --git a/packages/coding-agent/src/tools/__tests__/eval-description.test.ts b/packages/coding-agent/src/tools/__tests__/eval-description.test.ts index 4f34f74ec..49ff35477 100644 --- a/packages/coding-agent/src/tools/__tests__/eval-description.test.ts +++ b/packages/coding-agent/src/tools/__tests__/eval-description.test.ts @@ -6,7 +6,7 @@ describe("eval tool description", () => { const description = getEvalToolDescription({ py: true, js: false, spawns: "fact-finder,oracle" }); expect(description).toContain('agent(prompt, agent?="fact-finder"'); - expect(description).toContain("Allowed agents: `fact-finder`, `oracle`. omit it to use `fact-finder`."); + expect(description).toContain("Allowed: `fact-finder`, `oracle`."); }); it("omits agent() when spawning is disabled", () => { diff --git a/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts b/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts index eb0f5bd86..c8f2eb12b 100644 --- a/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-snapcompact-auto-fallback.test.ts @@ -24,11 +24,6 @@ interface Harness { interface HarnessOptions { activeModel: { provider: GeneratedProvider; id: string }; seedMessages?: Message[]; - // Prompt-token count billed on the triggering assistant turn. Stays at or - // below the active model's context window for a threshold-driven compaction; - // exceeding it routes through the overflow-recovery path, which drops the - // failed turn (and, in a minimal transcript, leaves nothing to summarize). - triggerInputTokens?: number; } async function createHarness(tempDir: TempDir, authStorage: AuthStorage, options: HarnessOptions): Promise { @@ -85,12 +80,10 @@ async function createHarness(tempDir: TempDir, authStorage: AuthStorage, options // snapcompact's renderability preflight can scan it, leaving nothing to // summarize). Derived from the live window so the fixture survives model // metadata changes (claude-sonnet-4-5's 200k window is narrower than the - // vision-role qwen's, so a fixed count would overflow one of them). Tests - // may override the billed input via triggerInputTokens. + // vision-role qwen's, so a fixed count would overflow one of them). const contextWindow = activeModel.contextWindow ?? 0; const thresholdTokens = compactionModule.resolveThresholdTokens(contextWindow, settings.getGroup("compaction")); - const derivedPromptTokens = contextWindow > 0 ? Math.floor((thresholdTokens + contextWindow) / 2) : 246_000; - const inputTokens = options.triggerInputTokens ?? derivedPromptTokens; + const promptTokens = contextWindow > 0 ? Math.floor((thresholdTokens + contextWindow) / 2) : 246_000; const assistantMsg = { role: "assistant" as const, content: [{ type: "text" as const, text: "Done." }], @@ -99,11 +92,11 @@ async function createHarness(tempDir: TempDir, authStorage: AuthStorage, options model: activeModel.id, stopReason: "stop" as const, usage: { - input: inputTokens, + input: promptTokens, output: 0, cacheRead: 0, cacheWrite: 0, - totalTokens: inputTokens, + totalTokens: promptTokens, cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0, total: 0 }, }, timestamp: Date.now(), @@ -159,10 +152,6 @@ describe("AgentSession auto-snapcompact local-blocker fallback", () => { authStorage = await AuthStorage.create(path.join(tempDir.path(), "auth.db")); const harness = await createHarness(tempDir, authStorage, { activeModel: { provider: "aimlapi", id: "claude-sonnet-4-5-20250929" }, - // claude-sonnet-4-5 has a 200K window; bill just under it so the turn - // trips the threshold compaction (which keeps the turn) instead of the - // overflow-recovery path (which drops it and empties the summarize set). - triggerInputTokens: 195_000, seedMessages: [ { role: "user", diff --git a/packages/coding-agent/test/interactive-mode-plan-review.test.ts b/packages/coding-agent/test/interactive-mode-plan-review.test.ts index bda65982e..399e9bcf1 100644 --- a/packages/coding-agent/test/interactive-mode-plan-review.test.ts +++ b/packages/coding-agent/test/interactive-mode-plan-review.test.ts @@ -446,7 +446,7 @@ describe("InteractiveMode plan review rendering", () => { expect(session.isPlanInternalAbortPending).toBe(false); }); - it("approves with in-overlay edits and mirrors them to the durable plan file", async () => { + it("approves with in-overlay edits and mirrors them to the plan file", async () => { const planFilePath = "local://PLAN.md"; const resolvedPlanPath = resolveLocalUrlToPath(planFilePath, { getArtifactsDir: () => session.sessionManager.getArtifactsDir(), @@ -477,11 +477,10 @@ describe("InteractiveMode plan review rendering", () => { title: "PLAN", }); - // The synthetic plan-approved prompt points execution at the durable - // local:// file instead of inlining stale plan content. + // The plan-approved prompt stays reference-only; approval must instead + // await the durable file mirror before dispatch so read sees the edit. const call = promptSpy.mock.calls.find(isPlanApprovedCall); expect(call).toBeDefined(); - expect(call?.[0] as string).toContain("local://PLAN.md"); expect(call?.[0] as string).not.toContain("edited body"); expect(call?.[0] as string).not.toContain("original body"); // onPlanEdited mirrored the edit to the plan file. diff --git a/packages/coding-agent/test/issue-2750-subagent-runtime-fallback.test.ts b/packages/coding-agent/test/issue-2750-subagent-runtime-fallback.test.ts index 6d9784631..85f15021d 100644 --- a/packages/coding-agent/test/issue-2750-subagent-runtime-fallback.test.ts +++ b/packages/coding-agent/test/issue-2750-subagent-runtime-fallback.test.ts @@ -1,11 +1,11 @@ import { afterEach, describe, expect, it, vi } from "bun:test"; import type { Api, Model } from "@oh-my-pi/pi-ai"; import { buildModel } from "@oh-my-pi/pi-catalog/build"; -import { Settings } from "../src/config/settings"; -import * as sdkModule from "../src/sdk"; -import type { AgentSession } from "../src/session/agent-session"; -import { runSubprocess } from "../src/task/executor"; -import type { AgentDefinition } from "../src/task/types"; +import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; +import * as sdkModule from "@oh-my-pi/pi-coding-agent/sdk"; +import type { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session"; +import { runSubprocess } from "@oh-my-pi/pi-coding-agent/task/executor"; +import type { AgentDefinition } from "@oh-my-pi/pi-coding-agent/task/types"; function model(provider: string, id: string): Model { return buildModel({ @@ -38,6 +38,12 @@ function createYieldingSession(): AgentSession { }, prompt: async () => { for (const listener of listeners) { + listener({ + type: "retry_fallback_applied", + from: "primary/bad-runtime-model", + to: "fallback/working-model", + role: "subagent:issue-2750", + }); listener({ type: "tool_execution_end", toolCallId: "tool-yield", @@ -45,12 +51,6 @@ function createYieldingSession(): AgentSession { result: { content: [{ type: "text", text: "Result submitted." }], details: { status: "success" } }, isError: false, }); - listener({ - type: "retry_fallback_applied", - from: "primary/bad-runtime-model", - to: "fallback/working-model", - role: "subagent:issue-2750", - }); } }, waitForIdle: async () => {}, diff --git a/packages/coding-agent/test/modes/components/tool-execution-spinner.test.ts b/packages/coding-agent/test/modes/components/tool-execution-spinner.test.ts index d3716977f..1eba10f41 100644 --- a/packages/coding-agent/test/modes/components/tool-execution-spinner.test.ts +++ b/packages/coding-agent/test/modes/components/tool-execution-spinner.test.ts @@ -21,8 +21,6 @@ describe("ToolExecutionComponent live preview spinners", () => { it("animates the eval pending cell while the call is live", () => { vi.useFakeTimers(); - let now = 0; - vi.spyOn(performance, "now").mockImplementation(() => now); const requestRender = vi.fn(); const requestComponentRender = vi.fn(); const component = new ToolExecutionComponent( @@ -36,7 +34,6 @@ describe("ToolExecutionComponent live preview spinners", () => { try { const firstFrame = stripVTControlCharacters(component.render(80).join("\n")); - now = 120; vi.advanceTimersByTime(120); const secondFrame = stripVTControlCharacters(component.render(80).join("\n")); diff --git a/packages/coding-agent/test/task/coordination-advisory.test.ts b/packages/coding-agent/test/task/coordination-advisory.test.ts index 0be14ccf3..a3bbd62cb 100644 --- a/packages/coding-agent/test/task/coordination-advisory.test.ts +++ b/packages/coding-agent/test/task/coordination-advisory.test.ts @@ -4,9 +4,9 @@ import type { TaskItem } from "@oh-my-pi/pi-coding-agent/task/types"; import { prompt } from "@oh-my-pi/pi-utils"; import subagentSystemPromptTemplate from "../../src/prompts/system/subagent-system-prompt.md" with { type: "text" }; -// Contract: a multi-sibling spawn with capacity and hub available draws -// a proactive coordination suggestion, and the subagent prompt tells peers -// to coordinate before overlapping edits. +// Contract: a multi-sibling spawn with spawn capacity and IRC available draws +// a proactive coordinate-via-irc suggestion, and the subagent COOP prompt +// actively tells peers to coordinate before overlapping edits. const item = (): TaskItem => ({ task: "do the thing" }); @@ -20,7 +20,8 @@ describe("buildCoordinationAdvisory", () => { it("stays silent for a single spawn", () => { expect(buildCoordinationAdvisory([item()], true, true)).toBeUndefined(); }); - it("stays silent when hub is unavailable", () => { + + it("stays silent when irc is unavailable", () => { expect(buildCoordinationAdvisory([item(), item()], true, false)).toBeUndefined(); });