diff --git a/packages/ai/test/deepseek-reasoning-content.test.ts b/packages/ai/test/deepseek-reasoning-content.test.ts index 502bb9a6d..4f770a5ba 100644 --- a/packages/ai/test/deepseek-reasoning-content.test.ts +++ b/packages/ai/test/deepseek-reasoning-content.test.ts @@ -68,7 +68,7 @@ function assistantToolCall( describe("DeepSeek reasoning_content tool-call replay", () => { // ---------------------------------------------------------------- // Fix 1: honest wire-exact ladders for DeepSeek-family on any provider — - // V4 Flash exposes [low, high, max] (#7668), V4 Pro stays [high, max]. + // V4 Flash and Pro expose [low, high, max] (#7668, #8405). // ---------------------------------------------------------------- describe("thinking ladder (Fix 1)", () => { it("bakes the honest [low, high, max] flash ladder with no effortMap on opencode-go", () => { @@ -91,13 +91,13 @@ describe("DeepSeek reasoning_content tool-call replay", () => { expect(model.thinking?.effortMap).toBeUndefined(); }); - it("bakes the honest [high, max] ladder with no effortMap on the official endpoint", () => { + it("bakes the honest [low, high, max] ladder with no effortMap on the official endpoint", () => { const model = deepseekModel({ provider: "deepseek", baseUrl: "https://api.deepseek.com/v1", id: "deepseek-v4-pro", }); - expect(model.thinking?.efforts).toEqual([Effort.High, Effort.Max]); + expect(model.thinking?.efforts).toEqual([Effort.Low, Effort.High, Effort.Max]); expect(model.thinking?.effortMap).toBeUndefined(); }); diff --git a/packages/coding-agent/test/agent-session-retry-fallback.test.ts b/packages/coding-agent/test/agent-session-retry-fallback.test.ts index 7573f3e96..3696e2e37 100644 --- a/packages/coding-agent/test/agent-session-retry-fallback.test.ts +++ b/packages/coding-agent/test/agent-session-retry-fallback.test.ts @@ -4108,7 +4108,7 @@ describe("AgentSession retry fallback", () => { it("skips usage fallbacks whose effort floor exceeds the session ceiling", async () => { const primaryModel = getBundledModel("anthropic", "claude-sonnet-4-5"); - const incompatibleFallback = getBundledModel("fireworks", "deepseek-v4-pro"); + const incompatibleFallback = getBundledModel("openrouter", "deepseek/deepseek-v4-pro"); const compatibleFallback = getBundledModel("openai", "gpt-4o-mini"); if (!primaryModel || !incompatibleFallback || !compatibleFallback) { throw new Error("Expected bundled usage fallback effort models"); diff --git a/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts b/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts index 30c299a20..f255a827b 100644 --- a/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts +++ b/packages/coding-agent/test/tools/browser-tab-evaluate.test.ts @@ -1,4 +1,4 @@ -import { describe, expect, it, vi } from "bun:test"; +import { afterAll, beforeAll, describe, expect, it, vi } from "bun:test"; import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings"; import type { ToolSession } from "@oh-my-pi/pi-coding-agent/sdk"; import { BrowserTool } from "@oh-my-pi/pi-coding-agent/tools/browser"; @@ -19,6 +19,23 @@ function makeSession(): ToolSession { } describe.skipIf(!CHROMIUM_AVAILABLE)("browser tab evaluation", () => { + const suiteTool = new BrowserTool(makeSession()); + const suiteTabName = `evaluation-suite-${process.pid}`; + + // Keep one browser lease across the suite. Tests still get isolated tabs and workers, + // while the chunked full run avoids relaunching Chromium for every test under load. + beforeAll(async () => { + await suiteTool.execute("open", { + action: "open", + name: suiteTabName, + url: "data:text/html,Browser evaluation suite", + }); + }, 30_000); + + afterAll(async () => { + await suiteTool.execute("close", { action: "close", name: suiteTabName, kill: true }); + }, 30_000); + // Launches real headless Chromium; CI cold start easily exceeds bun's 5s default. it("runs tab.evaluate in the page's main JavaScript world", async () => { const tool = new BrowserTool(makeSession());