d435385ab1
- Introduced `Max` as a first-class reasoning effort tier across all packages, including AI providers, coding agent configurations, and RPC protocols. - Refactored model effort ladders to use wire-exact mappings and removed legacy effort aliasing (e.g., `max-to-xhigh` mapping). - Updated model registry and provider configurations to support `Max` tier routing, color themes, and UI icon associations. - Expanded test suites to provide end-to-end coverage for the new reasoning tier, including updated compatibility and fallback scenarios.
659 lines
25 KiB
TypeScript
659 lines
25 KiB
TypeScript
import { afterEach, beforeEach, describe, expect, it, vi } from "bun:test";
|
|
import * as path from "node:path";
|
|
import { Agent } from "@oh-my-pi/pi-agent-core";
|
|
import { Effort } from "@oh-my-pi/pi-ai";
|
|
import { getBundledModel } from "@oh-my-pi/pi-catalog/models";
|
|
import * as autoThinkingClassifier from "@oh-my-pi/pi-coding-agent/auto-thinking/classifier";
|
|
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
|
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
|
|
import { AgentSession } from "@oh-my-pi/pi-coding-agent/session/agent-session";
|
|
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
|
import { SessionManager } from "@oh-my-pi/pi-coding-agent/session/session-manager";
|
|
import {
|
|
AUTO_THINKING,
|
|
clampAutoThinkingEffort,
|
|
resolveProvisionalAutoLevel,
|
|
} from "@oh-my-pi/pi-coding-agent/thinking";
|
|
import { TempDir } from "@oh-my-pi/pi-utils";
|
|
import { createAssistantMessage } from "./helpers/agent-session-setup";
|
|
|
|
describe("AgentSession role model thinking behavior", () => {
|
|
let tempDir: TempDir;
|
|
let session: AgentSession;
|
|
let sessionSettings: Settings;
|
|
const authStorages: AuthStorage[] = [];
|
|
|
|
beforeEach(() => {
|
|
tempDir = TempDir.createSync("@pi-role-thinking-");
|
|
});
|
|
|
|
afterEach(async () => {
|
|
vi.restoreAllMocks();
|
|
if (session) {
|
|
await session.dispose();
|
|
}
|
|
for (const authStorage of authStorages.splice(0)) {
|
|
authStorage.close();
|
|
}
|
|
tempDir.removeSync();
|
|
});
|
|
|
|
function getAnthropicModelOrThrow(id: string) {
|
|
const model = getBundledModel("anthropic", id);
|
|
if (!model) throw new Error(`Expected anthropic model ${id} to exist`);
|
|
return model;
|
|
}
|
|
|
|
async function createSession(options: {
|
|
initialModelId: string;
|
|
initialThinkingLevel: Effort;
|
|
modelRoles: Record<string, string>;
|
|
}) {
|
|
const model = getAnthropicModelOrThrow(options.initialModelId);
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: options.initialThinkingLevel,
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models.yml"));
|
|
|
|
sessionSettings = Settings.isolated();
|
|
for (const [role, modelRoleValue] of Object.entries(options.modelRoles)) {
|
|
sessionSettings.setModelRole(role, modelRoleValue);
|
|
}
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager: SessionManager.inMemory(),
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
});
|
|
}
|
|
|
|
it("re-applies explicit role thinking each time that role is selected", async () => {
|
|
const defaultModel = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const slowModel = getAnthropicModelOrThrow("claude-sonnet-4-6");
|
|
|
|
await createSession({
|
|
initialModelId: defaultModel.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: {
|
|
default: `${defaultModel.provider}/${defaultModel.id}`,
|
|
slow: `${slowModel.provider}/${slowModel.id}:off`,
|
|
},
|
|
});
|
|
|
|
const firstSwitch = await session.cycleRoleModels(["default", "slow"]);
|
|
expect(firstSwitch?.role).toBe("slow");
|
|
expect(firstSwitch?.model.id).toBe(slowModel.id);
|
|
expect(firstSwitch?.thinkingLevel).toBe("off");
|
|
expect(session.thinkingLevel).toBe("off");
|
|
|
|
session.setThinkingLevel(Effort.High);
|
|
expect(session.thinkingLevel).toBe(Effort.High);
|
|
|
|
const secondSwitch = await session.cycleRoleModels(["default", "slow"]);
|
|
expect(secondSwitch?.role).toBe("default");
|
|
expect(secondSwitch?.model.id).toBe(defaultModel.id);
|
|
expect(session.thinkingLevel).toBe(Effort.High);
|
|
|
|
const thirdSwitch = await session.cycleRoleModels(["default", "slow"]);
|
|
expect(thirdSwitch?.role).toBe("slow");
|
|
expect(thirdSwitch?.model.id).toBe(slowModel.id);
|
|
expect(thirdSwitch?.thinkingLevel).toBe("off");
|
|
expect(session.thinkingLevel).toBe("off");
|
|
});
|
|
|
|
it("activates auto thinking when cycling into a role whose value carries an explicit :auto suffix", async () => {
|
|
const defaultModel = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const smolModel = getAnthropicModelOrThrow("claude-sonnet-4-6");
|
|
|
|
await createSession({
|
|
initialModelId: defaultModel.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: {
|
|
default: `${defaultModel.provider}/${defaultModel.id}`,
|
|
smol: `${smolModel.provider}/${smolModel.id}:auto`,
|
|
},
|
|
});
|
|
|
|
const toSmol = await session.cycleRoleModels(["default", "smol"]);
|
|
expect(toSmol?.role).toBe("smol");
|
|
expect(toSmol?.model.id).toBe(smolModel.id);
|
|
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
|
});
|
|
|
|
it("preserves current thinking when switching into default/no-suffix role", async () => {
|
|
const defaultModel = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const slowModel = getAnthropicModelOrThrow("claude-sonnet-4-6");
|
|
|
|
await createSession({
|
|
initialModelId: defaultModel.id,
|
|
initialThinkingLevel: Effort.Low,
|
|
modelRoles: {
|
|
default: `${defaultModel.provider}/${defaultModel.id}`,
|
|
slow: `${slowModel.provider}/${slowModel.id}:high`,
|
|
},
|
|
});
|
|
|
|
const toSlow = await session.cycleRoleModels(["default", "slow"]);
|
|
expect(toSlow?.role).toBe("slow");
|
|
expect(toSlow?.thinkingLevel).toBe(Effort.High);
|
|
expect(session.thinkingLevel).toBe(Effort.High);
|
|
|
|
// `medium` is supported on both ladders (4-6 dropped `minimal`), so the
|
|
// selection survives the role switch unclamped.
|
|
session.setThinkingLevel(Effort.Medium);
|
|
expect(session.thinkingLevel).toBe(Effort.Medium);
|
|
|
|
const toDefault = await session.cycleRoleModels(["default", "slow"]);
|
|
expect(toDefault?.role).toBe("default");
|
|
expect(toDefault?.model.id).toBe(defaultModel.id);
|
|
expect(toDefault?.thinkingLevel).toBe(Effort.Medium);
|
|
expect(session.thinkingLevel).toBe(Effort.Medium);
|
|
});
|
|
|
|
it("applies slow role thinking even when plan shares the same model", async () => {
|
|
const defaultModel = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const smolModel = getAnthropicModelOrThrow("claude-sonnet-4-6");
|
|
const slowPlanModel = getAnthropicModelOrThrow("claude-opus-4-5");
|
|
|
|
await createSession({
|
|
initialModelId: defaultModel.id,
|
|
initialThinkingLevel: Effort.Medium,
|
|
modelRoles: {
|
|
default: `${defaultModel.provider}/${defaultModel.id}`,
|
|
smol: `${smolModel.provider}/${smolModel.id}:low`,
|
|
slow: `${slowPlanModel.provider}/${slowPlanModel.id}:high`,
|
|
plan: `${slowPlanModel.provider}/${slowPlanModel.id}:off`,
|
|
},
|
|
});
|
|
|
|
const toSmol = await session.cycleRoleModels(["slow", "default", "smol"]);
|
|
expect(toSmol?.role).toBe("smol");
|
|
expect(toSmol?.thinkingLevel).toBe(Effort.Low);
|
|
expect(session.thinkingLevel).toBe(Effort.Low);
|
|
|
|
const toSlow = await session.cycleRoleModels(["slow", "default", "smol"]);
|
|
expect(toSlow?.role).toBe("slow");
|
|
expect(toSlow?.model.id).toBe(slowPlanModel.id);
|
|
expect(toSlow?.thinkingLevel).toBe(Effort.High);
|
|
expect(session.thinkingLevel).toBe(Effort.High);
|
|
});
|
|
|
|
it("preserves explicit role thinking when updating default model despite unresolved previous model", async () => {
|
|
const defaultModel = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const slowModel = getAnthropicModelOrThrow("claude-sonnet-4-6");
|
|
|
|
await createSession({
|
|
initialModelId: defaultModel.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: {
|
|
default: "anthropic/nonexistent-model:off",
|
|
},
|
|
});
|
|
|
|
await session.setModel(slowModel, "default", { persist: true });
|
|
|
|
expect(sessionSettings.getModelRole("default")).toBe(`${slowModel.provider}/${slowModel.id}:off`);
|
|
});
|
|
|
|
it("clamps unsupported selections from model metadata", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-6");
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: undefined,
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-non-xhigh.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-non-xhigh.yml"));
|
|
|
|
sessionSettings = Settings.isolated();
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager: SessionManager.inMemory(),
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
});
|
|
|
|
session.setThinkingLevel(Effort.XHigh);
|
|
expect(session.thinkingLevel).toBe(Effort.High);
|
|
expect(session.getAvailableThinkingLevels()).not.toContain("xhigh");
|
|
});
|
|
|
|
it("clamps max selections down to the ladder ceiling on models without a max tier", async () => {
|
|
// Budget-mode sonnet-4-5 tops out at xhigh; a max request must clamp down.
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: undefined,
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-max-clamp.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-max-clamp.yml"));
|
|
|
|
sessionSettings = Settings.isolated();
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager: SessionManager.inMemory(),
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
});
|
|
|
|
session.setThinkingLevel(Effort.Max);
|
|
expect(session.thinkingLevel).toBe(Effort.XHigh);
|
|
expect(session.getAvailableThinkingLevels()).not.toContain("max");
|
|
});
|
|
|
|
it("cycles through off and auto before returning to effort levels", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: Effort.High,
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-cycle-thinking.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-cycle-thinking.yml"));
|
|
|
|
sessionSettings = Settings.isolated();
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager: SessionManager.inMemory(),
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
});
|
|
|
|
expect(session.cycleThinkingLevel()).toBe("off");
|
|
expect(session.thinkingLevel).toBe("off");
|
|
expect(agent.state.disableReasoning).toBe(true);
|
|
expect(session.cycleThinkingLevel()).toBe(AUTO_THINKING);
|
|
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
|
expect(session.thinkingLevel).toBe(resolveProvisionalAutoLevel(model));
|
|
expect(agent.state.disableReasoning).toBe(false);
|
|
expect(session.cycleThinkingLevel()).toBe(Effort.Minimal);
|
|
expect(session.thinkingLevel).toBe(Effort.Minimal);
|
|
});
|
|
|
|
it("cycles through max as the final tier on a max-capable model", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-opus-4-7");
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: Effort.XHigh,
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-cycle-max.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-cycle-max.yml"));
|
|
|
|
sessionSettings = Settings.isolated();
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager: SessionManager.inMemory(),
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
});
|
|
|
|
const available = session.getAvailableThinkingLevels();
|
|
expect(available.at(-1)).toBe(Effort.Max);
|
|
|
|
session.setThinkingLevel(Effort.XHigh);
|
|
expect(session.cycleThinkingLevel()).toBe(Effort.Max);
|
|
expect(session.thinkingLevel).toBe(Effort.Max);
|
|
// max is the last tier: the wheel wraps back to off.
|
|
expect(session.cycleThinkingLevel()).toBe("off");
|
|
});
|
|
|
|
it("keeps auto configured while applying the classifier result as the effective level", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
await createSession({
|
|
initialModelId: model.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: { default: `${model.provider}/${model.id}` },
|
|
});
|
|
const promptSpy = vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Medium);
|
|
|
|
session.setThinkingLevel(AUTO_THINKING);
|
|
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
|
expect(session.autoResolvedThinkingLevel()).toBeUndefined();
|
|
|
|
await session.prompt("Implement a focused parser fix");
|
|
|
|
expect(classifierSpy).toHaveBeenCalledTimes(1);
|
|
expect(promptSpy).toHaveBeenCalledTimes(1);
|
|
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
|
expect(session.thinkingLevel).toBe(Effort.Medium);
|
|
expect(session.autoResolvedThinkingLevel()).toBe(Effort.Medium);
|
|
expect(session.agent.state.thinkingLevel).toBe(Effort.Medium);
|
|
});
|
|
|
|
it("keeps auto active on resume (pending until the next turn reclassifies)", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: resolveProvisionalAutoLevel(model),
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-auto-resume.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-auto-resume.yml"));
|
|
const sessionManager = SessionManager.create(tempDir.path(), tempDir.path());
|
|
sessionSettings = Settings.isolated();
|
|
sessionSettings.set("defaultThinkingLevel", AUTO_THINKING);
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager,
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
thinkingLevel: AUTO_THINKING,
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Medium);
|
|
|
|
await session.prompt("Implement a focused parser fix");
|
|
|
|
expect(session.isAutoThinking).toBe(true);
|
|
expect(session.sessionManager.buildSessionContext().thinkingLevel).toBe(Effort.Medium);
|
|
session.sessionManager.appendMessage(createAssistantMessage("done"));
|
|
|
|
const sessionFile = session.sessionFile;
|
|
expect(sessionFile).toBeDefined();
|
|
await session.sessionManager.flush();
|
|
|
|
expect(await session.switchSession(sessionFile!)).toBe(true);
|
|
expect(session.isAutoThinking).toBe(true);
|
|
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
|
// Resumes in auto and pending — not frozen to the last resolved level, and
|
|
// not pre-seeded; the next user turn reclassifies.
|
|
expect(session.autoResolvedThinkingLevel()).toBeUndefined();
|
|
});
|
|
|
|
it("keeps a manual concrete pin (not auto) on resume even when the global default is auto", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: resolveProvisionalAutoLevel(model),
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-manual-resume.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-manual-resume.yml"));
|
|
const sessionManager = SessionManager.create(tempDir.path(), tempDir.path());
|
|
sessionSettings = Settings.isolated();
|
|
sessionSettings.set("defaultThinkingLevel", AUTO_THINKING);
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager,
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
thinkingLevel: AUTO_THINKING,
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Medium);
|
|
|
|
// User pins a concrete level mid-session; it must survive resume as-is and
|
|
// must not be reinterpreted as `auto` just because the global default is auto.
|
|
session.setThinkingLevel(Effort.Low);
|
|
expect(session.isAutoThinking).toBe(false);
|
|
await session.prompt("Pinned concrete turn");
|
|
expect(classifierSpy).not.toHaveBeenCalled();
|
|
session.sessionManager.appendMessage(createAssistantMessage("done"));
|
|
|
|
const sessionFile = session.sessionFile;
|
|
expect(sessionFile).toBeDefined();
|
|
await session.sessionManager.flush();
|
|
|
|
expect(await session.switchSession(sessionFile!)).toBe(true);
|
|
expect(session.isAutoThinking).toBe(false);
|
|
expect(session.configuredThinkingLevel()).toBe(Effort.Low);
|
|
expect(session.thinkingLevel).toBe(Effort.Low);
|
|
});
|
|
|
|
it("persists a concrete pin that matches the auto-resolved effort so resume stays concrete", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: resolveProvisionalAutoLevel(model),
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-pin-eq.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("anthropic", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-pin-eq.yml"));
|
|
const sessionManager = SessionManager.create(tempDir.path(), tempDir.path());
|
|
sessionSettings = Settings.isolated();
|
|
sessionSettings.set("defaultThinkingLevel", AUTO_THINKING);
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager,
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
thinkingLevel: AUTO_THINKING,
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Medium);
|
|
|
|
// Auto resolves to medium.
|
|
await session.prompt("Implement a focused parser fix");
|
|
expect(session.autoResolvedThinkingLevel()).toBe(Effort.Medium);
|
|
|
|
// User then pins the *same* effort: selector changes auto -> medium even though
|
|
// the effort is unchanged, so it must persist as a concrete pin (entry +
|
|
// defaultThinkingLevel), not silently stay `configured: "auto"`.
|
|
session.setThinkingLevel(Effort.Medium, true);
|
|
expect(session.isAutoThinking).toBe(false);
|
|
expect(sessionSettings.get("defaultThinkingLevel")).toBe(Effort.Medium);
|
|
session.sessionManager.appendMessage(createAssistantMessage("done"));
|
|
|
|
const sessionFile = session.sessionFile;
|
|
expect(sessionFile).toBeDefined();
|
|
await session.sessionManager.flush();
|
|
|
|
expect(await session.switchSession(sessionFile!)).toBe(true);
|
|
expect(session.isAutoThinking).toBe(false);
|
|
expect(session.configuredThinkingLevel()).toBe(Effort.Medium);
|
|
});
|
|
|
|
it("falls back to a concrete auto level when classification fails", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
await createSession({
|
|
initialModelId: model.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: { default: `${model.provider}/${model.id}` },
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockRejectedValue(new Error("classifier down"));
|
|
|
|
session.setThinkingLevel(AUTO_THINKING);
|
|
const fallback = resolveProvisionalAutoLevel(model);
|
|
await session.prompt("Investigate a regression");
|
|
|
|
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
|
expect(session.thinkingLevel).toBe(fallback);
|
|
expect(session.autoResolvedThinkingLevel()).toBe(fallback);
|
|
expect(session.agent.state.thinkingLevel).toBe(fallback);
|
|
});
|
|
|
|
it("skips classification for synthetic turns", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
await createSession({
|
|
initialModelId: model.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: { default: `${model.provider}/${model.id}` },
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.XHigh);
|
|
|
|
session.setThinkingLevel(AUTO_THINKING);
|
|
const provisional = resolveProvisionalAutoLevel(model);
|
|
await session.prompt("Synthetic maintenance turn", { synthetic: true });
|
|
|
|
expect(classifierSpy).not.toHaveBeenCalled();
|
|
expect(session.configuredThinkingLevel()).toBe(AUTO_THINKING);
|
|
expect(session.thinkingLevel).toBe(provisional);
|
|
expect(session.autoResolvedThinkingLevel()).toBeUndefined();
|
|
});
|
|
|
|
it("maps ultrathink prompts to the model's highest supported level, clamped below max", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
await createSession({
|
|
initialModelId: model.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: { default: `${model.provider}/${model.id}` },
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Low);
|
|
|
|
session.setThinkingLevel(AUTO_THINKING);
|
|
// sonnet-4-5 has no max tier, so the ultrathink jump clamps to xhigh.
|
|
const expected = clampAutoThinkingEffort(model, Effort.Max);
|
|
expect(expected).toBe(Effort.XHigh);
|
|
await session.prompt("ultrathink through the unsafe refactor");
|
|
|
|
expect(classifierSpy).not.toHaveBeenCalled();
|
|
expect(session.thinkingLevel).toBe(expected);
|
|
expect(session.autoResolvedThinkingLevel()).toBe(expected);
|
|
});
|
|
|
|
it("resolves ultrathink to max on max-capable models", async () => {
|
|
const model = getAnthropicModelOrThrow("claude-opus-4-7");
|
|
await createSession({
|
|
initialModelId: model.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: { default: `${model.provider}/${model.id}` },
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.Low);
|
|
|
|
session.setThinkingLevel(AUTO_THINKING);
|
|
await session.prompt("ultrathink through the unsafe refactor");
|
|
|
|
expect(classifierSpy).not.toHaveBeenCalled();
|
|
expect(session.thinkingLevel).toBe(Effort.Max);
|
|
expect(session.autoResolvedThinkingLevel()).toBe(Effort.Max);
|
|
});
|
|
|
|
it("keeps auto effectively off for non-reasoning models", async () => {
|
|
const model = getBundledModel("openai", "gpt-4o-mini");
|
|
if (!model) throw new Error("Expected bundled gpt-4o-mini model");
|
|
const agent = new Agent({
|
|
initialState: {
|
|
model,
|
|
systemPrompt: ["Test"],
|
|
tools: [],
|
|
messages: [],
|
|
thinkingLevel: undefined,
|
|
},
|
|
});
|
|
const authStorage = await AuthStorage.create(path.join(tempDir.path(), "testauth-non-reasoning-auto.db"));
|
|
authStorages.push(authStorage);
|
|
authStorage.setRuntimeApiKey("openai", "test-key");
|
|
const modelRegistry = new ModelRegistry(authStorage, path.join(tempDir.path(), "models-non-reasoning-auto.yml"));
|
|
sessionSettings = Settings.isolated();
|
|
sessionSettings.set("defaultThinkingLevel", AUTO_THINKING);
|
|
session = new AgentSession({
|
|
agent,
|
|
sessionManager: SessionManager.inMemory(),
|
|
settings: sessionSettings,
|
|
modelRegistry,
|
|
thinkingLevel: AUTO_THINKING,
|
|
});
|
|
vi.spyOn(session.agent, "prompt").mockResolvedValue(undefined);
|
|
const classifierSpy = vi.spyOn(autoThinkingClassifier, "classifyDifficulty").mockResolvedValue(Effort.XHigh);
|
|
|
|
expect(session.isAutoThinking).toBe(true);
|
|
expect(session.thinkingLevel).toBeUndefined();
|
|
expect(session.agent.state.thinkingLevel).toBeUndefined();
|
|
|
|
await session.prompt("Implement a tiny change");
|
|
|
|
expect(classifierSpy).not.toHaveBeenCalled();
|
|
expect(session.thinkingLevel).toBeUndefined();
|
|
expect(session.agent.state.thinkingLevel).toBeUndefined();
|
|
expect(session.autoResolvedThinkingLevel()).toBeUndefined();
|
|
});
|
|
|
|
it("ignores a stale recorded role and cycles from the active model", async () => {
|
|
const defaultModel = getAnthropicModelOrThrow("claude-sonnet-4-5");
|
|
const slowModel = getAnthropicModelOrThrow("claude-sonnet-4-6");
|
|
|
|
await createSession({
|
|
initialModelId: defaultModel.id,
|
|
initialThinkingLevel: Effort.High,
|
|
modelRoles: {
|
|
default: `${defaultModel.provider}/${defaultModel.id}`,
|
|
slow: `${slowModel.provider}/${slowModel.id}`,
|
|
},
|
|
});
|
|
|
|
// Record a model_change for the "slow" role WITHOUT switching the
|
|
// active model — the session still runs the default model. This is the
|
|
// stale state left behind when the model is changed through another
|
|
// surface (alt+m, temporary model, /model) after a role cycle.
|
|
session.sessionManager.appendModelChange(`${slowModel.provider}/${slowModel.id}`, "slow");
|
|
expect(session.sessionManager.getLastModelChangeRole()).toBe("slow");
|
|
expect(session.model?.id).toBe(defaultModel.id);
|
|
|
|
// The recorded role's resolved model (4-6) no longer equals the active
|
|
// model (4-5), so the cycle position must fall back to model equality
|
|
// and point at "default" — not trust the stale "slow" slot.
|
|
const cycle = session.getRoleModelCycle(["default", "slow"]);
|
|
if (!cycle) throw new Error("Expected a resolved role model cycle");
|
|
expect(cycle.models.map(entry => entry.role)).toEqual(["default", "slow"]);
|
|
expect(cycle.currentIndex).toBe(0);
|
|
expect(cycle.models[cycle.currentIndex]?.role).toBe("default");
|
|
|
|
// Cycling advances from the ACTIVE model's position: default → slow.
|
|
// With the stale slot trusted, the cycle would compute slow → default
|
|
// and "switch" right back onto the model already running.
|
|
const result = await session.cycleRoleModels(["default", "slow"]);
|
|
expect(result?.role).toBe("slow");
|
|
expect(result?.model.id).toBe(slowModel.id);
|
|
expect(session.model?.id).toBe(slowModel.id);
|
|
});
|
|
});
|