fix(ai): corrected thinking config format and model context windows across 100+ definitions

- Fixed thinking configuration format by replacing `levels` array with `minLevel`/`maxLevel` properties across 100+ model definitions.
- Corrected GPT-5.4 mini/nano context window from 400000 to 272000 tokens for accurate token limit reporting.
- Normalized GPT-5.4 variant priority handling to use parsed variant instead of raw model IDs for consistent behavior.
- Added "mini" variant support to OpenAI model parsing regex and updated thinking mode configuration for Claude models.
- Fixed test robustness by replacing exact string matching with numeric range comparison to handle BSD seq notation on macOS.
- Corrected model generation script execution order to apply policy overrides before promotion target linking.
This commit is contained in:
can1357
2026-03-21 17:33:51 +01:00
parent b89631541a
commit a4026c588e
7 changed files with 212 additions and 487 deletions
+10
View File
@@ -1,6 +1,16 @@
# Changelog
## [Unreleased]
### Changed
- Updated thinking configuration format from `levels` array to `minLevel` and `maxLevel` properties for improved clarity
- Corrected context window from 400000 to 272000 tokens for GPT-5.4 mini and nano variants on Codex transport
- Normalized GPT-5.4 variant priority handling to use parsed variant instead of special-casing raw model IDs
- Added support for `mini` variant in OpenAI model parsing regex
### Fixed
- Fixed inconsistent thinking level configuration across multiple model definitions
## [13.14.0] - 2026-03-20
+2 -3
View File
@@ -304,9 +304,6 @@ async function generateModels() {
}
}
applyGeneratedModelPolicies(allModels);
linkSparkPromotionTargets(allModels);
// Merge previous models.json entries as fallback for any provider/model
// not fetched dynamically. This replaces all hardcoded fallback lists —
// static-only providers (vertex, gemini-cli), auth-gated providers when
@@ -326,6 +323,8 @@ async function generateModels() {
allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels);
allModels = applyPremiumMultiplierOverrides(allModels);
applyGeneratedModelPolicies(allModels);
linkSparkPromotionTargets(allModels);
// Group by provider and sort each provider's models
const providers: Record<string, Record<string, Model>> = {};
+19 -5
View File
@@ -40,7 +40,13 @@ type SemVer = {
type GeminiKind = "pro" | "flash";
type AnthropicKind = "opus" | "sonnet";
type OpenAIVariant = "base" | "codex" | "codex-max" | "codex-mini" | "codex-spark" | "max" | "nano";
type OpenAIVariant = "base" | "codex" | "codex-max" | "codex-mini" | "codex-spark" | "mini" | "max" | "nano";
const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>> = {
base: 0,
mini: 1,
nano: 2,
};
interface GeminiModel {
family: "gemini";
@@ -299,9 +305,17 @@ function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel
return;
}
// GPT-5.4 mini/nano use plain OpenAI IDs on the Codex transport, but Codex still
// enforces the lower prompt budget for these variants.
if (model.api === "openai-codex-responses" && (model.id === "gpt-5.4-mini" || model.id === "gpt-5.4-nano")) {
model.contextWindow = 272000;
// enforces the lower prompt budget for these variants. Codex discovery can also
// report inconsistent priorities for the GPT-5.4 family, so normalize by parsed
// variant instead of special-casing raw model ids.
if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.4")) {
const normalizedPriority = CODEX_GPT_5_4_PRIORITY_BY_VARIANT[parsedModel.variant];
if (normalizedPriority !== undefined) {
model.priority = normalizedPriority;
}
if (parsedModel.variant === "mini" || parsedModel.variant === "nano") {
model.contextWindow = 272000;
}
}
}
@@ -489,7 +503,7 @@ function parseAnthropicModel(modelId: string): AnthropicModel | null {
}
function parseOpenAIModel(modelId: string): OpenAIModel | null {
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|max|nano))?\b/.exec(modelId);
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))?\b/.exec(modelId);
if (!match) {
return null;
}
File diff suppressed because it is too large Load Diff
+15
View File
@@ -207,6 +207,19 @@ describe("generated model policies", () => {
contextWindow: 400000,
maxTokens: 32000,
},
{
id: "gpt-5.4-mini",
name: "GPT-5.4 mini",
api: "openai-codex-responses",
provider: "openai-codex",
baseUrl: "https://example.com",
reasoning: true,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: 400000,
maxTokens: 32000,
priority: 2,
},
];
applyGeneratedModelPolicies(models);
@@ -227,6 +240,8 @@ describe("generated model policies", () => {
expect(models[1]?.cost.cacheWrite).toBe(6.25);
expect(models[1]?.contextWindow).toBe(1000000);
expect(models[2]?.contextWindow).toBe(272000);
expect(models[3]?.contextWindow).toBe(272000);
expect(models[3]?.priority).toBe(1);
});
it("links spark variants to their base models", () => {
@@ -248,10 +248,15 @@ describe("executeBash", () => {
// Truncated output should be within the spill threshold
expect(result.outputBytes).toBeLessThanOrEqual(DEFAULT_MAX_BYTES);
// The tail of the output should contain numbers near the end of the range.
// The exact last number may be split across a truncation boundary, so
// check for a number within the last 1000 lines.
expect(result.output).toContain(String(lineCount - 500));
// The tail should still contain numeric values near the end of the range.
// BSD `seq` on macOS formats large numbers in scientific notation, so parse
// the final lines numerically instead of matching one exact decimal string.
const tailValues = result.output
.split("\n")
.slice(-1000)
.map(line => Number(line.trim()))
.filter(Number.isFinite);
expect(tailValues.some(value => value >= lineCount - 500 && value <= lineCount)).toBe(true);
// With 64KB read buffer, ~40MB should produce ~600 chunks, not 5M.
// Allow generous headroom but ensure it's orders of magnitude below lineCount.
@@ -1,121 +0,0 @@
import { beforeEach, describe, expect, it, mock, vi } from "bun:test";
import type { SourceMeta } from "../src/capability/types";
import type { MCPServerConfig, MCPServerConnection, MCPToolDefinition, MCPTransport } from "../src/mcp/types";
const connectToServerMock = vi.fn();
const disconnectServerMock = vi.fn();
const listToolsMock = vi.fn();
mock.module("../src/mcp/client", () => ({
connectToServer: connectToServerMock,
disconnectServer: disconnectServerMock,
getPrompt: vi.fn(),
listPrompts: vi.fn(),
listResources: vi.fn(),
listResourceTemplates: vi.fn(),
listTools: listToolsMock,
readResource: vi.fn(),
serverSupportsPrompts: vi.fn(() => false),
serverSupportsResources: vi.fn(() => false),
subscribeToResources: vi.fn(),
unsubscribeFromResources: vi.fn(),
}));
import { MCPManager } from "../src/mcp/manager";
function createTransport(): MCPTransport {
return {
connected: true,
async request() {
throw new Error("request not implemented");
},
async notify() {},
async close() {},
};
}
function createConnection(name: string, config: MCPServerConfig, transport: MCPTransport): MCPServerConnection {
return {
name,
config,
transport,
serverInfo: { name: "mock", version: "1.0.0" },
capabilities: { tools: {} },
};
}
function createSource(path: string): SourceMeta {
return {
provider: "mcp-json",
providerName: "MCP JSON",
path,
level: "project",
};
}
beforeEach(() => {
connectToServerMock.mockReset();
disconnectServerMock.mockReset();
listToolsMock.mockReset();
disconnectServerMock.mockImplementation(async (connection: MCPServerConnection) => {
await connection.transport.close();
});
listToolsMock.mockResolvedValue([] satisfies MCPToolDefinition[]);
});
describe("MCPManager reconnect behavior", () => {
it("wires stdio transport onClose to reconnectServer", async () => {
const manager = new MCPManager("/tmp");
const serverName = "stdio-server";
const config: MCPServerConfig = { type: "stdio", command: "mock-server" };
const transport = createTransport();
connectToServerMock.mockResolvedValueOnce(createConnection(serverName, config, transport));
await manager.connectServers({ [serverName]: config }, { [serverName]: createSource("/tmp/.mcp.json") });
const connection = manager.getConnection(serverName);
expect(connection).toBeDefined();
expect(typeof connection?.transport.onClose).toBe("function");
const reconnectSpy = vi.spyOn(manager, "reconnectServer").mockResolvedValue(null);
connection?.transport.onClose?.();
expect(reconnectSpy).toHaveBeenCalledWith(serverName);
});
it("stops reconnect retries after disconnectAll increments epoch", async () => {
const manager = new MCPManager("/tmp");
const serverName = "epoch-server";
const config: MCPServerConfig = { type: "stdio", command: "mock-server" };
const firstReconnectAttempt = Promise.withResolvers<void>();
let connectCalls = 0;
connectToServerMock.mockImplementation(async () => {
connectCalls += 1;
if (connectCalls === 1) {
return createConnection(serverName, config, createTransport());
}
if (connectCalls === 2) {
firstReconnectAttempt.resolve();
throw new Error("ECONNREFUSED");
}
return createConnection(serverName, config, createTransport());
});
await manager.connectServers({ [serverName]: config }, { [serverName]: createSource("/tmp/.mcp.json") });
const sleepGate = Promise.withResolvers<void>();
const sleepSpy = vi.spyOn(Bun, "sleep").mockImplementation(async () => sleepGate.promise);
const reconnectPromise = manager.reconnectServer(serverName);
await firstReconnectAttempt.promise;
await manager.disconnectAll();
sleepGate.resolve();
const result = await reconnectPromise;
expect(result).toBeNull();
expect(connectCalls).toBe(2);
sleepSpy.mockRestore();
});
});