fix(ai): corrected thinking config format and model context windows across 100+ definitions
- Fixed thinking configuration format by replacing `levels` array with `minLevel`/`maxLevel` properties across 100+ model definitions. - Corrected GPT-5.4 mini/nano context window from 400000 to 272000 tokens for accurate token limit reporting. - Normalized GPT-5.4 variant priority handling to use parsed variant instead of raw model IDs for consistent behavior. - Added "mini" variant support to OpenAI model parsing regex and updated thinking mode configuration for Claude models. - Fixed test robustness by replacing exact string matching with numeric range comparison to handle BSD seq notation on macOS. - Corrected model generation script execution order to apply policy overrides before promotion target linking.
This commit is contained in:
@@ -1,6 +1,16 @@
|
||||
# Changelog
|
||||
|
||||
## [Unreleased]
|
||||
### Changed
|
||||
|
||||
- Updated thinking configuration format from `levels` array to `minLevel` and `maxLevel` properties for improved clarity
|
||||
- Corrected context window from 400000 to 272000 tokens for GPT-5.4 mini and nano variants on Codex transport
|
||||
- Normalized GPT-5.4 variant priority handling to use parsed variant instead of special-casing raw model IDs
|
||||
- Added support for `mini` variant in OpenAI model parsing regex
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed inconsistent thinking level configuration across multiple model definitions
|
||||
|
||||
## [13.14.0] - 2026-03-20
|
||||
|
||||
|
||||
@@ -304,9 +304,6 @@ async function generateModels() {
|
||||
}
|
||||
}
|
||||
|
||||
applyGeneratedModelPolicies(allModels);
|
||||
linkSparkPromotionTargets(allModels);
|
||||
|
||||
// Merge previous models.json entries as fallback for any provider/model
|
||||
// not fetched dynamically. This replaces all hardcoded fallback lists —
|
||||
// static-only providers (vertex, gemini-cli), auth-gated providers when
|
||||
@@ -326,6 +323,8 @@ async function generateModels() {
|
||||
|
||||
allModels = applyGlobalModelsDevFallback(allModels, modelsDevModels);
|
||||
allModels = applyPremiumMultiplierOverrides(allModels);
|
||||
applyGeneratedModelPolicies(allModels);
|
||||
linkSparkPromotionTargets(allModels);
|
||||
|
||||
// Group by provider and sort each provider's models
|
||||
const providers: Record<string, Record<string, Model>> = {};
|
||||
|
||||
@@ -40,7 +40,13 @@ type SemVer = {
|
||||
|
||||
type GeminiKind = "pro" | "flash";
|
||||
type AnthropicKind = "opus" | "sonnet";
|
||||
type OpenAIVariant = "base" | "codex" | "codex-max" | "codex-mini" | "codex-spark" | "max" | "nano";
|
||||
type OpenAIVariant = "base" | "codex" | "codex-max" | "codex-mini" | "codex-spark" | "mini" | "max" | "nano";
|
||||
|
||||
const CODEX_GPT_5_4_PRIORITY_BY_VARIANT: Partial<Record<OpenAIVariant, number>> = {
|
||||
base: 0,
|
||||
mini: 1,
|
||||
nano: 2,
|
||||
};
|
||||
|
||||
interface GeminiModel {
|
||||
family: "gemini";
|
||||
@@ -299,9 +305,17 @@ function applyOpenAICatalogPolicy(model: ApiModel<Api>, parsedModel: OpenAIModel
|
||||
return;
|
||||
}
|
||||
// GPT-5.4 mini/nano use plain OpenAI IDs on the Codex transport, but Codex still
|
||||
// enforces the lower prompt budget for these variants.
|
||||
if (model.api === "openai-codex-responses" && (model.id === "gpt-5.4-mini" || model.id === "gpt-5.4-nano")) {
|
||||
model.contextWindow = 272000;
|
||||
// enforces the lower prompt budget for these variants. Codex discovery can also
|
||||
// report inconsistent priorities for the GPT-5.4 family, so normalize by parsed
|
||||
// variant instead of special-casing raw model ids.
|
||||
if (model.api === "openai-codex-responses" && semverEqual(parsedModel.version, "5.4")) {
|
||||
const normalizedPriority = CODEX_GPT_5_4_PRIORITY_BY_VARIANT[parsedModel.variant];
|
||||
if (normalizedPriority !== undefined) {
|
||||
model.priority = normalizedPriority;
|
||||
}
|
||||
if (parsedModel.variant === "mini" || parsedModel.variant === "nano") {
|
||||
model.contextWindow = 272000;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -489,7 +503,7 @@ function parseAnthropicModel(modelId: string): AnthropicModel | null {
|
||||
}
|
||||
|
||||
function parseOpenAIModel(modelId: string): OpenAIModel | null {
|
||||
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|max|nano))?\b/.exec(modelId);
|
||||
const match = /gpt-(\d+(?:\.\d+){0,2})(?:-(codex-spark|codex-mini|codex-max|codex|mini|max|nano))?\b/.exec(modelId);
|
||||
if (!match) {
|
||||
return null;
|
||||
}
|
||||
|
||||
+157
-354
File diff suppressed because it is too large
Load Diff
@@ -207,6 +207,19 @@ describe("generated model policies", () => {
|
||||
contextWindow: 400000,
|
||||
maxTokens: 32000,
|
||||
},
|
||||
{
|
||||
id: "gpt-5.4-mini",
|
||||
name: "GPT-5.4 mini",
|
||||
api: "openai-codex-responses",
|
||||
provider: "openai-codex",
|
||||
baseUrl: "https://example.com",
|
||||
reasoning: true,
|
||||
input: ["text"],
|
||||
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
|
||||
contextWindow: 400000,
|
||||
maxTokens: 32000,
|
||||
priority: 2,
|
||||
},
|
||||
];
|
||||
|
||||
applyGeneratedModelPolicies(models);
|
||||
@@ -227,6 +240,8 @@ describe("generated model policies", () => {
|
||||
expect(models[1]?.cost.cacheWrite).toBe(6.25);
|
||||
expect(models[1]?.contextWindow).toBe(1000000);
|
||||
expect(models[2]?.contextWindow).toBe(272000);
|
||||
expect(models[3]?.contextWindow).toBe(272000);
|
||||
expect(models[3]?.priority).toBe(1);
|
||||
});
|
||||
|
||||
it("links spark variants to their base models", () => {
|
||||
|
||||
@@ -248,10 +248,15 @@ describe("executeBash", () => {
|
||||
// Truncated output should be within the spill threshold
|
||||
expect(result.outputBytes).toBeLessThanOrEqual(DEFAULT_MAX_BYTES);
|
||||
|
||||
// The tail of the output should contain numbers near the end of the range.
|
||||
// The exact last number may be split across a truncation boundary, so
|
||||
// check for a number within the last 1000 lines.
|
||||
expect(result.output).toContain(String(lineCount - 500));
|
||||
// The tail should still contain numeric values near the end of the range.
|
||||
// BSD `seq` on macOS formats large numbers in scientific notation, so parse
|
||||
// the final lines numerically instead of matching one exact decimal string.
|
||||
const tailValues = result.output
|
||||
.split("\n")
|
||||
.slice(-1000)
|
||||
.map(line => Number(line.trim()))
|
||||
.filter(Number.isFinite);
|
||||
expect(tailValues.some(value => value >= lineCount - 500 && value <= lineCount)).toBe(true);
|
||||
|
||||
// With 64KB read buffer, ~40MB should produce ~600 chunks, not 5M.
|
||||
// Allow generous headroom but ensure it's orders of magnitude below lineCount.
|
||||
|
||||
@@ -1,121 +0,0 @@
|
||||
import { beforeEach, describe, expect, it, mock, vi } from "bun:test";
|
||||
import type { SourceMeta } from "../src/capability/types";
|
||||
import type { MCPServerConfig, MCPServerConnection, MCPToolDefinition, MCPTransport } from "../src/mcp/types";
|
||||
|
||||
const connectToServerMock = vi.fn();
|
||||
const disconnectServerMock = vi.fn();
|
||||
const listToolsMock = vi.fn();
|
||||
|
||||
mock.module("../src/mcp/client", () => ({
|
||||
connectToServer: connectToServerMock,
|
||||
disconnectServer: disconnectServerMock,
|
||||
getPrompt: vi.fn(),
|
||||
listPrompts: vi.fn(),
|
||||
listResources: vi.fn(),
|
||||
listResourceTemplates: vi.fn(),
|
||||
listTools: listToolsMock,
|
||||
readResource: vi.fn(),
|
||||
serverSupportsPrompts: vi.fn(() => false),
|
||||
serverSupportsResources: vi.fn(() => false),
|
||||
subscribeToResources: vi.fn(),
|
||||
unsubscribeFromResources: vi.fn(),
|
||||
}));
|
||||
|
||||
import { MCPManager } from "../src/mcp/manager";
|
||||
|
||||
function createTransport(): MCPTransport {
|
||||
return {
|
||||
connected: true,
|
||||
async request() {
|
||||
throw new Error("request not implemented");
|
||||
},
|
||||
async notify() {},
|
||||
async close() {},
|
||||
};
|
||||
}
|
||||
|
||||
function createConnection(name: string, config: MCPServerConfig, transport: MCPTransport): MCPServerConnection {
|
||||
return {
|
||||
name,
|
||||
config,
|
||||
transport,
|
||||
serverInfo: { name: "mock", version: "1.0.0" },
|
||||
capabilities: { tools: {} },
|
||||
};
|
||||
}
|
||||
|
||||
function createSource(path: string): SourceMeta {
|
||||
return {
|
||||
provider: "mcp-json",
|
||||
providerName: "MCP JSON",
|
||||
path,
|
||||
level: "project",
|
||||
};
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
connectToServerMock.mockReset();
|
||||
disconnectServerMock.mockReset();
|
||||
listToolsMock.mockReset();
|
||||
|
||||
disconnectServerMock.mockImplementation(async (connection: MCPServerConnection) => {
|
||||
await connection.transport.close();
|
||||
});
|
||||
listToolsMock.mockResolvedValue([] satisfies MCPToolDefinition[]);
|
||||
});
|
||||
|
||||
describe("MCPManager reconnect behavior", () => {
|
||||
it("wires stdio transport onClose to reconnectServer", async () => {
|
||||
const manager = new MCPManager("/tmp");
|
||||
const serverName = "stdio-server";
|
||||
const config: MCPServerConfig = { type: "stdio", command: "mock-server" };
|
||||
const transport = createTransport();
|
||||
connectToServerMock.mockResolvedValueOnce(createConnection(serverName, config, transport));
|
||||
|
||||
await manager.connectServers({ [serverName]: config }, { [serverName]: createSource("/tmp/.mcp.json") });
|
||||
|
||||
const connection = manager.getConnection(serverName);
|
||||
expect(connection).toBeDefined();
|
||||
expect(typeof connection?.transport.onClose).toBe("function");
|
||||
|
||||
const reconnectSpy = vi.spyOn(manager, "reconnectServer").mockResolvedValue(null);
|
||||
connection?.transport.onClose?.();
|
||||
expect(reconnectSpy).toHaveBeenCalledWith(serverName);
|
||||
});
|
||||
|
||||
it("stops reconnect retries after disconnectAll increments epoch", async () => {
|
||||
const manager = new MCPManager("/tmp");
|
||||
const serverName = "epoch-server";
|
||||
const config: MCPServerConfig = { type: "stdio", command: "mock-server" };
|
||||
const firstReconnectAttempt = Promise.withResolvers<void>();
|
||||
let connectCalls = 0;
|
||||
|
||||
connectToServerMock.mockImplementation(async () => {
|
||||
connectCalls += 1;
|
||||
if (connectCalls === 1) {
|
||||
return createConnection(serverName, config, createTransport());
|
||||
}
|
||||
if (connectCalls === 2) {
|
||||
firstReconnectAttempt.resolve();
|
||||
throw new Error("ECONNREFUSED");
|
||||
}
|
||||
return createConnection(serverName, config, createTransport());
|
||||
});
|
||||
|
||||
await manager.connectServers({ [serverName]: config }, { [serverName]: createSource("/tmp/.mcp.json") });
|
||||
|
||||
const sleepGate = Promise.withResolvers<void>();
|
||||
const sleepSpy = vi.spyOn(Bun, "sleep").mockImplementation(async () => sleepGate.promise);
|
||||
|
||||
const reconnectPromise = manager.reconnectServer(serverName);
|
||||
await firstReconnectAttempt.promise;
|
||||
await manager.disconnectAll();
|
||||
sleepGate.resolve();
|
||||
|
||||
const result = await reconnectPromise;
|
||||
expect(result).toBeNull();
|
||||
expect(connectCalls).toBe(2);
|
||||
|
||||
sleepSpy.mockRestore();
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user