Files
oh-my-pi/packages/coding-agent/test/issue-1528-discovery-default-max-tokens.test.ts
T
roboomp bf75d80836 fix(coding-agent): resolve bundled reference in discoverOpenAIModelsList
Thin OpenAI-compatible proxies that omit context_length / max_model_len on
/v1/models made every discovered model fall back to
DISCOVERY_DEFAULT_CONTEXT_WINDOW (128K/33K), even when the id matched a
bundled model with a much larger intrinsic window. discoverProxyModels
and discoverLiteLLMModels already resolve ids against the bundled
reference index; discoverOpenAIModelsList (which also backs lm-studio
discovery) now does the same.

Behavior:
- Build the reference index once outside the loop and resolve each item
  via resolveModelReference().
- contextWindow precedence keeps provider-reported values authoritative:
  item.max_model_len ?? item.context_length ?? nativeMetadata?.contextWindow
  ?? reference?.contextWindow ?? DISCOVERY_DEFAULT_CONTEXT_WINDOW.
- maxTokens uses reference?.maxTokens when available, otherwise the
  api-specific discovery default, capped at contextWindow so a bundled
  ref for a larger sibling can never over-request output tokens.
- name / reasoning / thinking / input inherit from the reference; native
  lm-studio metadata still wins for input modality.
- Provider-specific baseUrl, headers, and local-unknown cost stay local.
- OpenAI-compat flags stay conservative (supportsStore / supportsDeveloperRole
  / supportsReasoningEffort all false) to match the proxy sibling.

Also updated two pre-existing regression tests that used
deepseek-v4-pro / deepseek-r1 / DeepSeek-V4-Flash as stand-in "fictional"
ids to exercise the default-fallback branch. Those model names have since
been added to the bundled catalog, so the tests were renamed to
vllm-lab-fork-* ids that unambiguously miss the reference index while
preserving each test's original default-fallback intent.

Fixes #3983
2026-07-01 03:07:14 +00:00

195 lines
6.7 KiB
TypeScript

import { afterEach, beforeEach, describe, expect, test } from "bun:test";
import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import type { FetchImpl } from "@oh-my-pi/pi-ai/types";
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
/**
* Issue #1528: auto-discovered OpenAI-compatible models defaulted to
* `maxTokens: 8192`, which made providers (DeepSeek, etc.) drop the streaming
* connection mid-response on large `write`/`edit` tool calls and surfaced as
* Bun's opaque "socket connection was closed unexpectedly". The cap is now
* `DISCOVERY_DEFAULT_MAX_TOKENS = 32_768` (`packages/coding-agent/src/config/
* model-registry.ts`). These tests pin the externally observable default for
* every discovery branch that previously hardcoded 8192.
*/
describe("issue #1528 discovery maxTokens default", () => {
let tempDir: string;
let modelsPath: string;
let authStorage: AuthStorage;
beforeEach(async () => {
tempDir = path.join(os.tmpdir(), `pi-test-issue-1528-${Snowflake.next()}`);
fs.mkdirSync(tempDir, { recursive: true });
modelsPath = path.join(tempDir, "models.yml");
authStorage = await AuthStorage.create(path.join(tempDir, "auth.db"));
});
afterEach(() => {
authStorage.close();
if (tempDir && fs.existsSync(tempDir)) {
removeSyncWithRetries(tempDir);
}
});
test("openai-models-list discovery returns maxTokens=32768 when API advertises no output limit", async () => {
fs.writeFileSync(
modelsPath,
[
"providers:",
" deepseek-compat:",
" baseUrl: https://api.example.com/v1",
" apiKey: sk-test",
" api: openai-completions",
" auth: apiKey",
" discovery:",
" type: openai-models-list",
].join("\n"),
);
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url !== "https://api.example.com/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
return new Response(JSON.stringify({ data: [{ id: "vllm-lab-fork-a1" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("deepseek-compat");
const model = registry.find("deepseek-compat", "vllm-lab-fork-a1");
expect(model?.maxTokens).toBe(32_768);
});
test("proxy (anthropic+openai) discovery returns maxTokens=32768 for openai-routed models without bundled limits", async () => {
fs.writeFileSync(
modelsPath,
[
"providers:",
" newapi-proxy:",
" baseUrl: https://proxy.example.com/v1",
" apiKey: sk-test",
" api: openai-completions",
" auth: apiKey",
" discovery:",
" type: proxy",
].join("\n"),
);
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url !== "https://proxy.example.com/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
return new Response(
JSON.stringify({
data: [{ id: "newapi-private-openai-model", supported_endpoint_types: ["openai"] }],
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
};
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("newapi-proxy");
const model = registry.find("newapi-proxy", "newapi-private-openai-model");
expect(model?.maxTokens).toBe(32_768);
});
test("proxy discovery keeps anthropic-routed models at the 8192 default to stay under Claude 3.x output caps", async () => {
// Anthropic's stream converter sends `max_tokens` as
// `(model.maxTokens / 3) | 0`. The raised 32K discovery cap would
// surface as 10,922 — above the 8,192 hard cap on classic Claude 3.x
// models — so the proxy branch keeps the conservative 8K default on
// the anthropic route.
fs.writeFileSync(
modelsPath,
[
"providers:",
" newapi-proxy:",
" baseUrl: https://proxy.example.com/v1",
" apiKey: sk-test",
" api: openai-completions",
" auth: apiKey",
" discovery:",
" type: proxy",
].join("\n"),
);
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url !== "https://proxy.example.com/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
return new Response(
JSON.stringify({
data: [
{ id: "claude-3-5-sonnet", supported_endpoint_types: ["anthropic"] },
{ id: "claude-3-5-haiku", supported_endpoint_types: ["anthropic", "openai"] },
],
}),
{ status: 200, headers: { "Content-Type": "application/json" } },
);
};
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("newapi-proxy");
const sonnet = registry.find("newapi-proxy", "claude-3-5-sonnet");
expect(sonnet?.api).toBe("anthropic-messages");
expect(sonnet?.maxTokens).toBe(8192);
// Dual-endpoint advertisements prefer the anthropic route in the proxy
// branch, so they also stay capped at 8K.
const haiku = registry.find("newapi-proxy", "claude-3-5-haiku");
expect(haiku?.api).toBe("anthropic-messages");
expect(haiku?.maxTokens).toBe(8192);
});
test("openai-models-list discovery keeps anthropic-messages providers at the 8192 default", async () => {
// The validator allows `api: anthropic-messages` with a bare
// openai-models-list discovery (e.g. third-party Anthropic catalogs
// served behind a `/v1/models` endpoint). Same divisor reasoning as
// the proxy branch applies: 32K would surface as 10,922 requested
// output tokens, above the Claude 3.x hard cap.
fs.writeFileSync(
modelsPath,
[
"providers:",
" third-party-anthropic:",
" baseUrl: https://anthropic-reseller.example.com/v1",
" apiKey: sk-test",
" api: anthropic-messages",
" auth: apiKey",
" discovery:",
" type: openai-models-list",
].join("\n"),
);
const fetchMock: FetchImpl = async input => {
const url = String(input);
if (url !== "https://anthropic-reseller.example.com/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
return new Response(JSON.stringify({ data: [{ id: "claude-3-5-sonnet" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("third-party-anthropic");
const sonnet = registry.find("third-party-anthropic", "claude-3-5-sonnet");
expect(sonnet?.api).toBe("anthropic-messages");
expect(sonnet?.maxTokens).toBe(8192);
});
});