bf75d80836
Thin OpenAI-compatible proxies that omit context_length / max_model_len on /v1/models made every discovered model fall back to DISCOVERY_DEFAULT_CONTEXT_WINDOW (128K/33K), even when the id matched a bundled model with a much larger intrinsic window. discoverProxyModels and discoverLiteLLMModels already resolve ids against the bundled reference index; discoverOpenAIModelsList (which also backs lm-studio discovery) now does the same. Behavior: - Build the reference index once outside the loop and resolve each item via resolveModelReference(). - contextWindow precedence keeps provider-reported values authoritative: item.max_model_len ?? item.context_length ?? nativeMetadata?.contextWindow ?? reference?.contextWindow ?? DISCOVERY_DEFAULT_CONTEXT_WINDOW. - maxTokens uses reference?.maxTokens when available, otherwise the api-specific discovery default, capped at contextWindow so a bundled ref for a larger sibling can never over-request output tokens. - name / reasoning / thinking / input inherit from the reference; native lm-studio metadata still wins for input modality. - Provider-specific baseUrl, headers, and local-unknown cost stay local. - OpenAI-compat flags stay conservative (supportsStore / supportsDeveloperRole / supportsReasoningEffort all false) to match the proxy sibling. Also updated two pre-existing regression tests that used deepseek-v4-pro / deepseek-r1 / DeepSeek-V4-Flash as stand-in "fictional" ids to exercise the default-fallback branch. Those model names have since been added to the bundled catalog, so the tests were renamed to vllm-lab-fork-* ids that unambiguously miss the reference index while preserving each test's original default-fallback intent. Fixes #3983
195 lines
6.7 KiB
TypeScript
195 lines
6.7 KiB
TypeScript
import { afterEach, beforeEach, describe, expect, test } from "bun:test";
|
|
import * as fs from "node:fs";
|
|
import * as os from "node:os";
|
|
import * as path from "node:path";
|
|
import type { FetchImpl } from "@oh-my-pi/pi-ai/types";
|
|
import { ModelRegistry } from "@oh-my-pi/pi-coding-agent/config/model-registry";
|
|
import { AuthStorage } from "@oh-my-pi/pi-coding-agent/session/auth-storage";
|
|
import { removeSyncWithRetries, Snowflake } from "@oh-my-pi/pi-utils";
|
|
|
|
/**
|
|
* Issue #1528: auto-discovered OpenAI-compatible models defaulted to
|
|
* `maxTokens: 8192`, which made providers (DeepSeek, etc.) drop the streaming
|
|
* connection mid-response on large `write`/`edit` tool calls and surfaced as
|
|
* Bun's opaque "socket connection was closed unexpectedly". The cap is now
|
|
* `DISCOVERY_DEFAULT_MAX_TOKENS = 32_768` (`packages/coding-agent/src/config/
|
|
* model-registry.ts`). These tests pin the externally observable default for
|
|
* every discovery branch that previously hardcoded 8192.
|
|
*/
|
|
describe("issue #1528 discovery maxTokens default", () => {
|
|
let tempDir: string;
|
|
let modelsPath: string;
|
|
let authStorage: AuthStorage;
|
|
|
|
beforeEach(async () => {
|
|
tempDir = path.join(os.tmpdir(), `pi-test-issue-1528-${Snowflake.next()}`);
|
|
fs.mkdirSync(tempDir, { recursive: true });
|
|
modelsPath = path.join(tempDir, "models.yml");
|
|
authStorage = await AuthStorage.create(path.join(tempDir, "auth.db"));
|
|
});
|
|
|
|
afterEach(() => {
|
|
authStorage.close();
|
|
if (tempDir && fs.existsSync(tempDir)) {
|
|
removeSyncWithRetries(tempDir);
|
|
}
|
|
});
|
|
|
|
test("openai-models-list discovery returns maxTokens=32768 when API advertises no output limit", async () => {
|
|
fs.writeFileSync(
|
|
modelsPath,
|
|
[
|
|
"providers:",
|
|
" deepseek-compat:",
|
|
" baseUrl: https://api.example.com/v1",
|
|
" apiKey: sk-test",
|
|
" api: openai-completions",
|
|
" auth: apiKey",
|
|
" discovery:",
|
|
" type: openai-models-list",
|
|
].join("\n"),
|
|
);
|
|
|
|
const fetchMock: FetchImpl = async input => {
|
|
const url = String(input);
|
|
if (url !== "https://api.example.com/v1/models") {
|
|
throw new Error(`Unexpected URL: ${url}`);
|
|
}
|
|
return new Response(JSON.stringify({ data: [{ id: "vllm-lab-fork-a1" }] }), {
|
|
status: 200,
|
|
headers: { "Content-Type": "application/json" },
|
|
});
|
|
};
|
|
|
|
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
|
|
await registry.refreshProvider("deepseek-compat");
|
|
|
|
const model = registry.find("deepseek-compat", "vllm-lab-fork-a1");
|
|
expect(model?.maxTokens).toBe(32_768);
|
|
});
|
|
|
|
test("proxy (anthropic+openai) discovery returns maxTokens=32768 for openai-routed models without bundled limits", async () => {
|
|
fs.writeFileSync(
|
|
modelsPath,
|
|
[
|
|
"providers:",
|
|
" newapi-proxy:",
|
|
" baseUrl: https://proxy.example.com/v1",
|
|
" apiKey: sk-test",
|
|
" api: openai-completions",
|
|
" auth: apiKey",
|
|
" discovery:",
|
|
" type: proxy",
|
|
].join("\n"),
|
|
);
|
|
|
|
const fetchMock: FetchImpl = async input => {
|
|
const url = String(input);
|
|
if (url !== "https://proxy.example.com/v1/models") {
|
|
throw new Error(`Unexpected URL: ${url}`);
|
|
}
|
|
return new Response(
|
|
JSON.stringify({
|
|
data: [{ id: "newapi-private-openai-model", supported_endpoint_types: ["openai"] }],
|
|
}),
|
|
{ status: 200, headers: { "Content-Type": "application/json" } },
|
|
);
|
|
};
|
|
|
|
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
|
|
await registry.refreshProvider("newapi-proxy");
|
|
|
|
const model = registry.find("newapi-proxy", "newapi-private-openai-model");
|
|
expect(model?.maxTokens).toBe(32_768);
|
|
});
|
|
|
|
test("proxy discovery keeps anthropic-routed models at the 8192 default to stay under Claude 3.x output caps", async () => {
|
|
// Anthropic's stream converter sends `max_tokens` as
|
|
// `(model.maxTokens / 3) | 0`. The raised 32K discovery cap would
|
|
// surface as 10,922 — above the 8,192 hard cap on classic Claude 3.x
|
|
// models — so the proxy branch keeps the conservative 8K default on
|
|
// the anthropic route.
|
|
fs.writeFileSync(
|
|
modelsPath,
|
|
[
|
|
"providers:",
|
|
" newapi-proxy:",
|
|
" baseUrl: https://proxy.example.com/v1",
|
|
" apiKey: sk-test",
|
|
" api: openai-completions",
|
|
" auth: apiKey",
|
|
" discovery:",
|
|
" type: proxy",
|
|
].join("\n"),
|
|
);
|
|
|
|
const fetchMock: FetchImpl = async input => {
|
|
const url = String(input);
|
|
if (url !== "https://proxy.example.com/v1/models") {
|
|
throw new Error(`Unexpected URL: ${url}`);
|
|
}
|
|
return new Response(
|
|
JSON.stringify({
|
|
data: [
|
|
{ id: "claude-3-5-sonnet", supported_endpoint_types: ["anthropic"] },
|
|
{ id: "claude-3-5-haiku", supported_endpoint_types: ["anthropic", "openai"] },
|
|
],
|
|
}),
|
|
{ status: 200, headers: { "Content-Type": "application/json" } },
|
|
);
|
|
};
|
|
|
|
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
|
|
await registry.refreshProvider("newapi-proxy");
|
|
|
|
const sonnet = registry.find("newapi-proxy", "claude-3-5-sonnet");
|
|
expect(sonnet?.api).toBe("anthropic-messages");
|
|
expect(sonnet?.maxTokens).toBe(8192);
|
|
|
|
// Dual-endpoint advertisements prefer the anthropic route in the proxy
|
|
// branch, so they also stay capped at 8K.
|
|
const haiku = registry.find("newapi-proxy", "claude-3-5-haiku");
|
|
expect(haiku?.api).toBe("anthropic-messages");
|
|
expect(haiku?.maxTokens).toBe(8192);
|
|
});
|
|
|
|
test("openai-models-list discovery keeps anthropic-messages providers at the 8192 default", async () => {
|
|
// The validator allows `api: anthropic-messages` with a bare
|
|
// openai-models-list discovery (e.g. third-party Anthropic catalogs
|
|
// served behind a `/v1/models` endpoint). Same divisor reasoning as
|
|
// the proxy branch applies: 32K would surface as 10,922 requested
|
|
// output tokens, above the Claude 3.x hard cap.
|
|
fs.writeFileSync(
|
|
modelsPath,
|
|
[
|
|
"providers:",
|
|
" third-party-anthropic:",
|
|
" baseUrl: https://anthropic-reseller.example.com/v1",
|
|
" apiKey: sk-test",
|
|
" api: anthropic-messages",
|
|
" auth: apiKey",
|
|
" discovery:",
|
|
" type: openai-models-list",
|
|
].join("\n"),
|
|
);
|
|
|
|
const fetchMock: FetchImpl = async input => {
|
|
const url = String(input);
|
|
if (url !== "https://anthropic-reseller.example.com/v1/models") {
|
|
throw new Error(`Unexpected URL: ${url}`);
|
|
}
|
|
return new Response(JSON.stringify({ data: [{ id: "claude-3-5-sonnet" }] }), {
|
|
status: 200,
|
|
headers: { "Content-Type": "application/json" },
|
|
});
|
|
};
|
|
|
|
const registry = new ModelRegistry(authStorage, modelsPath, { fetch: fetchMock });
|
|
await registry.refreshProvider("third-party-anthropic");
|
|
|
|
const sonnet = registry.find("third-party-anthropic", "claude-3-5-sonnet");
|
|
expect(sonnet?.api).toBe("anthropic-messages");
|
|
expect(sonnet?.maxTokens).toBe(8192);
|
|
});
|
|
});
|