Merge PR #2451: fix(coding-agent): preserve discovered context windows

This commit is contained in:
can1357
2026-06-14 17:54:50 +02:00
14 changed files with 412 additions and 27 deletions
+27
View File
@@ -641,6 +641,33 @@ providers:
type: openai-models-list
```
The built-in vLLM provider can be pointed at a non-default endpoint without declaring a custom discovery type. OMP uses vLLM's `/v1/models` metadata and preserves vLLM's `max_model_len` field as the discovered context window.
```yaml
providers:
vllm:
baseUrl: http://192.168.5.3:8085/v1
auth: none
```
For multiple vLLM endpoints, use arbitrary provider IDs with the generic OpenAI-compatible discovery path. Set `auth: none` for local no-auth servers or `apiKey` for authenticated ones. Generic discovery reads `max_model_len` first and then `context_length` as a generic OpenAI-compatible fallback.
```yaml
providers:
vllm-fast:
baseUrl: http://host-a:8000/v1
auth: none
api: openai-completions
discovery:
type: openai-models-list
vllm-long:
baseUrl: http://host-b:8000/v1
auth: none
api: openai-completions
discovery:
type: openai-models-list
```
### Hosted proxy with env-based key
```yaml
+1
View File
@@ -16,6 +16,7 @@
- Fixed OpenCode Go MiMo catalog metadata so title generation and other tool-enabled calls omit unsupported `tool_choice` instead of triggering provider 400s ([#2509](https://github.com/can1357/oh-my-pi/issues/2509)).
- Fixed OpenCode Go `kimi-k2.7-code` catalog metadata so resolve-gate requests use automatic tool selection instead of Moonshot-rejected forced `tool_choice` ([#2546](https://github.com/can1357/oh-my-pi/issues/2546)).
- Fixed Anthropic compat for the `github-copilot` host so `supportsEagerToolInputStreaming` defaults to `false` there, matching the Copilot proxy which rejects the per-tool `eager_input_streaming` field ([#2558](https://github.com/can1357/oh-my-pi/issues/2558)).
- Scoped vLLM model cache validity to the discovery base URL so changed endpoints refetch immediately, and bounded built-in vLLM discovery requests with a timeout.
## [15.12.6] - 2026-06-14
+7 -4
View File
@@ -33,6 +33,8 @@ export interface ModelManagerOptions<TApi extends Api = Api, TModelsDevPayload =
staticModels?: readonly ModelSpec<TApi>[];
/** Optional override for the cache database path. Default: <agent-dir>/models.db. */
cacheDbPath?: string;
/** Optional provider id override for cache namespacing. Defaults to providerId. */
cacheProviderId?: string;
/** Maximum cache age in milliseconds before considered stale. Default: 24h. */
cacheTtlMs?: number;
/** When true, a successful dynamic fetch is the complete provider catalog and prunes static-only models. */
@@ -107,13 +109,14 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
options: ModelManagerOptions<TApi, TModelsDevPayload>,
strategy: ModelRefreshStrategy = "online-if-uncached",
): Promise<ModelResolutionResult<TApi>> {
const cacheProviderId = options.cacheProviderId ?? options.providerId;
const now = options.now ?? Date.now;
const ttlMs = options.cacheTtlMs ?? DEFAULT_CACHE_TTL_MS;
const dbPath = options.cacheDbPath;
const staticModels = options.staticModels
? passModelList<TApi>(options.staticModels)
: (getBundledModels(options.providerId as GeneratedProvider) as Model<TApi>[]);
const cache = readModelCache<TApi>(options.providerId, ttlMs, now, dbPath);
const cache = readModelCache<TApi>(cacheProviderId, ttlMs, now, dbPath);
const dynamicModelsAuthoritative = options.dynamicModelsAuthoritative ?? false;
const staticFingerprint = fingerprintStatic(staticModels, dynamicModelsAuthoritative);
const cacheFingerprintMatches = cache?.staticFingerprint === staticFingerprint && staticFingerprint.length > 0;
@@ -160,7 +163,7 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
? retainModelIds(mergedSnapshot, dynamicModels)
: mergedSnapshot;
writeModelCache(
options.providerId,
cacheProviderId,
now(),
collapseBuiltModelVariants(snapshotModels),
true,
@@ -170,9 +173,9 @@ export async function resolveProviderModels<TApi extends Api = Api, TModelsDevPa
} else {
// Dynamic fetch failed — update cache with a non-authoritative snapshot so
// stale state remains visible while retry backoff still applies.
const latestCache = readModelCache<TApi>(options.providerId, ttlMs, now, dbPath);
const latestCache = readModelCache<TApi>(cacheProviderId, ttlMs, now, dbPath);
writeModelCache(
options.providerId,
cacheProviderId,
now(),
collapseBuiltModelVariants(
mergeDynamicModels(
@@ -2375,6 +2375,8 @@ export function litellmModelManagerOptions(
// 22. vLLM
// ---------------------------------------------------------------------------
const VLLM_DISCOVERY_TIMEOUT_MS = 10_000;
export interface VllmModelManagerConfig {
apiKey?: string;
baseUrl?: string;
@@ -2387,6 +2389,7 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
const references = createBundledReferenceMap<"openai-completions">("vllm" as Parameters<typeof getBundledModels>[0]);
return {
providerId: "vllm",
cacheProviderId: `vllm:${Bun.hash(baseUrl).toString(36)}`,
fetchDynamicModels: () =>
fetchOpenAICompatibleModels({
api: "openai-completions",
@@ -2401,6 +2404,7 @@ export function vllmModelManagerOptions(config?: VllmModelManagerConfig): ModelM
};
},
fetch: config?.fetch,
signal: AbortSignal.timeout(VLLM_DISCOVERY_TIMEOUT_MS),
}),
};
}
+1
View File
@@ -84,6 +84,7 @@
- Fixed `~/.agent[s]/skills` not appearing as `/skill:<name>` commands when every named source toggle (`skills.enableCodexUser`, `skills.enableClaudeUser`, `skills.enableClaudeProject`, `skills.enablePiUser`, `skills.enablePiProject`) was off: `loadSkills` gated the `agents` provider on `anyBuiltInSkillSourceEnabled`, so a user who turned off the Claude/Codex/Pi sources to clean noise also lost their own canonical OMP-native skills. The `agents` provider now reads the dedicated `enableAgentsUser`/`enableAgentsProject` toggles, and the unknown-third-party fall-through gate is restricted to the named third-party toggles so keeping the default agents toggles on no longer silently re-enables `opencode`/`github`/`claude-plugins`/`gemini` skill sources ([#2401](https://github.com/can1357/oh-my-pi/issues/2401)).
- Fixed Claude Code marketplace plugin skills installed under `skills/<name>/SKILL.md` to also appear as bare slash commands such as `/understand`, matching Claude-native plugin docs. The slash command name is taken from the skill directory basename so display-style frontmatter names like `name: Understand Anything` still resolve to `/understand` ([#2415](https://github.com/can1357/oh-my-pi/issues/2415)).
- Fixed ACP `/move` builtin test expectations to compare the resolved destination path so the test is portable on Windows and Unix ([#2381](https://github.com/can1357/oh-my-pi/pull/2381) by [@oldschoola](https://github.com/oldschoola)).
- Fixed vLLM discovery so `providers.vllm.baseUrl` drives the built-in endpoint, additional OpenAI-compatible vLLM provider IDs work through `openai-models-list`, and discovered `max_model_len` or fallback `context_length` values set context windows instead of falling back to 128k.
### Removed
@@ -393,12 +393,16 @@ export async function discoverOpenAIModelsList(
const response = apiKey
? await withAuth(apiKey, key => attempt({ ...baseHeaders, Authorization: `Bearer ${key}` }))
: await attempt(baseHeaders);
const payload = (await response.json()) as { data?: Array<{ id: string; context_length?: number }> };
const payload = (await response.json()) as {
data?: Array<{ id?: string; max_model_len?: unknown; context_length?: unknown }>;
};
const models = payload.data ?? [];
const discovered: Model<Api>[] = [];
for (const item of models) {
const id = item.id;
if (!id) continue;
const contextWindow =
toPositiveNumberOrUndefined(item.max_model_len) ?? toPositiveNumberOrUndefined(item.context_length) ?? 128000;
discovered.push(
buildModel({
id,
@@ -409,8 +413,8 @@ export async function discoverOpenAIModelsList(
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
contextWindow: toPositiveNumberOrUndefined(item.context_length) ?? 128000,
maxTokens: discoveryDefaultMaxTokens(providerConfig.api),
contextWindow,
maxTokens: Math.min(contextWindow, discoveryDefaultMaxTokens(providerConfig.api)),
headers,
compat: {
supportsStore: false,
@@ -24,9 +24,11 @@ import {
resolveVariantAlias,
} from "@oh-my-pi/pi-catalog/variant-collapse";
// Sentinel for local-only OAuth token (LM Studio, vLLM) — declared inline to avoid loading
// any provider module at startup. Must match `DEFAULT_LOCAL_TOKEN` in oauth/lm-studio.ts.
// Sentinels for local-only OAuth tokens — declared inline to avoid loading
// provider modules at startup. Must match packages/ai/src/registry/lm-studio.ts
// and packages/ai/src/registry/vllm.ts.
const DEFAULT_LOCAL_TOKEN = "lm-studio-local";
const DEFAULT_VLLM_LOCAL_TOKEN = "vllm-local";
const SPECIAL_MODEL_MANAGER_PROVIDER_IDS: readonly string[] = [
"google-antigravity",
@@ -82,6 +84,10 @@ export function isAuthenticated(apiKey: string | undefined | null): apiKey is st
return Boolean(apiKey) && apiKey !== kNoAuth;
}
function isDiscoveryBearerApiKey(apiKey: string | undefined | null): apiKey is string {
return isAuthenticated(apiKey) && apiKey !== DEFAULT_LOCAL_TOKEN && apiKey !== DEFAULT_VLLM_LOCAL_TOKEN;
}
/** Provider override config (baseUrl, headers, apiKey, compat, transport) without custom models */
interface ProviderOverride {
baseUrl?: string;
@@ -900,6 +906,18 @@ export class ModelRegistry {
});
}
#resolveStartupModelCacheProviderId(providerId: string): string {
const descriptor = PROVIDER_DESCRIPTORS.find(candidate => candidate.providerId === providerId);
if (!descriptor) {
return providerId;
}
const baseUrl =
this.#runtimeProviderOverrides.get(providerId)?.baseUrl ??
this.#providerOverrides.get(providerId)?.baseUrl ??
this.getProviderBaseUrl(providerId);
return descriptor.createModelManagerOptions({ baseUrl, fetch: this.#fetch }).cacheProviderId ?? providerId;
}
#loadCachedStandardProviderModels(): { models: Model<Api>[]; authoritativeFreshProviders: Set<string> } {
const configuredDiscoveryProviders = new Set(this.#discoverableProviders.map(provider => provider.provider));
const cachedModels: Model<Api>[] = [];
@@ -908,7 +926,8 @@ export class ModelRegistry {
if (configuredDiscoveryProviders.has(providerId)) {
continue;
}
const cache = readModelCache<Api>(providerId, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const cacheProviderId = this.#resolveStartupModelCacheProviderId(providerId);
const cache = readModelCache<Api>(cacheProviderId, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
if (!cache) {
continue;
}
@@ -938,7 +957,12 @@ export class ModelRegistry {
#loadCachedDiscoverableModels(): Model<Api>[] {
const cachedModels: Model<Api>[] = [];
for (const providerConfig of this.#discoverableProviders) {
const cache = readModelCache<Api>(providerConfig.provider, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const cache = readModelCache<Api>(
this.#configuredDiscoveryCacheProviderId(providerConfig),
24 * 60 * 60 * 1000,
Date.now,
this.#cacheDbPath,
);
if (!cache) {
this.#providerDiscoveryStates.set(providerConfig.provider, {
provider: providerConfig.provider,
@@ -1200,11 +1224,19 @@ export class ModelRegistry {
this.#rebuildCanonicalIndex();
}
#configuredDiscoveryCacheProviderId(providerConfig: DiscoveryProviderConfig): string {
if (providerConfig.discovery.type === "openai-models-list") {
return `${providerConfig.provider}:openai-models-list-context-v2`;
}
return providerConfig.provider;
}
async #discoverProviderModels(
providerConfig: DiscoveryProviderConfig,
strategy: ModelRefreshStrategy,
): Promise<Model<Api>[]> {
const cached = readModelCache<Api>(providerConfig.provider, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const cacheProviderId = this.#configuredDiscoveryCacheProviderId(providerConfig);
const cached = readModelCache<Api>(cacheProviderId, 24 * 60 * 60 * 1000, Date.now, this.#cacheDbPath);
const requiresAuth = !this.#keylessProviders.has(providerConfig.provider);
if (requiresAuth) {
const apiKey = await this.#peekApiKeyForProvider(providerConfig.provider);
@@ -1242,6 +1274,7 @@ export class ModelRegistry {
providerId,
staticModels: [],
cacheDbPath: this.#cacheDbPath,
cacheProviderId,
cacheTtlMs: 24 * 60 * 60 * 1000,
fetchDynamicModels,
});
@@ -1283,7 +1316,9 @@ export class ModelRegistry {
fetch: this.#fetch,
getBearerApiKeyResolver: async provider => {
const apiKey = await this.getApiKeyForProvider(provider);
if (!apiKey || apiKey === DEFAULT_LOCAL_TOKEN || apiKey === kNoAuth) return undefined;
if (!isDiscoveryBearerApiKey(apiKey)) {
return undefined;
}
return this.resolver(provider);
},
};
@@ -1388,11 +1423,20 @@ export class ModelRegistry {
for (let i = 0; i < standardProviderDescriptors.length; i++) {
const descriptor = standardProviderDescriptors[i];
const apiKey = standardProviderKeys[i];
if (isAuthenticated(apiKey) || descriptor.allowUnauthenticated) {
const hasExplicitVllmConfig =
descriptor.providerId === "vllm" &&
(this.#runtimeProviderOverrides.has(descriptor.providerId) ||
this.#providerOverrides.has(descriptor.providerId) ||
this.#keylessProviders.has(descriptor.providerId));
if (isAuthenticated(apiKey) || descriptor.allowUnauthenticated || hasExplicitVllmConfig) {
const discoveryBaseUrl =
this.#runtimeProviderOverrides.get(descriptor.providerId)?.baseUrl ??
this.#providerOverrides.get(descriptor.providerId)?.baseUrl ??
this.getProviderBaseUrl(descriptor.providerId);
options.push(
descriptor.createModelManagerOptions({
apiKey: isAuthenticated(apiKey) ? apiKey : undefined,
baseUrl: this.getProviderBaseUrl(descriptor.providerId),
apiKey: isDiscoveryBearerApiKey(apiKey) ? apiKey : undefined,
baseUrl: discoveryBaseUrl,
fetch: this.#fetch,
}),
);
@@ -50,12 +50,13 @@ export function validateProviderConfiguration(
!config.headers &&
!config.compat &&
!config.apiKey &&
config.auth !== "none" &&
!config.disableStrictTools &&
!hasModelOverrides &&
!config.discovery
) {
throw new Error(
`Provider ${providerName}: must specify "baseUrl", "headers", "apiKey", "compat", "disableStrictTools", "modelOverrides", "discovery", or "models"`,
`Provider ${providerName}: must specify "baseUrl", "headers", "apiKey", "auth: none", "compat", "disableStrictTools", "modelOverrides", "discovery", or "models"`,
);
}
}
@@ -3,6 +3,8 @@ import * as fs from "node:fs";
import * as os from "node:os";
import * as path from "node:path";
import { stripVTControlCharacters } from "node:util";
import { buildModel } from "@oh-my-pi/pi-catalog/build";
import { writeModelCache } from "@oh-my-pi/pi-catalog/model-cache";
import type { ModelRegistry, ProviderDiscoveryState } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { ModelRegistry as ModelRegistryImpl } from "@oh-my-pi/pi-coding-agent/config/model-registry";
import { Settings } from "@oh-my-pi/pi-coding-agent/config/settings";
@@ -171,4 +173,301 @@ describe("issue #970 custom provider discovery", () => {
expect(rendered).toContain("http://192.168.5.3:8085/v1/models returned 404");
expect(rendered).toContain("baseUrl");
});
test("discovers multiple configurable vllm instances and preserves advertised context metadata", async () => {
fs.writeFileSync(
modelsPath,
[
"providers:",
" vllm-fast:",
" baseUrl: http://192.168.5.3:8085/v1",
" auth: none",
" api: openai-completions",
" discovery:",
" type: openai-models-list",
" vllm-long:",
" baseUrl: http://192.168.5.4:8085/v1",
" auth: none",
" api: openai-completions",
" discovery:",
" type: openai-models-list",
].join("\n"),
);
const fetchMock: (input: string | URL | Request) => Promise<Response> = async input => {
const url = String(input);
if (url === "http://192.168.5.3:8085/v1/models") {
return new Response(JSON.stringify({ data: [{ id: "DeepSeek-V4-Flash", max_model_len: 262_144 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
if (url === "http://192.168.5.4:8085/v1/models") {
return new Response(JSON.stringify({ data: [{ id: "DeepSeek-V4-Long", context_length: "1048576" }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
throw new Error(`Unexpected URL: ${url}`);
};
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("vllm-fast");
await registry.refreshProvider("vllm-long");
const fast = registry.find("vllm-fast", "DeepSeek-V4-Flash");
expect(fast?.contextWindow).toBe(262_144);
expect(fast?.maxTokens).toBe(32_768);
const long = registry.find("vllm-long", "DeepSeek-V4-Long");
expect(long?.contextWindow).toBe(1_048_576);
expect(long?.maxTokens).toBe(32_768);
expect(registry.getProviderDiscoveryState("vllm-fast")?.status).toBe("ok");
expect(registry.getProviderDiscoveryState("vllm-long")?.status).toBe("ok");
});
test("ignores old configured openai-models-list cache namespaces after adding vllm context parsing", async () => {
fs.writeFileSync(
modelsPath,
[
"providers:",
" vllm-fast:",
" baseUrl: http://192.168.5.3:8085/v1",
" auth: none",
" api: openai-completions",
" discovery:",
" type: openai-models-list",
].join("\n"),
);
writeModelCache(
"vllm-fast",
Date.now(),
[
buildModel({
id: "Stale",
name: "Stale",
provider: "vllm-fast",
api: "openai-completions",
baseUrl: "http://192.168.5.3:8085/v1",
contextWindow: 128_000,
maxTokens: 32_768,
reasoning: false,
input: ["text"],
cost: {
input: 0,
output: 0,
cacheRead: 0,
cacheWrite: 0,
},
}),
],
true,
"",
path.join(tempDir, "models.db"),
);
const calls: string[] = [];
const fetchMock: (input: string | URL | Request) => Promise<Response> = async input => {
const url = String(input);
calls.push(url);
if (url !== "http://192.168.5.3:8085/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
return new Response(JSON.stringify({ data: [{ id: "Fresh", max_model_len: 262_144 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("vllm-fast", "online-if-uncached");
expect(calls).toEqual(["http://192.168.5.3:8085/v1/models"]);
expect(registry.find("vllm-fast", "Fresh")?.contextWindow).toBe(262_144);
expect(registry.find("vllm-fast", "Stale")).toBeUndefined();
});
test("uses default vllm baseUrl override for built-in discovery", async () => {
fs.writeFileSync(
modelsPath,
["providers:", " vllm:", " baseUrl: http://192.168.5.3:8085/v1", " auth: none"].join("\n"),
);
await authStorage.set("vllm", { type: "api_key", key: "vllm-local" });
const fetchMock: (input: string | URL | Request, init?: RequestInit) => Promise<Response> = async (
input,
init,
) => {
const url = String(input);
if (url !== "http://192.168.5.3:8085/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
const headers = init?.headers as Headers | Record<string, string> | undefined;
const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization;
expect(authHeader).toBeUndefined();
expect(init?.signal).toBeInstanceOf(AbortSignal);
return new Response(JSON.stringify({ data: [{ id: "DeepSeek-V4-Flash", max_model_len: 262_144 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("vllm");
const model = registry.find("vllm", "DeepSeek-V4-Flash");
expect(model?.baseUrl).toBe("http://192.168.5.3:8085/v1");
expect(model?.contextWindow).toBe(262_144);
expect(model?.provider).toBe("vllm");
});
test("does not probe built-in vllm unless it is explicitly configured", async () => {
fs.writeFileSync(modelsPath, ["providers: {}"].join("\n"));
const urls: string[] = [];
const fetchMock: (input: string | URL | Request) => Promise<Response> = async input => {
const url = String(input);
urls.push(url);
if (url === "http://127.0.0.1:8000/v1/models") {
throw new Error("Unexpected default vLLM probe");
}
return new Response(JSON.stringify({ data: [] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await registry.refresh();
expect(urls).not.toContain("http://127.0.0.1:8000/v1/models");
});
test("treats auth none only vllm config as explicit built-in discovery", async () => {
fs.writeFileSync(modelsPath, ["providers:", " vllm:", " auth: none"].join("\n"));
const fetchMock: (input: string | URL | Request, init?: RequestInit) => Promise<Response> = async (
input,
init,
) => {
const url = String(input);
if (url !== "http://127.0.0.1:8000/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
expect(init?.signal).toBeInstanceOf(AbortSignal);
return new Response(JSON.stringify({ data: [{ id: "DefaultVllm", max_model_len: 262_144 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("vllm");
expect(registry.find("vllm", "DefaultVllm")?.contextWindow).toBe(262_144);
});
test("refetches built-in vllm discovery when the configured baseUrl changes", async () => {
fs.writeFileSync(
modelsPath,
["providers:", " vllm:", " baseUrl: http://192.168.5.3:8085/v1", " auth: none"].join("\n"),
);
const calls: string[] = [];
const fetchMock: (input: string | URL | Request) => Promise<Response> = async input => {
const url = String(input);
if (url === "http://192.168.5.3:8085/v1/models") {
calls.push(url);
return new Response(JSON.stringify({ data: [{ id: "Old", max_model_len: 262_144 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
if (url === "http://192.168.5.4:8085/v1/models") {
calls.push(url);
return new Response(JSON.stringify({ data: [{ id: "New", max_model_len: 524_288 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
}
throw new Error(`Unexpected URL: ${url}`);
};
const firstRegistry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await firstRegistry.refreshProvider("vllm");
expect(firstRegistry.find("vllm", "Old")?.contextWindow).toBe(262_144);
fs.writeFileSync(
modelsPath,
["providers:", " vllm:", " baseUrl: http://192.168.5.4:8085/v1", " auth: none"].join("\n"),
);
const secondRegistry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await secondRegistry.refresh();
expect(secondRegistry.find("vllm", "New")?.contextWindow).toBe(524_288);
expect(calls).toEqual(["http://192.168.5.3:8085/v1/models", "http://192.168.5.4:8085/v1/models"]);
});
test("loads built-in vllm cache from the configured baseUrl namespace", async () => {
fs.writeFileSync(
modelsPath,
["providers:", " vllm:", " baseUrl: http://192.168.5.3:8085/v1", " auth: none"].join("\n"),
);
const fetchMock: (input: string | URL | Request) => Promise<Response> = async input => {
const url = String(input);
if (url !== "http://192.168.5.3:8085/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
return new Response(JSON.stringify({ data: [{ id: "Cached", max_model_len: 262_144 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const firstRegistry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await firstRegistry.refreshProvider("vllm");
expect(firstRegistry.find("vllm", "Cached")?.contextWindow).toBe(262_144);
const cachedRegistry = new ModelRegistryImpl(authStorage, modelsPath, {
fetch: async input => {
throw new Error(`Unexpected online fetch: ${String(input)}`);
},
});
expect(cachedRegistry.find("vllm", "Cached")?.contextWindow).toBe(262_144);
});
test("does not send vllm-local placeholder as discovery bearer", async () => {
fs.writeFileSync(
modelsPath,
[
"providers:",
" vllm:",
" baseUrl: http://192.168.5.3:8085/v1",
" apiKey: vllm-local",
" api: openai-completions",
" discovery:",
" type: openai-models-list",
].join("\n"),
);
const fetchMock: (input: string | URL | Request, init?: RequestInit) => Promise<Response> = async (
input,
init,
) => {
const url = String(input);
if (url !== "http://192.168.5.3:8085/v1/models") {
throw new Error(`Unexpected URL: ${url}`);
}
const headers = init?.headers as Headers | Record<string, string> | undefined;
const authHeader = headers instanceof Headers ? headers.get("Authorization") : headers?.Authorization;
expect(authHeader).toBeUndefined();
return new Response(JSON.stringify({ data: [{ id: "DeepSeek-V4-Flash", max_model_len: 262_144 }] }), {
status: 200,
headers: { "Content-Type": "application/json" },
});
};
const registry = new ModelRegistryImpl(authStorage, modelsPath, { fetch: fetchMock });
await registry.refreshProvider("vllm");
expect(registry.getProviderDiscoveryState("vllm")?.status).toBe("ok");
});
});
@@ -14,7 +14,7 @@ beforeAll(async () => {
resetSettingsForTest();
await Settings.init({ inMemory: true, cwd: process.cwd() });
await initTheme(false, undefined, undefined, "dark", "light");
});
}, 15_000);
async function getRequiredTheme() {
const uiTheme = await getThemeByName("dark");
@@ -118,7 +118,7 @@ describe("MCP tool rendering", () => {
expect(makeDeferredTool().mergeCallAndResult).toBe(true);
expect(rendered).toContain(`${doneIcon} sentry/search_events`);
expect(rendered).not.toContain(`${pendingIcon} sentry/search_events`);
});
}, 15_000);
it("replaces the pending call header with an error header for MCP errors", async () => {
const uiTheme = await getRequiredTheme();
@@ -129,5 +129,5 @@ describe("MCP tool rendering", () => {
expect(rendered).toContain(`${errorIcon} sentry/search_events`);
expect(rendered).not.toContain(`${pendingIcon} sentry/search_events`);
});
}, 15_000);
});
@@ -106,7 +106,7 @@ describe("createAgentSession deferred MCP auto discovery", () => {
// fire-and-forget with no completion promise or event exposed — fake
// timers cannot drive a child process, so poll the live session with
// a generous ceiling, exiting the instant discovery flips on.
const deadline = Date.now() + 12_000;
const deadline = Date.now() + 30_000;
while (!session.isMCPDiscoveryEnabled() && Date.now() < deadline) {
await Bun.sleep(50);
}
@@ -121,7 +121,7 @@ describe("createAgentSession deferred MCP auto discovery", () => {
} finally {
await session.dispose();
}
}, 20_000);
}, 40_000);
it("disposing mid-connect disconnects the manager and never resurrects tools", async () => {
// Stall `initialize` in the real fixture subprocess so the connect is
@@ -142,7 +142,7 @@ describe("createAgentSession deferred MCP auto discovery", () => {
// Genuine integration wait (see above): the deferred task notices the
// disposed session once the stalled connect resolves and must disconnect
// instead of refreshing tools. Exits the instant the spy fires.
const deadline = Date.now() + 12_000;
const deadline = Date.now() + 30_000;
while (disconnectSpy.mock.calls.length === 0 && Date.now() < deadline) {
await Bun.sleep(50);
}
@@ -150,5 +150,5 @@ describe("createAgentSession deferred MCP auto discovery", () => {
expect(session.getActiveToolNames().filter(name => name.startsWith("mcp__"))).toEqual([]);
expect(session.getActiveToolNames()).not.toContain("search_tool_bm25");
expect(session.isMCPDiscoveryEnabled()).toBe(false);
}, 20_000);
}, 40_000);
});
@@ -271,6 +271,7 @@ describe("createAgentSession deferred model pattern resolution", () => {
slashCommands: [],
enableMCP: false,
enableLsp: false,
skipPythonPreflight: true,
});
try {
@@ -196,5 +196,5 @@ describe("SessionManager signature persistence", () => {
expect(await fs.readFile(sessionFile, "utf8")).toBe(persistedBefore);
expect((await fs.stat(sessionFile)).mtimeMs).toBe(initialMtimeMs);
await reloaded.close();
});
}, 15_000);
});
+2 -2
View File
@@ -900,7 +900,7 @@ describe("github tool", () => {
await tempHome.cleanup();
await fs.rm(fixture.baseDir, { recursive: true, force: true });
}
});
}, 30_000);
it("rejects PR pushes from branches without checkout metadata", async () => {
const fixture = await createPrFixture();
@@ -930,7 +930,7 @@ describe("github tool", () => {
} finally {
await fs.rm(fixture.baseDir, { recursive: true, force: true });
}
});
}, 30_000);
it("exposes a flat op-based schema without legacy run_watch parameters", () => {
const tool = new GithubTool(createSession());